@observertc/observer-js 1.0.0-beta.9 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1174 -166
- package/dist/index.d.mts +4012 -208
- package/dist/index.d.ts +4012 -208
- package/dist/index.js +4149 -601
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +4091 -600
- package/dist/index.mjs.map +1 -1
- package/package.json +15 -4
package/dist/index.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { EventEmitter } from 'events';
|
|
2
2
|
import { types } from 'mediasoup';
|
|
3
3
|
|
|
4
|
+
declare const schemaVersion = "3.7.0";
|
|
4
5
|
/**
|
|
5
6
|
* The WebRTC app provided custom stats payload
|
|
6
7
|
*/
|
|
@@ -12,7 +13,7 @@ type ExtensionStat = {
|
|
|
12
13
|
/**
|
|
13
14
|
* The payload of the extension stats the custom app provides
|
|
14
15
|
*/
|
|
15
|
-
payload?: string
|
|
16
|
+
payload?: Record<string, unknown>;
|
|
16
17
|
};
|
|
17
18
|
/**
|
|
18
19
|
* A list of additional client events.
|
|
@@ -23,9 +24,9 @@ type ClientMetaData = {
|
|
|
23
24
|
*/
|
|
24
25
|
type: string;
|
|
25
26
|
/**
|
|
26
|
-
* The
|
|
27
|
+
* The attributes of the meta data entry, if applicable.
|
|
27
28
|
*/
|
|
28
|
-
payload?: string
|
|
29
|
+
payload?: Record<string, unknown>;
|
|
29
30
|
/**
|
|
30
31
|
* The unique identifier of the peer connection for which the event was generated.
|
|
31
32
|
*/
|
|
@@ -52,9 +53,13 @@ type ClientIssue = {
|
|
|
52
53
|
*/
|
|
53
54
|
type: string;
|
|
54
55
|
/**
|
|
55
|
-
*
|
|
56
|
+
* Identifier of the related issue or resolution when it is provided.
|
|
56
57
|
*/
|
|
57
|
-
|
|
58
|
+
key?: string;
|
|
59
|
+
/**
|
|
60
|
+
* The attributes of the issue, if applicable.
|
|
61
|
+
*/
|
|
62
|
+
payload?: Record<string, unknown>;
|
|
58
63
|
/**
|
|
59
64
|
* The timestamp in epoch format when the event was generated.
|
|
60
65
|
*/
|
|
@@ -69,16 +74,16 @@ type ClientEvent = {
|
|
|
69
74
|
*/
|
|
70
75
|
type: string;
|
|
71
76
|
/**
|
|
72
|
-
* The
|
|
77
|
+
* The attributes of the event, if applicable.
|
|
73
78
|
*/
|
|
74
|
-
payload?: string
|
|
79
|
+
payload?: Record<string, unknown>;
|
|
75
80
|
/**
|
|
76
81
|
* The timestamp in epoch format when the event was generated.
|
|
77
82
|
*/
|
|
78
83
|
timestamp?: number;
|
|
79
84
|
};
|
|
80
85
|
/**
|
|
81
|
-
*
|
|
86
|
+
* Certificate Stats
|
|
82
87
|
*/
|
|
83
88
|
type CertificateStats = {
|
|
84
89
|
/**
|
|
@@ -102,7 +107,7 @@ type CertificateStats = {
|
|
|
102
107
|
*/
|
|
103
108
|
base64Certificate?: string;
|
|
104
109
|
/**
|
|
105
|
-
* The certificate ID of the issuer
|
|
110
|
+
* The certificate ID of the issuer.
|
|
106
111
|
*/
|
|
107
112
|
issuerCertificateId?: string;
|
|
108
113
|
/**
|
|
@@ -119,7 +124,7 @@ type IceCandidatePairStats = {
|
|
|
119
124
|
*/
|
|
120
125
|
id: string;
|
|
121
126
|
/**
|
|
122
|
-
* The timestamp of when the stats were recorded, in
|
|
127
|
+
* The timestamp of when the stats were recorded, in milliseconds.
|
|
123
128
|
*/
|
|
124
129
|
timestamp: number;
|
|
125
130
|
/**
|
|
@@ -134,7 +139,10 @@ type IceCandidatePairStats = {
|
|
|
134
139
|
* The ID of the remote ICE candidate in this pair.
|
|
135
140
|
*/
|
|
136
141
|
remoteCandidateId?: string;
|
|
137
|
-
|
|
142
|
+
/**
|
|
143
|
+
* The checklist state of this candidate pair. Values follow the W3C RTCStatsIceCandidatePairState enum (frozen, waiting, in-progress, failed, succeeded). Two further values are accepted for backward compatibility and are not part of the current spec: `new` (never standardised) and `cancelled` (removed from the spec after 2016).
|
|
144
|
+
*/
|
|
145
|
+
state?: "new" | "frozen" | "in-progress" | "waiting" | "failed" | "succeeded" | "cancelled" | "inprogress";
|
|
138
146
|
/**
|
|
139
147
|
* Whether this candidate pair has been nominated.
|
|
140
148
|
*/
|
|
@@ -229,7 +237,7 @@ type IceCandidateStats = {
|
|
|
229
237
|
*/
|
|
230
238
|
transportId?: string;
|
|
231
239
|
/**
|
|
232
|
-
* The IP address of the ICE candidate
|
|
240
|
+
* The IP address of the ICE candidate.
|
|
233
241
|
*/
|
|
234
242
|
address?: string;
|
|
235
243
|
/**
|
|
@@ -358,7 +366,15 @@ type IceTransportStats = {
|
|
|
358
366
|
*/
|
|
359
367
|
selectedCandidatePairChanges?: number;
|
|
360
368
|
/**
|
|
361
|
-
*
|
|
369
|
+
* Number of congestion control feedback (CCFB) messages sent on this transport.
|
|
370
|
+
*/
|
|
371
|
+
ccfbMessagesSent?: number;
|
|
372
|
+
/**
|
|
373
|
+
* Number of congestion control feedback (CCFB) messages received on this transport.
|
|
374
|
+
*/
|
|
375
|
+
ccfbMessagesReceived?: number;
|
|
376
|
+
/**
|
|
377
|
+
* Additional information attached to this stats.
|
|
362
378
|
*/
|
|
363
379
|
attachments?: Record<string, unknown>;
|
|
364
380
|
};
|
|
@@ -539,7 +555,7 @@ type MediaSourceStats = {
|
|
|
539
555
|
attachments?: Record<string, unknown>;
|
|
540
556
|
};
|
|
541
557
|
/**
|
|
542
|
-
* Remote Outbound
|
|
558
|
+
* Remote Outbound RTP Stats
|
|
543
559
|
*/
|
|
544
560
|
type RemoteOutboundRtpStats = {
|
|
545
561
|
/**
|
|
@@ -625,7 +641,24 @@ type QualityLimitationDurations = {
|
|
|
625
641
|
other: number;
|
|
626
642
|
};
|
|
627
643
|
/**
|
|
628
|
-
*
|
|
644
|
+
* Cumulative PSNR measurements for Y, U, V components.
|
|
645
|
+
*/
|
|
646
|
+
type PsnrSum = {
|
|
647
|
+
/**
|
|
648
|
+
* PSNR value for the Y (luminance) component.
|
|
649
|
+
*/
|
|
650
|
+
y: number;
|
|
651
|
+
/**
|
|
652
|
+
* PSNR value for the U (chrominance) component.
|
|
653
|
+
*/
|
|
654
|
+
u: number;
|
|
655
|
+
/**
|
|
656
|
+
* PSNR value for the V (chrominance) component.
|
|
657
|
+
*/
|
|
658
|
+
v: number;
|
|
659
|
+
};
|
|
660
|
+
/**
|
|
661
|
+
* Outbound RTP Stats
|
|
629
662
|
*/
|
|
630
663
|
type OutboundRtpStats = {
|
|
631
664
|
/**
|
|
@@ -677,6 +710,10 @@ type OutboundRtpStats = {
|
|
|
677
710
|
*/
|
|
678
711
|
rid?: string;
|
|
679
712
|
/**
|
|
713
|
+
* Index of the encoding in the encoding array.
|
|
714
|
+
*/
|
|
715
|
+
encodingIndex?: number;
|
|
716
|
+
/**
|
|
680
717
|
* The total number of header bytes sent on this stream.
|
|
681
718
|
*/
|
|
682
719
|
headerBytesSent?: number;
|
|
@@ -733,6 +770,14 @@ type OutboundRtpStats = {
|
|
|
733
770
|
*/
|
|
734
771
|
qpSum?: number;
|
|
735
772
|
/**
|
|
773
|
+
* Cumulative PSNR measurements for Y, U, V components.
|
|
774
|
+
*/
|
|
775
|
+
psnrSum?: PsnrSum;
|
|
776
|
+
/**
|
|
777
|
+
* Total number of PSNR measurements collected.
|
|
778
|
+
*/
|
|
779
|
+
psnrMeasurements?: number;
|
|
780
|
+
/**
|
|
736
781
|
* The total time spent encoding frames on this stream in seconds.
|
|
737
782
|
*/
|
|
738
783
|
totalEncodeTime?: number;
|
|
@@ -741,10 +786,14 @@ type OutboundRtpStats = {
|
|
|
741
786
|
*/
|
|
742
787
|
totalPacketSendDelay?: number;
|
|
743
788
|
/**
|
|
744
|
-
* The reason for any quality limitation on this stream.
|
|
789
|
+
* The reason for any quality limitation on this stream (e.g., 'cpu', 'bandwidth', 'other').
|
|
745
790
|
*/
|
|
746
791
|
qualityLimitationReason?: string;
|
|
747
792
|
/**
|
|
793
|
+
* The duration of quality limitation reasons categorized by type.
|
|
794
|
+
*/
|
|
795
|
+
qualityLimitationDurations?: QualityLimitationDurations;
|
|
796
|
+
/**
|
|
748
797
|
* The number of resolution changes due to quality limitations.
|
|
749
798
|
*/
|
|
750
799
|
qualityLimitationResolutionChanges?: number;
|
|
@@ -765,7 +814,7 @@ type OutboundRtpStats = {
|
|
|
765
814
|
*/
|
|
766
815
|
encoderImplementation?: string;
|
|
767
816
|
/**
|
|
768
|
-
* Indicates whether the encoder is power
|
|
817
|
+
* Indicates whether the encoder is power-efficient.
|
|
769
818
|
*/
|
|
770
819
|
powerEfficientEncoder?: boolean;
|
|
771
820
|
/**
|
|
@@ -777,16 +826,16 @@ type OutboundRtpStats = {
|
|
|
777
826
|
*/
|
|
778
827
|
scalabilityMode?: string;
|
|
779
828
|
/**
|
|
780
|
-
*
|
|
829
|
+
* Number of packets sent with ECT(1) congestion marking.
|
|
781
830
|
*/
|
|
782
|
-
|
|
831
|
+
packetsSentWithEct1?: number;
|
|
783
832
|
/**
|
|
784
833
|
* Additional information attached to this stats.
|
|
785
834
|
*/
|
|
786
835
|
attachments?: Record<string, unknown>;
|
|
787
836
|
};
|
|
788
837
|
/**
|
|
789
|
-
* Remote Inbound
|
|
838
|
+
* Remote Inbound RTP Stats
|
|
790
839
|
*/
|
|
791
840
|
type RemoteInboundRtpStats = {
|
|
792
841
|
/**
|
|
@@ -818,6 +867,22 @@ type RemoteInboundRtpStats = {
|
|
|
818
867
|
*/
|
|
819
868
|
packetsReceived?: number;
|
|
820
869
|
/**
|
|
870
|
+
* Total number of RTP packets received for this SSRC marked with the ECT(1) marking.
|
|
871
|
+
*/
|
|
872
|
+
packetsReceivedWithEct1?: number;
|
|
873
|
+
/**
|
|
874
|
+
* Total number of RTP packets received for this SSRC marked with the CE marking.
|
|
875
|
+
*/
|
|
876
|
+
packetsReceivedWithCe?: number;
|
|
877
|
+
/**
|
|
878
|
+
* Total number of RTP packets for which an RFC8888 report has been sent with a zero R bit.
|
|
879
|
+
*/
|
|
880
|
+
packetsReportedAsLost?: number;
|
|
881
|
+
/**
|
|
882
|
+
* Total number of RTP packets reported as lost but later recovered in a subsequent RFC8888 report.
|
|
883
|
+
*/
|
|
884
|
+
packetsReportedAsLostButRecovered?: number;
|
|
885
|
+
/**
|
|
821
886
|
* The total number of packets lost on this stream.
|
|
822
887
|
*/
|
|
823
888
|
packetsLost?: number;
|
|
@@ -846,12 +911,16 @@ type RemoteInboundRtpStats = {
|
|
|
846
911
|
*/
|
|
847
912
|
roundTripTimeMeasurements?: number;
|
|
848
913
|
/**
|
|
914
|
+
* Number of packets with ECT(1) marking that were bleached by a middlebox.
|
|
915
|
+
*/
|
|
916
|
+
packetsWithBleachedEct1Marking?: number;
|
|
917
|
+
/**
|
|
849
918
|
* Additional information attached to this stats
|
|
850
919
|
*/
|
|
851
920
|
attachments?: Record<string, unknown>;
|
|
852
921
|
};
|
|
853
922
|
/**
|
|
854
|
-
* Inbound
|
|
923
|
+
* Inbound RTP Stats
|
|
855
924
|
*/
|
|
856
925
|
type InboundRtpStats = {
|
|
857
926
|
/**
|
|
@@ -887,6 +956,22 @@ type InboundRtpStats = {
|
|
|
887
956
|
*/
|
|
888
957
|
packetsReceived?: number;
|
|
889
958
|
/**
|
|
959
|
+
* Total number of RTP packets received for this SSRC marked with the ECT(1) marking.
|
|
960
|
+
*/
|
|
961
|
+
packetsReceivedWithEct1?: number;
|
|
962
|
+
/**
|
|
963
|
+
* Total number of RTP packets received for this SSRC marked with the CE marking.
|
|
964
|
+
*/
|
|
965
|
+
packetsReceivedWithCe?: number;
|
|
966
|
+
/**
|
|
967
|
+
* Total number of RTP packets for which an RFC8888 report has been sent with a zero R bit.
|
|
968
|
+
*/
|
|
969
|
+
packetsReportedAsLost?: number;
|
|
970
|
+
/**
|
|
971
|
+
* Total number of RTP packets reported as lost but later recovered in a subsequent RFC8888 report.
|
|
972
|
+
*/
|
|
973
|
+
packetsReportedAsLostButRecovered?: number;
|
|
974
|
+
/**
|
|
890
975
|
* Number of packets lost on the RTP stream.
|
|
891
976
|
*/
|
|
892
977
|
packetsLost?: number;
|
|
@@ -895,7 +980,7 @@ type InboundRtpStats = {
|
|
|
895
980
|
*/
|
|
896
981
|
jitter?: number;
|
|
897
982
|
/**
|
|
898
|
-
* The
|
|
983
|
+
* The media stream identification tag from the SDP media section.
|
|
899
984
|
*/
|
|
900
985
|
mid?: string;
|
|
901
986
|
/**
|
|
@@ -991,15 +1076,15 @@ type InboundRtpStats = {
|
|
|
991
1076
|
*/
|
|
992
1077
|
bytesReceived?: number;
|
|
993
1078
|
/**
|
|
994
|
-
* Number of NACKs
|
|
1079
|
+
* Number of NACKs received.
|
|
995
1080
|
*/
|
|
996
1081
|
nackCount?: number;
|
|
997
1082
|
/**
|
|
998
|
-
* Number of Full Intra Requests
|
|
1083
|
+
* Number of Full Intra Requests received.
|
|
999
1084
|
*/
|
|
1000
1085
|
firCount?: number;
|
|
1001
1086
|
/**
|
|
1002
|
-
* Number of Picture Loss Indications
|
|
1087
|
+
* Number of Picture Loss Indications received.
|
|
1003
1088
|
*/
|
|
1004
1089
|
pliCount?: number;
|
|
1005
1090
|
/**
|
|
@@ -1177,10 +1262,14 @@ type OutboundTrackSample = {
|
|
|
1177
1262
|
*/
|
|
1178
1263
|
kind: string;
|
|
1179
1264
|
/**
|
|
1180
|
-
* Calculated score for track (details should be added to
|
|
1265
|
+
* Calculated score for track (details should be added to scoreReasons)
|
|
1181
1266
|
*/
|
|
1182
1267
|
score?: number;
|
|
1183
1268
|
/**
|
|
1269
|
+
* Reasons for the score calculation, mapping each reason to how much it contributed to the score
|
|
1270
|
+
*/
|
|
1271
|
+
scoreReasons?: Record<string, number>;
|
|
1272
|
+
/**
|
|
1184
1273
|
* Additional information attached to this stats
|
|
1185
1274
|
*/
|
|
1186
1275
|
attachments?: Record<string, unknown>;
|
|
@@ -1202,16 +1291,20 @@ type InboundTrackSample = {
|
|
|
1202
1291
|
*/
|
|
1203
1292
|
kind: string;
|
|
1204
1293
|
/**
|
|
1205
|
-
* Calculated score for track (details should be added to
|
|
1294
|
+
* Calculated score for track (details should be added to scoreReasons)
|
|
1206
1295
|
*/
|
|
1207
1296
|
score?: number;
|
|
1208
1297
|
/**
|
|
1298
|
+
* Reasons for the score calculation, mapping each reason to how much it contributed to the score
|
|
1299
|
+
*/
|
|
1300
|
+
scoreReasons?: Record<string, number>;
|
|
1301
|
+
/**
|
|
1209
1302
|
* Additional information attached to this stats
|
|
1210
1303
|
*/
|
|
1211
1304
|
attachments?: Record<string, unknown>;
|
|
1212
1305
|
};
|
|
1213
1306
|
/**
|
|
1214
|
-
*
|
|
1307
|
+
* A sample containing statistics and metrics for a WebRTC peer connection
|
|
1215
1308
|
*/
|
|
1216
1309
|
type PeerConnectionSample = {
|
|
1217
1310
|
/**
|
|
@@ -1223,10 +1316,14 @@ type PeerConnectionSample = {
|
|
|
1223
1316
|
*/
|
|
1224
1317
|
attachments?: Record<string, unknown>;
|
|
1225
1318
|
/**
|
|
1226
|
-
* Calculated score for peer connection (details should be added to
|
|
1319
|
+
* Calculated score for peer connection (details should be added to scoreReasons)
|
|
1227
1320
|
*/
|
|
1228
1321
|
score?: number;
|
|
1229
1322
|
/**
|
|
1323
|
+
* Reasons for the score calculation, mapping each reason to how much it contributed to the score
|
|
1324
|
+
*/
|
|
1325
|
+
scoreReasons?: Record<string, number>;
|
|
1326
|
+
/**
|
|
1230
1327
|
* Inbound Track Stats items
|
|
1231
1328
|
*/
|
|
1232
1329
|
inboundTracks?: InboundTrackSample[];
|
|
@@ -1239,19 +1336,19 @@ type PeerConnectionSample = {
|
|
|
1239
1336
|
*/
|
|
1240
1337
|
codecs?: CodecStats[];
|
|
1241
1338
|
/**
|
|
1242
|
-
* Inbound
|
|
1339
|
+
* Inbound RTP Stats
|
|
1243
1340
|
*/
|
|
1244
1341
|
inboundRtps?: InboundRtpStats[];
|
|
1245
1342
|
/**
|
|
1246
|
-
* Remote Inbound
|
|
1343
|
+
* Remote Inbound RTP Stats
|
|
1247
1344
|
*/
|
|
1248
1345
|
remoteInboundRtps?: RemoteInboundRtpStats[];
|
|
1249
1346
|
/**
|
|
1250
|
-
* Outbound
|
|
1347
|
+
* Outbound RTP Stats
|
|
1251
1348
|
*/
|
|
1252
1349
|
outboundRtps?: OutboundRtpStats[];
|
|
1253
1350
|
/**
|
|
1254
|
-
* Remote Outbound
|
|
1351
|
+
* Remote Outbound RTP Stats
|
|
1255
1352
|
*/
|
|
1256
1353
|
remoteOutboundRtps?: RemoteOutboundRtpStats[];
|
|
1257
1354
|
/**
|
|
@@ -1283,7 +1380,7 @@ type PeerConnectionSample = {
|
|
|
1283
1380
|
*/
|
|
1284
1381
|
iceCandidatePairs?: IceCandidatePairStats[];
|
|
1285
1382
|
/**
|
|
1286
|
-
*
|
|
1383
|
+
* Certificate Stats
|
|
1287
1384
|
*/
|
|
1288
1385
|
certificates?: CertificateStats[];
|
|
1289
1386
|
};
|
|
@@ -1308,10 +1405,14 @@ type ClientSample = {
|
|
|
1308
1405
|
*/
|
|
1309
1406
|
attachments?: Record<string, unknown>;
|
|
1310
1407
|
/**
|
|
1311
|
-
* Calculated score for client (details should be added to
|
|
1408
|
+
* Calculated score for client (details should be added to scoreReasons)
|
|
1312
1409
|
*/
|
|
1313
1410
|
score?: number;
|
|
1314
1411
|
/**
|
|
1412
|
+
* Reasons for the score calculation, mapping each reason to how much it contributed to the score
|
|
1413
|
+
*/
|
|
1414
|
+
scoreReasons?: Record<string, number>;
|
|
1415
|
+
/**
|
|
1315
1416
|
* Samples taken PeerConnections
|
|
1316
1417
|
*/
|
|
1317
1418
|
peerConnections?: PeerConnectionSample[];
|
|
@@ -1585,6 +1686,24 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
|
|
|
1585
1686
|
bitPerPixel: number;
|
|
1586
1687
|
deltaPacketsSent: number;
|
|
1587
1688
|
deltaBytesSent: number;
|
|
1689
|
+
deltaFramesSent: number;
|
|
1690
|
+
deltaFramesEncoded: number;
|
|
1691
|
+
deltaKeyFramesEncoded: number;
|
|
1692
|
+
deltaNackCount: number;
|
|
1693
|
+
deltaPliCount: number;
|
|
1694
|
+
deltaFirCount: number;
|
|
1695
|
+
deltaRetransmittedPacketsSent: number;
|
|
1696
|
+
deltaRetransmittedBytesSent: number;
|
|
1697
|
+
deltaEncodeTime: number;
|
|
1698
|
+
deltaQualityLimitationResolutionChanges: number;
|
|
1699
|
+
/**
|
|
1700
|
+
* `true` when the codec, encoder implementation or scalability mode changed in this tick.
|
|
1701
|
+
*
|
|
1702
|
+
* Chrome resets `packetsSent`/`bytesSent` on the SSRC when the codec switches
|
|
1703
|
+
* (crbug.com/webrtc/5361), producing sawtooth spikes and negative bitrates. All deltas for the
|
|
1704
|
+
* tick are suppressed so a codec change is never mistaken for a traffic event.
|
|
1705
|
+
*/
|
|
1706
|
+
counterResetBoundary: boolean;
|
|
1588
1707
|
remoteRttInMs?: number;
|
|
1589
1708
|
remoteFractionLost?: number;
|
|
1590
1709
|
remoteJitter?: number;
|
|
@@ -1599,6 +1718,241 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
|
|
|
1599
1718
|
update(stats: OutboundRtpStats): void;
|
|
1600
1719
|
}
|
|
1601
1720
|
|
|
1721
|
+
/**
|
|
1722
|
+
* Small statistics helpers used by the aggregators and detectors.
|
|
1723
|
+
*
|
|
1724
|
+
* Rationale: a mean is a poor summary for call telemetry — one participant with a 1500 ms RTT
|
|
1725
|
+
* skews the average for nine healthy ones. Detectors should reason with medians, high percentiles
|
|
1726
|
+
* and "affected ratios" instead.
|
|
1727
|
+
*/
|
|
1728
|
+
/** A distribution summary of a numeric sample set. */
|
|
1729
|
+
type StatsSummary = {
|
|
1730
|
+
count: number;
|
|
1731
|
+
min: number;
|
|
1732
|
+
max: number;
|
|
1733
|
+
mean: number;
|
|
1734
|
+
median: number;
|
|
1735
|
+
p25: number;
|
|
1736
|
+
p75: number;
|
|
1737
|
+
p95: number;
|
|
1738
|
+
};
|
|
1739
|
+
/**
|
|
1740
|
+
* The p-th percentile (0..1) using linear interpolation between closest ranks.
|
|
1741
|
+
* Returns `undefined` for an empty input.
|
|
1742
|
+
*/
|
|
1743
|
+
declare function percentile(values: number[], p: number): number | undefined;
|
|
1744
|
+
/** The median (50th percentile). `undefined` for an empty input. */
|
|
1745
|
+
declare function median(values: number[]): number | undefined;
|
|
1746
|
+
/**
|
|
1747
|
+
* Median absolute deviation: `median(|value - median(values)|)`. A robust alternative to standard
|
|
1748
|
+
* deviation for describing how spread out `values` are — one wild outlier shifts a mean-based
|
|
1749
|
+
* deviation a lot, but barely moves a median. `undefined` for an empty input.
|
|
1750
|
+
*/
|
|
1751
|
+
declare function medianAbsoluteDeviation(values: number[]): number | undefined;
|
|
1752
|
+
/**
|
|
1753
|
+
* A robust z-score: how many (MAD-based) standard deviations `value` sits above/below the median of
|
|
1754
|
+
* `baseline`.
|
|
1755
|
+
*
|
|
1756
|
+
* Uses the median and MAD instead of the mean and standard deviation so a handful of baseline
|
|
1757
|
+
* outliers can't inflate the "normal" spread and mask a genuine spike — the same reasoning behind
|
|
1758
|
+
* {@link summarize}'s percentiles applies here to a single scalar spread. `1.4826` is the constant
|
|
1759
|
+
* that makes MAD estimate the standard deviation of a normal distribution, so the result reads on
|
|
1760
|
+
* the same scale as a classic z-score.
|
|
1761
|
+
*
|
|
1762
|
+
* A classic z-score divides by zero once every baseline value is identical (`MAD === 0`). Here:
|
|
1763
|
+
* `value` strictly above that constant baseline reads as `Infinity` — a spike with no precedent
|
|
1764
|
+
* whatsoever, however small; `value` at or below it reads as `0`, indistinguishable from the (flat)
|
|
1765
|
+
* baseline rather than a division error.
|
|
1766
|
+
*
|
|
1767
|
+
* ### Worked example
|
|
1768
|
+
*
|
|
1769
|
+
* A baseline of "share of clients reporting congestion", one entry per 10 s bucket, on a healthy
|
|
1770
|
+
* fleet: `[0.02, 0.01, 0.03, 0.04, 0.02]`. Median `0.02`, MAD `0.01`, so the scale is
|
|
1771
|
+
* `1.4826 * 0.01 ≈ 0.0148`.
|
|
1772
|
+
*
|
|
1773
|
+
* ```ts
|
|
1774
|
+
* robustZScore(0.03, baseline); // ≈ 0.67 — an ordinary bucket
|
|
1775
|
+
* robustZScore(0.25, baseline); // ≈ 15.5 — a quarter of the fleet at once; nothing like it before
|
|
1776
|
+
* ```
|
|
1777
|
+
*
|
|
1778
|
+
* Now add one bad bucket to the *baseline* — `[0.02, 0.01, 0.03, 0.04, 0.02, 0.40]`. A mean/stddev
|
|
1779
|
+
* z-score would absorb it: the mean climbs to `0.087` and the stddev to `~0.14`, so a genuine `0.25`
|
|
1780
|
+
* spike scores about `1.2` and looks unremarkable. **One past incident would hide the next one.**
|
|
1781
|
+
* Median and MAD barely move (median `0.025`, MAD `0.01`), so `0.25` still scores `≈15.2`. That
|
|
1782
|
+
* resistance is the entire reason for this function.
|
|
1783
|
+
*
|
|
1784
|
+
* The `Infinity` case is not an edge case in practice — it is a fleet that has been perfectly quiet:
|
|
1785
|
+
*
|
|
1786
|
+
* ```ts
|
|
1787
|
+
* robustZScore(0.10, [ 0, 0, 0, 0, 0 ]); // Infinity — first congestion ever seen
|
|
1788
|
+
* robustZScore(0, [ 0, 0, 0, 0, 0 ]); // 0 — still nothing happening
|
|
1789
|
+
* ```
|
|
1790
|
+
*
|
|
1791
|
+
* `Infinity` clears any finite threshold, which is intended: "we have never seen this" *is* the
|
|
1792
|
+
* strongest possible statistical statement. It is also why a caller must gate on practical
|
|
1793
|
+
* significance too — see `SfuCongestionDetector`, which additionally requires a minimum number of
|
|
1794
|
+
* affected clients, so a single client on a quiet fleet cannot page anyone.
|
|
1795
|
+
*
|
|
1796
|
+
* `undefined` when `baseline` is empty — there is nothing to compare against.
|
|
1797
|
+
*/
|
|
1798
|
+
declare function robustZScore(value: number, baseline: number[]): number | undefined;
|
|
1799
|
+
/** Summarize a numeric sample set. Returns `undefined` for an empty input. */
|
|
1800
|
+
declare function summarize(values: number[]): StatsSummary | undefined;
|
|
1801
|
+
/**
|
|
1802
|
+
* Linear-interpolated percentile of an **already ascending** array. The building block behind
|
|
1803
|
+
* {@link percentile} and {@link summarize}; exported so callers computing several percentiles over
|
|
1804
|
+
* the same data can sort once themselves.
|
|
1805
|
+
*
|
|
1806
|
+
* Passing an unsorted array yields a meaningless number rather than an error — sort first.
|
|
1807
|
+
*/
|
|
1808
|
+
declare function percentileOfSorted(sorted: number[], p: number): number;
|
|
1809
|
+
/**
|
|
1810
|
+
* A counter-reset-safe delta: the increase of a cumulative counter between two observations.
|
|
1811
|
+
*
|
|
1812
|
+
* Returns `0` when the counter went backwards (reset / SSRC reuse) or when either side is missing.
|
|
1813
|
+
* NOTE the guard is `>=` on **defined** values — a previous value of `0` is a perfectly valid
|
|
1814
|
+
* baseline, so `0 -> 5` correctly yields `5` (using a truthiness check here silently drops the
|
|
1815
|
+
* first interval of every counter, which is exactly when the first loss/freeze event happens).
|
|
1816
|
+
*/
|
|
1817
|
+
declare function counterDelta(previous: number | undefined, current: number | undefined): number;
|
|
1818
|
+
/**
|
|
1819
|
+
* Pearson correlation of two equal-length series, clamped to `0..1`.
|
|
1820
|
+
*
|
|
1821
|
+
* Negative or undefined relationships read as `0`, because every caller here asks "does A follow B?"
|
|
1822
|
+
* — an inverse relationship is not a weaker yes, it is a no.
|
|
1823
|
+
*/
|
|
1824
|
+
declare function correlation(xs: number[], ys: number[]): number;
|
|
1825
|
+
/** Per-step result of {@link pageHinkley}. */
|
|
1826
|
+
type PageHinkleyResult = {
|
|
1827
|
+
/**
|
|
1828
|
+
* The Page-Hinkley statistic after each observation, same length as the input — one value per
|
|
1829
|
+
* `values[i]`, so `statistic[i]` is "how far the cumulative deviation has grown above its own
|
|
1830
|
+
* historical low, using only `values[0..i]`". Never negative (it's a gap to a *minimum*), and
|
|
1831
|
+
* `0` for as long as the process looks stable.
|
|
1832
|
+
*/
|
|
1833
|
+
statistic: number[];
|
|
1834
|
+
/**
|
|
1835
|
+
* Index of the **first** observation whose statistic exceeded `lambda`, else `undefined`.
|
|
1836
|
+
*
|
|
1837
|
+
* This is a one-shot latch over the call's whole input: once found, later observations are not
|
|
1838
|
+
* re-checked, even if the statistic subsequently falls back down (e.g. after a single transient
|
|
1839
|
+
* spike — see the class doc example). "Is it *still* elevated right now" is a different
|
|
1840
|
+
* question, answered by looking at the *tail* of {@link statistic}, or — for a live stream — by
|
|
1841
|
+
* re-running this over a recent window each time (which is what `TrendTester` does) rather than
|
|
1842
|
+
* over the whole history once.
|
|
1843
|
+
*/
|
|
1844
|
+
changePointIndex?: number;
|
|
1845
|
+
/** `true` when {@link changePointIndex} is defined. */
|
|
1846
|
+
changeDetected: boolean;
|
|
1847
|
+
};
|
|
1848
|
+
/**
|
|
1849
|
+
* Page-Hinkley test: sequential (online) detection of a sustained **increase** in the mean of
|
|
1850
|
+
* `values` — "has this metric settled onto a durably higher level", as opposed to "did one sample
|
|
1851
|
+
* come in high". A single noisy point should not read as a regression; ten points that are all a
|
|
1852
|
+
* bit higher than before should.
|
|
1853
|
+
*
|
|
1854
|
+
* ### How it works
|
|
1855
|
+
*
|
|
1856
|
+
* At step `i`, three numbers are tracked:
|
|
1857
|
+
*
|
|
1858
|
+
* 1. `mean` — the running average of `values[0..i]` (**not** a fixed baseline — it is recomputed
|
|
1859
|
+
* from everything seen so far, including `values[i]` itself, which is what makes this
|
|
1860
|
+
* *adaptive*: after a real shift, `mean` keeps drifting up to meet the new level, and the
|
|
1861
|
+
* signal below fades back out on its own rather than staying triggered forever).
|
|
1862
|
+
* 2. `cumulative` — running sum of `(values[i] - mean - delta)`. Subtracting `mean` centres each
|
|
1863
|
+
* term on "surprise relative to what we've seen so far"; subtracting `delta` on top means a
|
|
1864
|
+
* small positive surprise still nets to a *negative* contribution, so it doesn't accumulate.
|
|
1865
|
+
* 3. `runningMinimum` — the lowest `cumulative` has ever been.
|
|
1866
|
+
*
|
|
1867
|
+
* The **Page-Hinkley statistic** is `cumulative - runningMinimum`: how far the running sum has
|
|
1868
|
+
* climbed above its own historical floor. Pure noise pulls `cumulative` up and down around a flat
|
|
1869
|
+
* trend, so the gap to `runningMinimum` stays small. A sustained increase pushes `cumulative`
|
|
1870
|
+
* mostly one direction — up — so `runningMinimum` stops updating and the gap grows every step,
|
|
1871
|
+
* crossing `lambda` once the shift is large/long enough to be sure it isn't noise. That first
|
|
1872
|
+
* crossing is {@link PageHinkleyResult.changePointIndex}.
|
|
1873
|
+
*
|
|
1874
|
+
* ### Parameters
|
|
1875
|
+
*
|
|
1876
|
+
* - `delta` — the **drift tolerance** (in the same units as `values`, e.g. ms of RTT): how much of
|
|
1877
|
+
* a step-to-step increase is written off as noise rather than counted towards the cumulative sum.
|
|
1878
|
+
* `0` means even a razor-thin, consistent upward creep eventually accumulates enough to trigger;
|
|
1879
|
+
* raising it requires each observation to clear that bar above the running mean before it
|
|
1880
|
+
* contributes anything (see the class doc's hand-worked example — the same jump that triggers
|
|
1881
|
+
* with `delta: 0` is completely absorbed at `delta: 10`).
|
|
1882
|
+
* - `lambda` — the **detection threshold** the statistic must exceed. It is in "surprise units" (a
|
|
1883
|
+
* sum of deviations, not a single observation's units), so there's no shortcut for picking it
|
|
1884
|
+
* other than trying it against representative data — see the RTT examples in `stats.spec.ts` for
|
|
1885
|
+
* a worked comparison of a low vs. a high `lambda` on the same series. Raising it delays
|
|
1886
|
+
* detection but makes a false positive from a lucky run of noise less likely.
|
|
1887
|
+
*
|
|
1888
|
+
* O(n) — one pass, unlike {@link mannKendall}'s O(n²). To detect a **decrease** instead of an
|
|
1889
|
+
* increase, negate `values` before calling (or negate the result's meaning if you'd rather).
|
|
1890
|
+
*/
|
|
1891
|
+
declare function pageHinkley(values: number[], delta?: number, lambda?: number): PageHinkleyResult;
|
|
1892
|
+
/** Result of {@link mannKendall}. */
|
|
1893
|
+
type MannKendallResult = {
|
|
1894
|
+
/** Sum of pairwise signs (`sign(values[j] - values[i])` for every `i < j`). */
|
|
1895
|
+
s: number;
|
|
1896
|
+
/** Variance of {@link s}, corrected for tied values. */
|
|
1897
|
+
variance: number;
|
|
1898
|
+
/** Standard-normal score derived from `s`. `0` when `s` is `0` — no evidence either way. */
|
|
1899
|
+
z: number;
|
|
1900
|
+
/** Two-tailed p-value for the null hypothesis "no monotonic trend". */
|
|
1901
|
+
pValue: number;
|
|
1902
|
+
/** `'increasing'` / `'decreasing'` when significant at `alpha`, else `'no-trend'`. */
|
|
1903
|
+
trend: 'increasing' | 'decreasing' | 'no-trend';
|
|
1904
|
+
};
|
|
1905
|
+
/**
|
|
1906
|
+
* Mann-Kendall trend test: a non-parametric test for a **monotonic** trend in `values`, without
|
|
1907
|
+
* assuming a distribution or a constant rate of change — it only asks "are later values
|
|
1908
|
+
* consistently larger (or smaller) than earlier ones more often than chance would allow".
|
|
1909
|
+
*
|
|
1910
|
+
* Every pair `i < j` votes `+1` (`values[j] > values[i]`), `-1` (`values[j] < values[i]`) or `0`
|
|
1911
|
+
* (tie); `s` is the sum of those votes. Under the null hypothesis of no trend, `s` is
|
|
1912
|
+
* approximately normal with mean `0` and a known variance (corrected here for tied values), which
|
|
1913
|
+
* turns `s` into a Z score and a two-tailed p-value. `trend` is only `'increasing'` / `'decreasing'`
|
|
1914
|
+
* when that p-value clears `alpha` — a handful of mostly-ascending points is exactly what noise
|
|
1915
|
+
* looks like half the time, and this is what keeps that from reading as a trend.
|
|
1916
|
+
*
|
|
1917
|
+
* O(n²) (every pair is compared); fine for the small, per-tick sample counts detectors work with,
|
|
1918
|
+
* not for large historical series.
|
|
1919
|
+
*/
|
|
1920
|
+
declare function mannKendall(values: number[], alpha?: number): MannKendallResult;
|
|
1921
|
+
/**
|
|
1922
|
+
* Turns a Mann-Kendall `s` statistic and its `variance` into a Z score, p-value and verdict — the
|
|
1923
|
+
* back half of {@link mannKendall}, split out so an incremental caller that maintains `s` /
|
|
1924
|
+
* `variance` itself (e.g. over a sliding window, correcting for evicted points rather than
|
|
1925
|
+
* recomputing every pair from scratch) doesn't have to reimplement the normal approximation.
|
|
1926
|
+
*/
|
|
1927
|
+
declare function mannKendallVerdict(s: number, variance: number, alpha?: number): Pick<MannKendallResult, 'z' | 'pValue' | 'trend'>;
|
|
1928
|
+
|
|
1929
|
+
/** The per-receiver view the aggregator builds for one subscribed track. */
|
|
1930
|
+
type PublishedTrackReceivingDistributionEntry = {
|
|
1931
|
+
numberOfReceivers: number;
|
|
1932
|
+
numberOfHealthyReceivers: number;
|
|
1933
|
+
numberOfDegradedReceivers: number;
|
|
1934
|
+
/** degradedReceivers / receivers (0..1); `0` when there are no receivers. */
|
|
1935
|
+
degradedRatio: number;
|
|
1936
|
+
/** Distribution summaries across receivers (undefined when no receiver reported the metric). */
|
|
1937
|
+
bitrate?: StatsSummary;
|
|
1938
|
+
fractionLost?: StatsSummary;
|
|
1939
|
+
jitter?: StatsSummary;
|
|
1940
|
+
rttInMs?: StatsSummary;
|
|
1941
|
+
jitterBufferDelayInMs?: StatsSummary;
|
|
1942
|
+
concealmentRatio?: StatsSummary;
|
|
1943
|
+
/** Fan-out counters: how many receivers saw the symptom, and the total across them. */
|
|
1944
|
+
freezes: {
|
|
1945
|
+
affectedReceivers: number;
|
|
1946
|
+
total: number;
|
|
1947
|
+
};
|
|
1948
|
+
plis: {
|
|
1949
|
+
affectedReceivers: number;
|
|
1950
|
+
total: number;
|
|
1951
|
+
};
|
|
1952
|
+
concealment: {
|
|
1953
|
+
affectedReceivers: number;
|
|
1954
|
+
};
|
|
1955
|
+
};
|
|
1602
1956
|
declare class ObservedOutboundTrack implements OutboundTrackSample {
|
|
1603
1957
|
timestamp: number;
|
|
1604
1958
|
readonly id: string;
|
|
@@ -1609,18 +1963,28 @@ declare class ObservedOutboundTrack implements OutboundTrackSample {
|
|
|
1609
1963
|
private _visited;
|
|
1610
1964
|
appData?: Record<string, unknown>;
|
|
1611
1965
|
readonly remoteInboundTracks: Set<ObservedInboundTrack>;
|
|
1966
|
+
readonly detectors: Detectors;
|
|
1967
|
+
receivingDistribution?: PublishedTrackReceivingDistributionEntry;
|
|
1612
1968
|
readonly calculatedScore: CalculatedScore;
|
|
1613
1969
|
addedAt?: number | undefined;
|
|
1614
1970
|
removedAt?: number | undefined;
|
|
1615
1971
|
muted?: boolean;
|
|
1616
1972
|
attachments?: Record<string, unknown> | undefined;
|
|
1973
|
+
degradedReasons?: string[] | undefined;
|
|
1974
|
+
bitrate?: number | undefined;
|
|
1975
|
+
deltaPacketsSent?: number | undefined;
|
|
1976
|
+
remoteFractionLost?: number | undefined;
|
|
1977
|
+
remoteRttInMs?: number | undefined;
|
|
1978
|
+
qualityLimitationReason?: string | undefined;
|
|
1617
1979
|
constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _outboundRtps?: ObservedOutboundRtp[] | undefined, _mediaSource?: ObservedMediaSource | undefined);
|
|
1618
1980
|
get score(): number | undefined;
|
|
1619
1981
|
get visited(): boolean;
|
|
1982
|
+
get degraded(): boolean;
|
|
1620
1983
|
getPeerConnection(): ObservedPeerConnection;
|
|
1621
1984
|
getOutboundRtps(): ObservedOutboundRtp[] | undefined;
|
|
1622
1985
|
getMediaSource(): ObservedMediaSource | undefined;
|
|
1623
1986
|
update(stats: OutboundTrackSample): void;
|
|
1987
|
+
private createReceivingDistribution;
|
|
1624
1988
|
}
|
|
1625
1989
|
|
|
1626
1990
|
declare class ObservedInboundTrack implements InboundTrackSample {
|
|
@@ -1638,6 +2002,8 @@ declare class ObservedInboundTrack implements InboundTrackSample {
|
|
|
1638
2002
|
removedAt?: number | undefined;
|
|
1639
2003
|
muted?: boolean;
|
|
1640
2004
|
attachments?: Record<string, unknown> | undefined;
|
|
2005
|
+
degradationReasons: string[];
|
|
2006
|
+
get degraded(): boolean;
|
|
1641
2007
|
constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _inboundRtp?: ObservedInboundRtp | undefined, _mediaPlayout?: ObservedMediaPlayout | undefined);
|
|
1642
2008
|
get score(): number | undefined;
|
|
1643
2009
|
get visited(): boolean;
|
|
@@ -1645,6 +2011,7 @@ declare class ObservedInboundTrack implements InboundTrackSample {
|
|
|
1645
2011
|
getInboundRtp(): ObservedInboundRtp | undefined;
|
|
1646
2012
|
getMediaPlayout(): ObservedMediaPlayout | undefined;
|
|
1647
2013
|
update(stats: InboundTrackSample): void;
|
|
2014
|
+
private checkDegradation;
|
|
1648
2015
|
}
|
|
1649
2016
|
|
|
1650
2017
|
declare class ObservedRemoteOutboundRtp implements RemoteOutboundRtpStats {
|
|
@@ -1752,6 +2119,46 @@ declare class ObservedInboundRtp implements InboundRtpStats {
|
|
|
1752
2119
|
deltaBytesReceived: number;
|
|
1753
2120
|
deltaReceivedSamples: number;
|
|
1754
2121
|
deltaSilentConcealedSamples: number;
|
|
2122
|
+
deltaConcealedSamples: number;
|
|
2123
|
+
deltaConcealmentEvents: number;
|
|
2124
|
+
deltaFreezeCount: number;
|
|
2125
|
+
deltaFreezesDuration: number;
|
|
2126
|
+
deltaPliCount: number;
|
|
2127
|
+
deltaNackCount: number;
|
|
2128
|
+
deltaFirCount: number;
|
|
2129
|
+
deltaPacketsDiscarded: number;
|
|
2130
|
+
deltaFramesDecoded: number;
|
|
2131
|
+
deltaFramesReceived: number;
|
|
2132
|
+
deltaFramesRendered: number;
|
|
2133
|
+
deltaFramesDropped: number;
|
|
2134
|
+
deltaKeyFramesDecoded: number;
|
|
2135
|
+
deltaDecodeTime: number;
|
|
2136
|
+
deltaJitterBufferDelay: number;
|
|
2137
|
+
deltaJitterBufferEmittedCount: number;
|
|
2138
|
+
deltaRetransmittedPacketsReceived: number;
|
|
2139
|
+
deltaFecPacketsReceived: number;
|
|
2140
|
+
deltaFecPacketsDiscarded: number;
|
|
2141
|
+
deltaPausesDuration: number;
|
|
2142
|
+
/**
|
|
2143
|
+
* Mean jitter-buffer delay for the frames/samples emitted in this tick (seconds), derived from
|
|
2144
|
+
* the cumulative `jitterBufferDelay` / `jitterBufferEmittedCount` pair — the only correct way to
|
|
2145
|
+
* read those two counters.
|
|
2146
|
+
*/
|
|
2147
|
+
jitterBufferDelayInMs?: number;
|
|
2148
|
+
/** Fraction of the samples received in this tick that were concealed (0..1). */
|
|
2149
|
+
concealmentRatio?: number;
|
|
2150
|
+
/** Fraction of the frames received in this tick that were dropped before rendering (0..1). */
|
|
2151
|
+
framesDroppedRatio?: number;
|
|
2152
|
+
/**
|
|
2153
|
+
* `true` when the codec or decoder implementation changed in this tick.
|
|
2154
|
+
*
|
|
2155
|
+
* Chrome resets `packetsReceived`/`bytesReceived` on an SSRC when the codec switches
|
|
2156
|
+
* (crbug.com/webrtc/5361, open since 2015), which shows up as a sawtooth spike or a negative
|
|
2157
|
+
* rate. Every delta in this tick is therefore suppressed to `0` rather than reported as traffic
|
|
2158
|
+
* — otherwise a room-wide codec rollout produces a synchronized fake-degradation alert across
|
|
2159
|
+
* every participant at once.
|
|
2160
|
+
*/
|
|
2161
|
+
counterResetBoundary: boolean;
|
|
1755
2162
|
remoteRttInMs?: number;
|
|
1756
2163
|
remoteBytesSent?: number;
|
|
1757
2164
|
remotePacketsSent?: number;
|
|
@@ -1941,7 +2348,34 @@ declare class ObservedPeerConnection extends EventEmitter {
|
|
|
1941
2348
|
sendingVideoBitrate: number;
|
|
1942
2349
|
receivingAudioBitrate: number;
|
|
1943
2350
|
receivingVideoBitrate: number;
|
|
2351
|
+
/**
|
|
2352
|
+
* Median round-trip time of the tick, in ms.
|
|
2353
|
+
*
|
|
2354
|
+
* Prefers {@link rtcpRttInMs} and falls back to {@link iceRttInMs}, so within one tick it always
|
|
2355
|
+
* reports **one** kind of round trip. It used to be the median of both mixed together, which was
|
|
2356
|
+
* a bug: the mixing ratio changed as streams came and went, so the value moved for reasons that
|
|
2357
|
+
* had nothing to do with the network.
|
|
2358
|
+
*/
|
|
1944
2359
|
currentRttInMs?: number;
|
|
2360
|
+
/**
|
|
2361
|
+
* RTT measured by ICE/STUN consent checks, in ms — the trip to **whatever terminates ICE**. In an
|
|
2362
|
+
* SFU topology that is the SFU, so this is the client↔SFU leg, not client↔client.
|
|
2363
|
+
*/
|
|
2364
|
+
iceRttInMs?: number;
|
|
2365
|
+
/**
|
|
2366
|
+
* RTT reported by RTCP receiver reports, in ms — an **end-to-end** media-path round trip.
|
|
2367
|
+
*
|
|
2368
|
+
* Not the same trip as {@link iceRttInMs}; the difference between the two is roughly the far side
|
|
2369
|
+
* of the SFU (see {@link sfuHopRttInMs}).
|
|
2370
|
+
*/
|
|
2371
|
+
rtcpRttInMs?: number;
|
|
2372
|
+
/**
|
|
2373
|
+
* `rtcpRttInMs - iceRttInMs`, when both are known — an estimate of everything *past* the SFU.
|
|
2374
|
+
*
|
|
2375
|
+
* Useful for splitting "this client's own last mile is slow" (high `iceRttInMs`) from "the path
|
|
2376
|
+
* beyond the SFU is slow" (low ICE, high hop).
|
|
2377
|
+
*/
|
|
2378
|
+
sfuHopRttInMs?: number;
|
|
1945
2379
|
currentJitter?: number;
|
|
1946
2380
|
usingTCP: boolean;
|
|
1947
2381
|
usingTURN: boolean;
|
|
@@ -2036,6 +2470,487 @@ type OperationSystem = {
|
|
|
2036
2470
|
version: string;
|
|
2037
2471
|
};
|
|
2038
2472
|
|
|
2473
|
+
/**
|
|
2474
|
+
* Turning a correlation into a **conclusion**.
|
|
2475
|
+
*
|
|
2476
|
+
* Every detector in this library ultimately reports the same shape of observation: *N clients have
|
|
2477
|
+
* issue X open at once, and here is what they have in common*. That is useful but not yet
|
|
2478
|
+
* actionable — someone still has to know that congestion spread across unrelated calls means the
|
|
2479
|
+
* server, while CPU limitation spread across unrelated calls means a bad client release. This module
|
|
2480
|
+
* holds that interpretation step so it is stated once, consistently, instead of being re-derived by
|
|
2481
|
+
* whoever reads the alert at 3am.
|
|
2482
|
+
*
|
|
2483
|
+
* ### Two functions, because there are two questions
|
|
2484
|
+
*
|
|
2485
|
+
* A detector already knows its scope — it was constructed with an `ObservedCall` or with the
|
|
2486
|
+
* `Observer`. Handing that scope back to a single generic function meant every caller supplied
|
|
2487
|
+
* fields the other scope needed and its own scope ignored: a call-scoped detector passing
|
|
2488
|
+
* `affectedCalls: 1, totalCalls: 1` forever, an observer-scoped one passing a participant ratio that
|
|
2489
|
+
* was deliberately never read. Placeholders like that are a standing invitation to read them as if
|
|
2490
|
+
* they meant something.
|
|
2491
|
+
*
|
|
2492
|
+
* So there are two entry points, each taking only the facts its scope actually has:
|
|
2493
|
+
*
|
|
2494
|
+
* - {@link concludeCallIssue} — within one call. The axis is *how much of the meeting*, and whether
|
|
2495
|
+
* the affected clients all subscribe to one published track.
|
|
2496
|
+
* - {@link concludeObserverIssue} — across calls. The axis is *how many independent calls*, which is
|
|
2497
|
+
* the only thing that separates "one bad room" from "our infrastructure".
|
|
2498
|
+
*
|
|
2499
|
+
* Neither the issue family nor the spread concludes anything alone: congestion in one call is a
|
|
2500
|
+
* meeting problem, congestion in six calls is an infrastructure problem, and the issue type is
|
|
2501
|
+
* identical in both.
|
|
2502
|
+
*/
|
|
2503
|
+
/** Where the fault most likely sits, given who is affected. */
|
|
2504
|
+
type IssueFaultDomain =
|
|
2505
|
+
/** Independent calls affected at once — they share only the servers and the network. */
|
|
2506
|
+
'infrastructure'
|
|
2507
|
+
/** One call, broadly affected — something that call shares (its SFU worker, room, or host). */
|
|
2508
|
+
| 'call'
|
|
2509
|
+
/** The subscribers of one published track — the publisher's path or the forwarding of it. */
|
|
2510
|
+
| 'published-track'
|
|
2511
|
+
/** A single endpoint — its own device or last mile. */
|
|
2512
|
+
| 'endpoint'
|
|
2513
|
+
/** Independent calls affected, but by something endpoints own — a client build, not a server. */
|
|
2514
|
+
| 'client-population'
|
|
2515
|
+
/** Not enough signal to attribute. */
|
|
2516
|
+
| 'unknown';
|
|
2517
|
+
/** A stated verdict, attached to the raised issue payload. */
|
|
2518
|
+
type IssueConclusion = {
|
|
2519
|
+
/** Where to look. */
|
|
2520
|
+
faultDomain: IssueFaultDomain;
|
|
2521
|
+
/** One line, written to be readable in an alert without opening a dashboard. */
|
|
2522
|
+
summary: string;
|
|
2523
|
+
/** What to check first. Omitted when the issue family is unknown to this module. */
|
|
2524
|
+
recommendation?: string;
|
|
2525
|
+
/**
|
|
2526
|
+
* How much the spread alone justifies the verdict, `0..1`. Not a probability — a coarse ranking
|
|
2527
|
+
* so alerting can threshold on it. More independent calls, or a tighter onset, means higher.
|
|
2528
|
+
*/
|
|
2529
|
+
confidence: number;
|
|
2530
|
+
};
|
|
2531
|
+
/** The facts a **call-scoped** conclusion is drawn from. */
|
|
2532
|
+
type CallIssueSpread = {
|
|
2533
|
+
issueType: string;
|
|
2534
|
+
/** Distinct clients of this call with the issue open. */
|
|
2535
|
+
affectedClients: number;
|
|
2536
|
+
/** Participants in the call — the denominator. */
|
|
2537
|
+
totalClients: number;
|
|
2538
|
+
/** True when the onsets clustered — a shared trigger rather than drift. */
|
|
2539
|
+
onsetBurst: boolean;
|
|
2540
|
+
/**
|
|
2541
|
+
* Set when the affected clients are the subscriber set of **one published track**.
|
|
2542
|
+
*
|
|
2543
|
+
* The strongest call-scoped statement available: those clients share a publisher and nothing
|
|
2544
|
+
* else, so the receivers are exonerated and the source's path is implicated.
|
|
2545
|
+
*/
|
|
2546
|
+
publishedTrackId?: string;
|
|
2547
|
+
};
|
|
2548
|
+
/** The facts an **observer-scoped** conclusion is drawn from. */
|
|
2549
|
+
type ObserverIssueSpread = {
|
|
2550
|
+
issueType: string;
|
|
2551
|
+
/** Distinct clients across the fleet with the issue open. */
|
|
2552
|
+
affectedClients: number;
|
|
2553
|
+
/** Clients in the fleet. Reported for context; it does not gate anything at this scope. */
|
|
2554
|
+
totalClients: number;
|
|
2555
|
+
/** Distinct calls containing at least one affected client. The dimension that matters here. */
|
|
2556
|
+
affectedCalls: number;
|
|
2557
|
+
/** Calls in flight. */
|
|
2558
|
+
totalCalls: number;
|
|
2559
|
+
/** True when the onsets clustered. */
|
|
2560
|
+
onsetBurst: boolean;
|
|
2561
|
+
};
|
|
2562
|
+
/**
|
|
2563
|
+
* Draw the conclusion for a group of clients **within one call**.
|
|
2564
|
+
*
|
|
2565
|
+
* Ordered most-to-least specific: a track-scoped group is a stronger statement than a call-wide one,
|
|
2566
|
+
* and a single affected endpoint is not a statement about the call at all.
|
|
2567
|
+
*/
|
|
2568
|
+
declare function concludeCallIssue(spread: CallIssueSpread): IssueConclusion;
|
|
2569
|
+
/**
|
|
2570
|
+
* Draw the conclusion for a group of clients spanning **several calls**.
|
|
2571
|
+
*
|
|
2572
|
+
* One affected call is not an observer-scoped finding — it has an obvious local explanation and the
|
|
2573
|
+
* call-scoped detector has already reported it — so that case returns `call` and says so rather than
|
|
2574
|
+
* dressing it up as a fleet event.
|
|
2575
|
+
*
|
|
2576
|
+
* Which domain breadth implicates depends on the family, and this is the whole reason the module
|
|
2577
|
+
* exists: `congestion` across unrelated calls points at the servers, `cpulimitation` across unrelated
|
|
2578
|
+
* calls points at what those *endpoints* share — a client release, a browser version, shared
|
|
2579
|
+
* virtualised hardware — and pointing an SFU team at the second one wastes a night.
|
|
2580
|
+
*/
|
|
2581
|
+
declare function concludeObserverIssue(spread: ObserverIssueSpread): IssueConclusion;
|
|
2582
|
+
|
|
2583
|
+
/**
|
|
2584
|
+
* What every server-raised finding carries, whatever raised it.
|
|
2585
|
+
*
|
|
2586
|
+
* ### Why this is not `ClientIssue`
|
|
2587
|
+
*
|
|
2588
|
+
* `ClientIssue` is a **wire** type: it arrives inside a `ClientSample`, so its `payload` is bound to
|
|
2589
|
+
* what the schema can carry — a record of primitives since schema 3.5.0, a JSON string on samples
|
|
2590
|
+
* from earlier clients. Server-raised findings were reusing it, which forced every detector to `JSON.stringify` a
|
|
2591
|
+
* perfectly good object on the way out and every handler to `JSON.parse` it back on the way in —
|
|
2592
|
+
* paying serialisation on a path where nothing is ever serialised, and losing type information in
|
|
2593
|
+
* both directions. These go straight to an in-process handler, so they carry the object.
|
|
2594
|
+
*/
|
|
2595
|
+
type IssueBase = {
|
|
2596
|
+
/** What was found, e.g. `'CROSS_CALL_ISSUE_ONSET_BURST'`. */
|
|
2597
|
+
type: string;
|
|
2598
|
+
/** Observer clock, when the finding was raised. */
|
|
2599
|
+
timestamp: number;
|
|
2600
|
+
/**
|
|
2601
|
+
* What the finding *means* — where to look, and how much the evidence justifies it.
|
|
2602
|
+
*
|
|
2603
|
+
* A first-class field rather than a key inside {@link payload}, because it is the one part every
|
|
2604
|
+
* finding has in common and the one part an alerting rule reads. Burying it in the evidence made
|
|
2605
|
+
* `payload.conclusion.faultDomain` the path to the most important thing in the object.
|
|
2606
|
+
*/
|
|
2607
|
+
conclusion?: IssueConclusion;
|
|
2608
|
+
/**
|
|
2609
|
+
* The evidence, and **only** the evidence.
|
|
2610
|
+
*
|
|
2611
|
+
* Deliberately does not repeat `type`, `scope`, or the ids already present on the event that
|
|
2612
|
+
* delivers it. A payload that restates its envelope invites the two to disagree — and they did,
|
|
2613
|
+
* because nothing kept them in step.
|
|
2614
|
+
*/
|
|
2615
|
+
payload?: Record<string, unknown>;
|
|
2616
|
+
};
|
|
2617
|
+
/**
|
|
2618
|
+
* A finding about **one call**, raised by `observedCall.addIssue()` and delivered as `call-issue`.
|
|
2619
|
+
*
|
|
2620
|
+
* The call is the event's scope (`{ observedCall, observer }`), so the payload does not carry a
|
|
2621
|
+
* `callId` — read it from the event.
|
|
2622
|
+
*/
|
|
2623
|
+
type CallIssue = IssueBase & {
|
|
2624
|
+
scope: 'call';
|
|
2625
|
+
};
|
|
2626
|
+
/**
|
|
2627
|
+
* A finding about the **fleet**, raised by `observer.addIssue()` and delivered as `observer-issue`.
|
|
2628
|
+
*
|
|
2629
|
+
* Raised by detectors and validators that reason across calls, so no single call owns it. Where the
|
|
2630
|
+
* finding does concern specific calls — a cross-call correlation, say — they are named in the
|
|
2631
|
+
* evidence, because that *is* the evidence.
|
|
2632
|
+
*/
|
|
2633
|
+
type ObserverIssue = IssueBase & {
|
|
2634
|
+
scope: 'observer';
|
|
2635
|
+
};
|
|
2636
|
+
/**
|
|
2637
|
+
* Either kind, discriminated by {@link IssueBase} + `scope`.
|
|
2638
|
+
*
|
|
2639
|
+
* `scope` is on the issue and not merely implied by which event fired, so a finding stays
|
|
2640
|
+
* self-describing once it leaves the bus — funnelled into one handler, a log line, or a queue.
|
|
2641
|
+
*/
|
|
2642
|
+
type Issue = CallIssue | ObserverIssue;
|
|
2643
|
+
/**
|
|
2644
|
+
* The payload as a JSON string, for the boundaries that genuinely need text — a log line, an HTTP
|
|
2645
|
+
* body, a message queue.
|
|
2646
|
+
*
|
|
2647
|
+
* Returns `undefined` for a missing payload, or for one that cannot be serialised (a circular
|
|
2648
|
+
* reference from something an application attached): the caller wanted text, not an exception.
|
|
2649
|
+
*/
|
|
2650
|
+
declare function issuePayloadAsString(issue: Pick<IssueBase, 'payload'>): string | undefined;
|
|
2651
|
+
|
|
2652
|
+
/**
|
|
2653
|
+
* The bus events that carry an `observedCall`, i.e. the ones an enricher can attribute to a summary.
|
|
2654
|
+
*
|
|
2655
|
+
* Derived from the event map rather than listed by hand, so it cannot drift: adding a call-scoped
|
|
2656
|
+
* event makes it enrichable automatically, and an enricher on an observer-scoped event
|
|
2657
|
+
* (`observer-issue`, `validation-ready`) will not compile — there is no single call it belongs to,
|
|
2658
|
+
* and quietly writing a fleet-wide fact into every open summary would be worse than a type error.
|
|
2659
|
+
*/
|
|
2660
|
+
type CallScopedEventName = {
|
|
2661
|
+
[K in keyof ObserverEvents]: ObserverEvents[K][0] extends ObservedCallScope ? K : never;
|
|
2662
|
+
}[keyof ObserverEvents];
|
|
2663
|
+
/** A function that folds one event into the summary. Runs on every occurrence, for its own call. */
|
|
2664
|
+
type CallSummaryEnricher<K extends CallScopedEventName> = (summary: CallSummary, ...args: ObserverEvents[K]) => void;
|
|
2665
|
+
/** The enricher map: any subset of the call-scoped events, each fully typed against its payload. */
|
|
2666
|
+
type CallSummaryEnrichers = {
|
|
2667
|
+
[K in CallScopedEventName]?: CallSummaryEnricher<K>;
|
|
2668
|
+
};
|
|
2669
|
+
/** The built-in sections. Ask for what you want; anything absent is simply not collected. */
|
|
2670
|
+
type CallSummarySection = 'clients' | 'issues' | 'turnServers' | 'scores';
|
|
2671
|
+
type CallSummaryConfig = {
|
|
2672
|
+
/**
|
|
2673
|
+
* Which built-in sections to accumulate. Empty (the default) collects none of them — a summary
|
|
2674
|
+
* with only `enrich` is a perfectly good summary.
|
|
2675
|
+
*/
|
|
2676
|
+
include: CallSummarySection[];
|
|
2677
|
+
/**
|
|
2678
|
+
* Fold arbitrary state in from any call-scoped event, into `summary.attachments`. See
|
|
2679
|
+
* {@link CallSummaryEnrichers}.
|
|
2680
|
+
*
|
|
2681
|
+
* Each enricher runs on **every** occurrence of its event, for its own call, so prefer the
|
|
2682
|
+
* low-frequency lifecycle events (`client-joined`, `call-issue`) over `client-updated`, which fires
|
|
2683
|
+
* once per sample per client. Keep them cheap and side-effect free: an enricher that throws is logged
|
|
2684
|
+
* and skipped rather than allowed to disturb the call, but one that is slow is on the ingestion path.
|
|
2685
|
+
*/
|
|
2686
|
+
enrich?: CallSummaryEnrichers;
|
|
2687
|
+
/**
|
|
2688
|
+
* Cap on retained issues. Default `500`.
|
|
2689
|
+
*
|
|
2690
|
+
* A bound on memory per call, so scale it by how long your calls run and how noisy they are, not by
|
|
2691
|
+
* taste — a two-hour call with a struggling participant can raise hundreds. Past the cap issues are
|
|
2692
|
+
* dropped and `summary.truncated.issues` counts them, so the real total stays recoverable as
|
|
2693
|
+
* `issues.length + (truncated?.issues ?? 0)`. `0` collects the count only, keeping no issue objects.
|
|
2694
|
+
*/
|
|
2695
|
+
maxIssues: number;
|
|
2696
|
+
/**
|
|
2697
|
+
* Cap on retained client ids. Default `10_000`.
|
|
2698
|
+
*
|
|
2699
|
+
* High because the elements are short strings and the usual reason to read a summary is *who was in
|
|
2700
|
+
* this call*. It exists so a webinar-scale room cannot grow the summary without limit. Overflow is
|
|
2701
|
+
* counted in `summary.truncated.clientIds`; `joined`, `left` and `peak` are unaffected by the cap,
|
|
2702
|
+
* since they are counters rather than a list.
|
|
2703
|
+
*/
|
|
2704
|
+
maxClientIds: number;
|
|
2705
|
+
};
|
|
2706
|
+
/**
|
|
2707
|
+
* Who was in the call over its whole life — not just who is in it now.
|
|
2708
|
+
*
|
|
2709
|
+
* Deliberately identifiers and counts only. Anything *about* a client — browser, platform, region —
|
|
2710
|
+
* is already on `observedClient` while the call is live, and belongs in `attachments` via an enricher
|
|
2711
|
+
* if you want it kept. Duplicating it here would mean the library deciding which client attributes
|
|
2712
|
+
* matter, and it would mean reading the highest-frequency event on the bus to do it.
|
|
2713
|
+
*/
|
|
2714
|
+
type CallSummaryClients = {
|
|
2715
|
+
/** Every client id seen, in join order, capped by `maxClientIds`. */
|
|
2716
|
+
clientIds: string[];
|
|
2717
|
+
/** The most participants present at any one moment. */
|
|
2718
|
+
peak: number;
|
|
2719
|
+
joined: number;
|
|
2720
|
+
left: number;
|
|
2721
|
+
};
|
|
2722
|
+
/** Which TURN relays carried this call's media. */
|
|
2723
|
+
type CallSummaryTurnServers = {
|
|
2724
|
+
serverUrls: string[];
|
|
2725
|
+
/** Distinct clients seen relaying through any of them. */
|
|
2726
|
+
clientsRelayed: number;
|
|
2727
|
+
};
|
|
2728
|
+
/** The call score over time. Percentiles, not a mean — see `utils/stats`. */
|
|
2729
|
+
type CallSummaryScores = {
|
|
2730
|
+
min?: number;
|
|
2731
|
+
max?: number;
|
|
2732
|
+
median?: number;
|
|
2733
|
+
/** How many score readings went into the above. `0` means nothing was measured. */
|
|
2734
|
+
samples: number;
|
|
2735
|
+
};
|
|
2736
|
+
/** What had to be dropped to stay within the caps. Absent when nothing was. */
|
|
2737
|
+
type CallSummaryTruncation = {
|
|
2738
|
+
issues?: number;
|
|
2739
|
+
clientIds?: number;
|
|
2740
|
+
};
|
|
2741
|
+
/**
|
|
2742
|
+
* An accumulating record of one call's life, finalised when the call closes.
|
|
2743
|
+
*
|
|
2744
|
+
* ### Why this exists
|
|
2745
|
+
*
|
|
2746
|
+
* Everything else in this library is about *now*. Detectors answer "is something wrong right now",
|
|
2747
|
+
* validators answer a structural question once, and both read state that the call throws away when
|
|
2748
|
+
* it ends. Nothing kept the answer to *"what happened in that meeting?"* — who was in it, what was
|
|
2749
|
+
* raised, how it scored — and that is the question asked after the call, by support, by billing, by
|
|
2750
|
+
* whoever is writing the incident note.
|
|
2751
|
+
*
|
|
2752
|
+
* ### It is opt-in, and its sections are opt-in
|
|
2753
|
+
*
|
|
2754
|
+
* `observedCall.summary` is `undefined` unless a summary was configured, and each section is present
|
|
2755
|
+
* only if it was requested. **An absent section means "not collected", never "nothing happened"** —
|
|
2756
|
+
* the same rule as `inconclusive` on a validator. Reading `summary.issues` as "this call had no
|
|
2757
|
+
* issues" when `'issues'` was never in `include` is the one misreading this type invites, so it does
|
|
2758
|
+
* not offer a default-empty section to make it easy.
|
|
2759
|
+
*
|
|
2760
|
+
* ### Read it live, receive it once
|
|
2761
|
+
*
|
|
2762
|
+
* The object is live: read `observedCall.summary` at any point during the call. It is also delivered
|
|
2763
|
+
* on the `call-summary` event, emitted inside `close()` while the call is still reachable — after
|
|
2764
|
+
* that the call is gone from `observer.observedCalls` and there is nothing left to ask.
|
|
2765
|
+
*/
|
|
2766
|
+
type CallSummary = {
|
|
2767
|
+
callId: string;
|
|
2768
|
+
/** First client join (client clock), as `ObservedCall` computed it. */
|
|
2769
|
+
startedAt?: number;
|
|
2770
|
+
/** Last client leave. */
|
|
2771
|
+
endedAt?: number;
|
|
2772
|
+
/** `endedAt - startedAt`, when both are known. */
|
|
2773
|
+
durationInMs?: number;
|
|
2774
|
+
/** When the summary itself was finalised (observer clock). Set by `close()`. */
|
|
2775
|
+
closedAt?: number;
|
|
2776
|
+
clients?: CallSummaryClients;
|
|
2777
|
+
/**
|
|
2778
|
+
* Every issue raised against this call, in the order they were raised, capped by `maxIssues`.
|
|
2779
|
+
*
|
|
2780
|
+
* Just the issues — no derived tallies. A count is `issues.length`, a per-type count is one
|
|
2781
|
+
* `filter`, and either is cheaper to write at the call site than to keep correct here. The one
|
|
2782
|
+
* thing you cannot derive is what the cap discarded, which is why `truncated.issues` exists:
|
|
2783
|
+
* the issues actually raised is `issues.length + (truncated?.issues ?? 0)`.
|
|
2784
|
+
*/
|
|
2785
|
+
issues?: CallIssue[];
|
|
2786
|
+
turnServers?: CallSummaryTurnServers;
|
|
2787
|
+
scores?: CallSummaryScores;
|
|
2788
|
+
/**
|
|
2789
|
+
* Whatever your enrichers put here. The library never writes to it, so it cannot collide with a
|
|
2790
|
+
* section added in a future version.
|
|
2791
|
+
*
|
|
2792
|
+
* **`attachments`, not `appData`, and the distinction is load-bearing.** `appData` is the live
|
|
2793
|
+
* working state an application hangs off an entity for the entity's lifetime, and it may hold
|
|
2794
|
+
* references that cannot be serialised — a mediasoup router, an `RTCPeerConnection`, a socket. A
|
|
2795
|
+
* summary is the opposite: it outlives the call precisely so it can be *shipped* — archived,
|
|
2796
|
+
* queued, written to a column — and it is handed to you on `call-summary` at the moment the call
|
|
2797
|
+
* it came from is being torn down. Anything unserialisable in it is a reference to something
|
|
2798
|
+
* already gone.
|
|
2799
|
+
*
|
|
2800
|
+
* So put serialisable facts here, the same contract as `attachments` on a `ClientSample`. If you
|
|
2801
|
+
* need the live object, read it off `observedCall` / `observedClient` inside the enricher and
|
|
2802
|
+
* attach what you can serialise: the router's `id`, not the router.
|
|
2803
|
+
*/
|
|
2804
|
+
attachments: Record<string, unknown>;
|
|
2805
|
+
/**
|
|
2806
|
+
* What the caps discarded, and how much.
|
|
2807
|
+
*
|
|
2808
|
+
* Present **only** when something was actually dropped. A silently truncated summary is worse
|
|
2809
|
+
* than no summary — someone will count `log.length` and report it as the issue count — so the
|
|
2810
|
+
* shortfall is stated rather than left to be inferred from a suspiciously round number.
|
|
2811
|
+
*/
|
|
2812
|
+
truncated?: CallSummaryTruncation;
|
|
2813
|
+
};
|
|
2814
|
+
/** The defaults a summary is created with. Caps are generous but finite; see {@link CallSummary}. */
|
|
2815
|
+
declare const defaultCallSummaryConfig: CallSummaryConfig;
|
|
2816
|
+
/** A fresh summary for `callId`, with only the requested sections present. */
|
|
2817
|
+
declare function createCallSummary(callId: string, config: CallSummaryConfig): CallSummary;
|
|
2818
|
+
|
|
2819
|
+
/**
|
|
2820
|
+
* What a validator concluded, once it is done.
|
|
2821
|
+
*
|
|
2822
|
+
* `S` is the validator's own payload — a discriminated union on `verdict` plus whatever evidence
|
|
2823
|
+
* belongs to each outcome — so a report reads in the language of the thing being checked rather than
|
|
2824
|
+
* in generic pass/fail.
|
|
2825
|
+
*
|
|
2826
|
+
* Note this is a plain intersection with `S`, not `{ [K in keyof S]: S[K] }`. A mapped type over a
|
|
2827
|
+
* union collapses to the keys its members *share* (`keyof (A | B)` is the intersection), which would
|
|
2828
|
+
* silently drop every per-verdict evidence field and leave only `verdict` behind.
|
|
2829
|
+
*/
|
|
2830
|
+
type ValidationReport<S extends Record<string, unknown> = Record<string, unknown>> =
|
|
2831
|
+
/** Still running — it has not seen the conditions it needs to judge anything yet. */
|
|
2832
|
+
{
|
|
2833
|
+
ready: false;
|
|
2834
|
+
}
|
|
2835
|
+
/** Done. `verdict` is narrowed by `S` to that validator's own vocabulary. */
|
|
2836
|
+
| ({
|
|
2837
|
+
ready: true;
|
|
2838
|
+
verdict: string;
|
|
2839
|
+
decidedAt: number;
|
|
2840
|
+
} & S);
|
|
2841
|
+
/**
|
|
2842
|
+
* A **one-shot structural check**: something true of the deployment rather than of this moment.
|
|
2843
|
+
*
|
|
2844
|
+
* The distinction from a `Detector` is what changes over time. A detector answers "is something wrong
|
|
2845
|
+
* *right now*" — congestion, a dead relay, a track nobody receives — and the answer legitimately
|
|
2846
|
+
* differs every tick, so it runs every tick, forever. A validator answers "is this deployment built
|
|
2847
|
+
* correctly" — does the SFU pick layers per receiver — and that only changes when you deploy. So a
|
|
2848
|
+
* validator **runs until it knows, then finishes**: it calls {@link onDone} once, the observer drops
|
|
2849
|
+
* it, and nothing more is computed.
|
|
2850
|
+
*
|
|
2851
|
+
* To check again — after a deploy, say — start a new one with `observer.addValidator(...)`. There is
|
|
2852
|
+
* no revalidation timer, because a deploy, not the passage of time, is what makes a structural
|
|
2853
|
+
* verdict stale.
|
|
2854
|
+
*
|
|
2855
|
+
* Validators are held in `observer.validators` and driven by `observer.update()`.
|
|
2856
|
+
*/
|
|
2857
|
+
interface Validator<S extends Record<string, unknown> = Record<string, unknown>> {
|
|
2858
|
+
readonly name: string;
|
|
2859
|
+
/**
|
|
2860
|
+
* The conclusion so far. `{ ready: false }` until {@link onDone} fires — and **that is not a
|
|
2861
|
+
* pass**. A validator usually needs specific conditions to occur before it can judge anything,
|
|
2862
|
+
* and plenty of deployments never present them.
|
|
2863
|
+
*/
|
|
2864
|
+
readonly report: ValidationReport<S>;
|
|
2865
|
+
/** Called exactly once, when the validator finishes. The observer uses this to unregister it. */
|
|
2866
|
+
onDone: (report: ValidationReport<S>) => void;
|
|
2867
|
+
/** Gather evidence; decide if there is now enough. Called on every `observer.update()`. */
|
|
2868
|
+
update(): void;
|
|
2869
|
+
/**
|
|
2870
|
+
* Give up without a verdict. Finishes with `inconclusive`, so a caller waiting on it is freed.
|
|
2871
|
+
*
|
|
2872
|
+
* `reason` is carried into the report. Worth passing something specific — "cancelled" tells the
|
|
2873
|
+
* reader nothing, whereas "sfu redeployed" explains why a check that was running has no verdict.
|
|
2874
|
+
*/
|
|
2875
|
+
cancel: (reason?: string) => void;
|
|
2876
|
+
}
|
|
2877
|
+
/**
|
|
2878
|
+
* The part of a validator the observer needs in order to drive it.
|
|
2879
|
+
*
|
|
2880
|
+
* `Validator<S>` is invariant in `S` — `onDone` takes a `ValidationReport<S>` and `report` returns one
|
|
2881
|
+
* — so a `Set<Validator>` cannot hold validators with different payloads. Driving one only needs the
|
|
2882
|
+
* three members that don't mention `S`.
|
|
2883
|
+
*/
|
|
2884
|
+
type RunningValidator = Pick<Validator, 'name' | 'update' | 'cancel'>;
|
|
2885
|
+
|
|
2886
|
+
/** The suffix client-monitor-js appends to the type of a resolution entry. */
|
|
2887
|
+
declare const RESOLVED_ISSUE_SUFFIX = "-resolved";
|
|
2888
|
+
/**
|
|
2889
|
+
* A **stateful** client issue the server currently believes to be open.
|
|
2890
|
+
*
|
|
2891
|
+
* client-monitor-js (>= 4.6.0, `sendResolvedIssuesToServer`) puts an issue's whole lifecycle on the
|
|
2892
|
+
* wire as two entries sharing a `key`:
|
|
2893
|
+
*
|
|
2894
|
+
* ```
|
|
2895
|
+
* raise: { type: 'stuck-decoder', key, payload, timestamp: raisedAt }
|
|
2896
|
+
* resolution: { type: 'stuck-decoder-resolved', key, payload: { raisedAt, comment, …final }, timestamp: resolvedAt }
|
|
2897
|
+
* ```
|
|
2898
|
+
*
|
|
2899
|
+
* The observer opens an `ActiveClientIssue` on the raise and closes it on the matching key, which
|
|
2900
|
+
* turns a stream of point-in-time symptom reports into **intervals**. That is what makes the
|
|
2901
|
+
* difference between "several clients reported congestion in the last 10 seconds" (a guess built on
|
|
2902
|
+
* an arbitrary window) and "several clients are congested *right now, simultaneously*" — the latter
|
|
2903
|
+
* being real evidence of a shared cause.
|
|
2904
|
+
*
|
|
2905
|
+
* Issues still active when the client monitor closes are auto-resolved by the client, so a clean
|
|
2906
|
+
* departure does not leak.
|
|
2907
|
+
*/
|
|
2908
|
+
type ActiveClientIssue = {
|
|
2909
|
+
/** Identity shared by the raise and its resolution. Unique per client. */
|
|
2910
|
+
key: string;
|
|
2911
|
+
/** The issue type **without** the `-resolved` suffix (e.g. `'congestion'`). */
|
|
2912
|
+
type: string;
|
|
2913
|
+
/** The client that reported it. */
|
|
2914
|
+
clientId: string;
|
|
2915
|
+
/**
|
|
2916
|
+
* The call that client belongs to. The dimension that separates "one bad meeting" from "our
|
|
2917
|
+
* infrastructure": at observer scope, clients in *different* calls share nothing but the server.
|
|
2918
|
+
*/
|
|
2919
|
+
callId: string;
|
|
2920
|
+
/** When the client raised it (client clock). */
|
|
2921
|
+
raisedAt: number;
|
|
2922
|
+
/** When the observer first saw it (server clock) — skew-free, use this for cross-client timing. */
|
|
2923
|
+
observedAt: number;
|
|
2924
|
+
/** Parsed raise payload, when it was JSON. */
|
|
2925
|
+
payload?: Record<string, unknown>;
|
|
2926
|
+
/** `payload.peerConnectionId`, when present — most client detectors report it. */
|
|
2927
|
+
peerConnectionId?: string;
|
|
2928
|
+
/** `payload.trackId`, when present — the join key to a track (and thus to a publisher). */
|
|
2929
|
+
trackId?: string;
|
|
2930
|
+
};
|
|
2931
|
+
/** A closed interval: an {@link ActiveClientIssue} plus how it ended. */
|
|
2932
|
+
type ResolvedActiveClientIssue = ActiveClientIssue & {
|
|
2933
|
+
/** When the client resolved it (client clock). */
|
|
2934
|
+
resolvedAt: number;
|
|
2935
|
+
/** Observer-clock resolution time. */
|
|
2936
|
+
observedResolvedAt: number;
|
|
2937
|
+
/** `resolvedAt - raisedAt` as reported by the client, else derived from observer clocks. */
|
|
2938
|
+
durationInMs: number;
|
|
2939
|
+
/** Free-form note passed to `resolveIssue`. */
|
|
2940
|
+
comment?: string;
|
|
2941
|
+
/** Payload explicitly passed at resolution (the built-in detectors pass their final payload). */
|
|
2942
|
+
resolutionPayload?: Record<string, unknown>;
|
|
2943
|
+
/**
|
|
2944
|
+
* How the interval ended: the client said so, the observer expired it, or the client left
|
|
2945
|
+
* without resolving.
|
|
2946
|
+
*/
|
|
2947
|
+
resolvedBy: 'client' | 'timeout' | 'client-closed';
|
|
2948
|
+
};
|
|
2949
|
+
/** `true` when the entry is a resolution companion rather than a raise. */
|
|
2950
|
+
declare function isClientIssueResolutionEntry(issue: ClientIssue): boolean;
|
|
2951
|
+
/** Strip the `-resolved` suffix, so both entries of a lifecycle share one logical type. */
|
|
2952
|
+
declare function baseIssueType(type: string): string;
|
|
2953
|
+
|
|
2039
2954
|
/** The lifecycle events a sink may emit (a subset of Node's writable-stream events). */
|
|
2040
2955
|
type ClientSampleSinkEvents = {
|
|
2041
2956
|
/** The destination is fully written and closed (e.g. a file flushed and its fd closed). */
|
|
@@ -2103,11 +3018,11 @@ type RtpCodecParameters = {
|
|
|
2103
3018
|
parameter?: string;
|
|
2104
3019
|
}[];
|
|
2105
3020
|
};
|
|
2106
|
-
type SampleHistoryItem<T extends string> =
|
|
3021
|
+
type SampleHistoryItem<T extends string> = {
|
|
2107
3022
|
type: T;
|
|
2108
3023
|
timestamp: number;
|
|
2109
3024
|
};
|
|
2110
|
-
type MediasoupRouterSample =
|
|
3025
|
+
type MediasoupRouterSample = {
|
|
2111
3026
|
routerId: string;
|
|
2112
3027
|
attachments: Record<string, unknown>;
|
|
2113
3028
|
createdAt: number;
|
|
@@ -2151,7 +3066,9 @@ type MediasoupDirectTransportSample = {
|
|
|
2151
3066
|
type: 'direct';
|
|
2152
3067
|
history: MediasoupDirectTransportSampleEventMap[];
|
|
2153
3068
|
};
|
|
2154
|
-
type MediasoupTransportSample =
|
|
3069
|
+
type MediasoupTransportSample = {
|
|
3070
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
3071
|
+
attachments?: Record<string, unknown>;
|
|
2155
3072
|
id: string;
|
|
2156
3073
|
createdAt: number;
|
|
2157
3074
|
connectedAt?: number;
|
|
@@ -2167,7 +3084,9 @@ type MediasoupProducerSampleEventMap = {
|
|
|
2167
3084
|
type MediasoupProducerSampleEvent = {
|
|
2168
3085
|
[K in keyof MediasoupProducerSampleEventMap]: SampleHistoryItem<K>;
|
|
2169
3086
|
}[keyof MediasoupProducerSampleEventMap];
|
|
2170
|
-
type MediasoupProducerSample =
|
|
3087
|
+
type MediasoupProducerSample = {
|
|
3088
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
3089
|
+
attachments?: Record<string, unknown>;
|
|
2171
3090
|
id: string;
|
|
2172
3091
|
transportId: string;
|
|
2173
3092
|
createdAt: number;
|
|
@@ -2191,7 +3110,9 @@ type MediasoupConsumerSampleEventMap = {
|
|
|
2191
3110
|
type MediasoupConsumerSampleEvent = {
|
|
2192
3111
|
[K in keyof MediasoupConsumerSampleEventMap]: SampleHistoryItem<K>;
|
|
2193
3112
|
}[keyof MediasoupConsumerSampleEventMap];
|
|
2194
|
-
type MediasoupConsumerSample =
|
|
3113
|
+
type MediasoupConsumerSample = {
|
|
3114
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
3115
|
+
attachments?: Record<string, unknown>;
|
|
2195
3116
|
id: string;
|
|
2196
3117
|
producerId: string;
|
|
2197
3118
|
transportId: string;
|
|
@@ -2200,7 +3121,9 @@ type MediasoupConsumerSample = Record<string, unknown> & {
|
|
|
2200
3121
|
kind: 'audio' | 'video';
|
|
2201
3122
|
history: MediasoupConsumerSampleEvent[];
|
|
2202
3123
|
};
|
|
2203
|
-
type MediasoupDataProducerSample =
|
|
3124
|
+
type MediasoupDataProducerSample = {
|
|
3125
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
3126
|
+
attachments?: Record<string, unknown>;
|
|
2204
3127
|
id: string;
|
|
2205
3128
|
transportId: string;
|
|
2206
3129
|
createdAt: number;
|
|
@@ -2208,7 +3131,9 @@ type MediasoupDataProducerSample = Record<string, unknown> & {
|
|
|
2208
3131
|
label: string;
|
|
2209
3132
|
protocol: string;
|
|
2210
3133
|
};
|
|
2211
|
-
type MediasoupDataConsumerSample =
|
|
3134
|
+
type MediasoupDataConsumerSample = {
|
|
3135
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
3136
|
+
attachments?: Record<string, unknown>;
|
|
2212
3137
|
id: string;
|
|
2213
3138
|
dataProducerId: string;
|
|
2214
3139
|
transportId: string;
|
|
@@ -2218,30 +3143,142 @@ type MediasoupDataConsumerSample = Record<string, unknown> & {
|
|
|
2218
3143
|
protocol: string;
|
|
2219
3144
|
};
|
|
2220
3145
|
|
|
3146
|
+
/**
|
|
3147
|
+
* Declarative enrichment: return the `attachments` to stamp onto an entity's sample the moment it is
|
|
3148
|
+
* created. Called once per entity, before the corresponding `*-sample-added` event.
|
|
3149
|
+
*
|
|
3150
|
+
* The mediasoup object is handed in, so the common case — mirroring mediasoup's own `appData`, where
|
|
3151
|
+
* applications already keep `participantId`, `purpose` and friends — is a one-liner. Returning
|
|
3152
|
+
* `undefined` attaches nothing.
|
|
3153
|
+
*/
|
|
3154
|
+
type MediasoupSampleEnricher = {
|
|
3155
|
+
transport?: (transport: types.Transport) => Record<string, unknown> | undefined;
|
|
3156
|
+
producer?: (producer: types.Producer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
3157
|
+
consumer?: (consumer: types.Consumer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
3158
|
+
dataProducer?: (dataProducer: types.DataProducer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
3159
|
+
dataConsumer?: (dataConsumer: types.DataConsumer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
3160
|
+
};
|
|
2221
3161
|
type ObservedMediasoupRouterSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
2222
|
-
routerId: string;
|
|
2223
3162
|
router: types.Router;
|
|
2224
3163
|
appData?: AppData;
|
|
2225
3164
|
attachments?: Record<string, unknown>;
|
|
3165
|
+
/** Stamp `attachments` onto each entity sample as it is created. See {@link MediasoupSampleEnricher}. */
|
|
3166
|
+
enrich?: MediasoupSampleEnricher;
|
|
2226
3167
|
};
|
|
3168
|
+
/**
|
|
3169
|
+
* Lifecycle hooks for building your own report.
|
|
3170
|
+
*
|
|
3171
|
+
* Each entity announces itself as `<entity>-sample-added` when it appears and
|
|
3172
|
+
* `<entity>-sample-closed` when it goes away, carrying **the live sample object** plus the mediasoup
|
|
3173
|
+
* object it came from. Mutating `sample.attachments` inside a handler is the intended way to extend
|
|
3174
|
+
* a sample on the fly — the object you receive is the one held in `observedRouter.sample`, not a copy.
|
|
3175
|
+
*/
|
|
2227
3176
|
type ObservedMediasoupRouterEvents = {
|
|
2228
3177
|
close: [];
|
|
2229
|
-
|
|
3178
|
+
'transport-sample-added': [{
|
|
3179
|
+
sample: MediasoupTransportSample;
|
|
3180
|
+
transport: types.Transport;
|
|
3181
|
+
}];
|
|
3182
|
+
'transport-sample-closed': [{
|
|
3183
|
+
sample: MediasoupTransportSample;
|
|
3184
|
+
transport: types.Transport;
|
|
3185
|
+
}];
|
|
3186
|
+
'producer-sample-added': [{
|
|
3187
|
+
sample: MediasoupProducerSample;
|
|
3188
|
+
producer: types.Producer;
|
|
3189
|
+
transport: types.Transport;
|
|
3190
|
+
}];
|
|
3191
|
+
'producer-sample-closed': [{
|
|
3192
|
+
sample: MediasoupProducerSample;
|
|
3193
|
+
producer: types.Producer;
|
|
3194
|
+
transport: types.Transport;
|
|
3195
|
+
}];
|
|
3196
|
+
'consumer-sample-added': [{
|
|
3197
|
+
sample: MediasoupConsumerSample;
|
|
3198
|
+
consumer: types.Consumer;
|
|
3199
|
+
transport: types.Transport;
|
|
3200
|
+
}];
|
|
3201
|
+
'consumer-sample-closed': [{
|
|
3202
|
+
sample: MediasoupConsumerSample;
|
|
3203
|
+
consumer: types.Consumer;
|
|
3204
|
+
transport: types.Transport;
|
|
3205
|
+
}];
|
|
3206
|
+
'data-producer-sample-added': [{
|
|
3207
|
+
sample: MediasoupDataProducerSample;
|
|
3208
|
+
dataProducer: types.DataProducer;
|
|
3209
|
+
transport: types.Transport;
|
|
3210
|
+
}];
|
|
3211
|
+
'data-producer-sample-closed': [{
|
|
3212
|
+
sample: MediasoupDataProducerSample;
|
|
3213
|
+
dataProducer: types.DataProducer;
|
|
3214
|
+
transport: types.Transport;
|
|
3215
|
+
}];
|
|
3216
|
+
'data-consumer-sample-added': [{
|
|
3217
|
+
sample: MediasoupDataConsumerSample;
|
|
3218
|
+
dataConsumer: types.DataConsumer;
|
|
3219
|
+
transport: types.Transport;
|
|
3220
|
+
}];
|
|
3221
|
+
'data-consumer-sample-closed': [{
|
|
3222
|
+
sample: MediasoupDataConsumerSample;
|
|
3223
|
+
dataConsumer: types.DataConsumer;
|
|
3224
|
+
transport: types.Transport;
|
|
3225
|
+
}];
|
|
3226
|
+
};
|
|
2230
3227
|
declare interface ObservedMediasoupRouter {
|
|
2231
3228
|
on<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
|
|
2232
3229
|
off<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
|
|
2233
3230
|
once<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
|
|
2234
3231
|
emit<U extends keyof ObservedMediasoupRouterEvents>(event: U, ...args: ObservedMediasoupRouterEvents[U]): boolean;
|
|
2235
3232
|
}
|
|
3233
|
+
/**
|
|
3234
|
+
* Observes a live mediasoup `Router` by subscribing to its `observer` API and **accumulates** its
|
|
3235
|
+
* topology and lifecycle into an in-memory `MediasoupRouterSample` (`observedRouter.sample`):
|
|
3236
|
+
* transports, producers, consumers, data producers/consumers, their state-change history and
|
|
3237
|
+
* `createdAt` / `closedAt`. The sample grows for the life of the router (closed entities are kept,
|
|
3238
|
+
* with their `closedAt` set) and is yours to read, snapshot, or persist.
|
|
3239
|
+
*
|
|
3240
|
+
* NOTE: this is intentionally the simplest approach — everything is held in memory. For very large
|
|
3241
|
+
* routers (e.g. ~100 participants producing and consuming on one router, where consumers grow as
|
|
3242
|
+
* O(N²)) this can become substantial; in that case do your own periodic sampling/persistence and
|
|
3243
|
+
* discard what you don't need (see the README's "Memory & large meetings" note).
|
|
3244
|
+
*/
|
|
2236
3245
|
declare class ObservedMediasoupRouter<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
|
|
2237
3246
|
readonly router: types.Router;
|
|
2238
|
-
readonly sample: MediasoupRouterSample;
|
|
2239
3247
|
appData: AppData;
|
|
3248
|
+
readonly sample: MediasoupRouterSample;
|
|
2240
3249
|
readonly webrtcTransportIds: Set<string>;
|
|
2241
|
-
get attachments(): Record<string, unknown>;
|
|
2242
3250
|
closed: boolean;
|
|
3251
|
+
private readonly _transportSamples;
|
|
3252
|
+
private readonly _producerSamples;
|
|
3253
|
+
private readonly _consumerSamples;
|
|
3254
|
+
private readonly _dataProducerSamples;
|
|
3255
|
+
private readonly _dataConsumerSamples;
|
|
3256
|
+
private readonly _enrich?;
|
|
2243
3257
|
constructor(settings: ObservedMediasoupRouterSettings<AppData>);
|
|
2244
3258
|
get id(): string;
|
|
3259
|
+
get attachments(): Record<string, unknown>;
|
|
3260
|
+
getTransportSample(id: string): MediasoupTransportSample | undefined;
|
|
3261
|
+
getProducerSample(id: string): MediasoupProducerSample | undefined;
|
|
3262
|
+
getConsumerSample(id: string): MediasoupConsumerSample | undefined;
|
|
3263
|
+
getDataProducerSample(id: string): MediasoupDataProducerSample | undefined;
|
|
3264
|
+
getDataConsumerSample(id: string): MediasoupDataConsumerSample | undefined;
|
|
3265
|
+
/**
|
|
3266
|
+
* Merge `attachments` into an entity's sample, whichever kind it is.
|
|
3267
|
+
*
|
|
3268
|
+
* Ids are unique across mediasoup entity kinds, so one method covers all of them. Returns `false`
|
|
3269
|
+
* when the id is unknown — a real answer instead of failing quietly, which matters when the
|
|
3270
|
+
* annotation is driven by application events that may race the mediasoup ones.
|
|
3271
|
+
*/
|
|
3272
|
+
attachTo(id: string, attachments: Record<string, unknown>): boolean;
|
|
3273
|
+
/**
|
|
3274
|
+
* A **detached deep copy** of the current sample — the basis for building your own report.
|
|
3275
|
+
*
|
|
3276
|
+
* `this.sample` is live: its arrays grow and its `history` entries are appended as the router
|
|
3277
|
+
* runs, so a report built directly on it keeps changing after you think you're done. This returns
|
|
3278
|
+
* a snapshot that never moves.
|
|
3279
|
+
*/
|
|
3280
|
+
snapshot(): MediasoupRouterSample;
|
|
3281
|
+
close(): void;
|
|
2245
3282
|
addTransport: (transport: types.Transport) => void;
|
|
2246
3283
|
addWebRtcTransport(transport: types.WebRtcTransport): void;
|
|
2247
3284
|
addPlainTransport(transport: types.PlainTransport): void;
|
|
@@ -2251,8 +3288,19 @@ declare class ObservedMediasoupRouter<AppData extends Record<string, unknown> =
|
|
|
2251
3288
|
addConsumer(transport: types.Transport, consumer: types.Consumer): void;
|
|
2252
3289
|
addDataProducer(transport: types.Transport, dataProducer: types.DataProducer): void;
|
|
2253
3290
|
addDataConsumer(transport: types.Transport, dataConsumer: types.DataConsumer): void;
|
|
2254
|
-
close(): void;
|
|
2255
3291
|
private attachRouterListeners;
|
|
3292
|
+
private _addTransportSample;
|
|
3293
|
+
private _addProducerSample;
|
|
3294
|
+
private _addConsumerSample;
|
|
3295
|
+
private _addDataProducerSample;
|
|
3296
|
+
private _addDataConsumerSample;
|
|
3297
|
+
/**
|
|
3298
|
+
* Run an enricher and merge what it returns.
|
|
3299
|
+
*
|
|
3300
|
+
* Takes a thunk rather than a value so the **invocation** is inside the guard — application code
|
|
3301
|
+
* runs here, and a throwing enricher must not take the router's bookkeeping down with it.
|
|
3302
|
+
*/
|
|
3303
|
+
private _applyEnrichment;
|
|
2256
3304
|
private _attachTransportObserverListeners;
|
|
2257
3305
|
}
|
|
2258
3306
|
|
|
@@ -2293,16 +3341,36 @@ type ObserverEvents = {
|
|
|
2293
3341
|
reason: SampleRejectedReason;
|
|
2294
3342
|
sample: ClientSample;
|
|
2295
3343
|
}];
|
|
3344
|
+
/** An observer-scoped (cross-call / SFU-wide) finding raised by `observer.addIssue(...)`. */
|
|
3345
|
+
'observer-issue': [ObserverEventBase & {
|
|
3346
|
+
issue: ObserverIssue;
|
|
3347
|
+
}];
|
|
3348
|
+
/**
|
|
3349
|
+
* A validator decided. Fires once per settle, not per tick — the point of a validator is that it
|
|
3350
|
+
* stops talking once it knows.
|
|
3351
|
+
*/
|
|
3352
|
+
'validation-ready': [ObserverEventBase & {
|
|
3353
|
+
validator: string;
|
|
3354
|
+
report: ValidationReport;
|
|
3355
|
+
}];
|
|
2296
3356
|
'mediasoup-router-added': [ObservedMediasoupRouterScope];
|
|
2297
3357
|
'mediasoup-router-removed': [ObservedMediasoupRouterScope];
|
|
2298
|
-
'mediasoup-router-matched-with-
|
|
3358
|
+
'mediasoup-router-matched-with-peer-connection': [ObservedMediasoupRouterScope & ObservedPeerConnectionScope];
|
|
2299
3359
|
'call-added': [ObservedCallScope];
|
|
2300
3360
|
'call-updated': [ObservedCallScope];
|
|
2301
3361
|
'call-closed': [ObservedCallScope];
|
|
2302
3362
|
'call-empty': [ObservedCallScope];
|
|
2303
3363
|
'call-not-empty': [ObservedCallScope];
|
|
2304
3364
|
'call-issue': [ObservedCallScope & {
|
|
2305
|
-
issue:
|
|
3365
|
+
issue: CallIssue;
|
|
3366
|
+
}];
|
|
3367
|
+
/**
|
|
3368
|
+
* A call's summary was finalised. Emitted from inside `close()`, while the call is still in
|
|
3369
|
+
* `observer.observedCalls` — after that there is nothing left to ask. Only fires for calls that
|
|
3370
|
+
* had a summary configured.
|
|
3371
|
+
*/
|
|
3372
|
+
'call-summary': [ObservedCallScope & {
|
|
3373
|
+
summary: CallSummary;
|
|
2306
3374
|
}];
|
|
2307
3375
|
'client-added': [ObservedClientScope];
|
|
2308
3376
|
'client-sink-created': [ObservedClientScope & {
|
|
@@ -2321,6 +3389,14 @@ type ObserverEvents = {
|
|
|
2321
3389
|
'client-issue': [ObservedClientScope & {
|
|
2322
3390
|
issue: ClientIssue;
|
|
2323
3391
|
}];
|
|
3392
|
+
/**
|
|
3393
|
+
* A stateful client issue ended — the client sent its `<type>-resolved` companion, or the
|
|
3394
|
+
* observer force-closed it because the client went away. Carries the finished **interval**
|
|
3395
|
+
* (`raisedAt` → `resolvedAt`, `durationInMs`).
|
|
3396
|
+
*/
|
|
3397
|
+
'client-issue-resolved': [ObservedClientScope & {
|
|
3398
|
+
resolvedIssue: ResolvedActiveClientIssue;
|
|
3399
|
+
}];
|
|
2324
3400
|
'client-metadata': [ObservedClientScope & {
|
|
2325
3401
|
metaData: ClientMetaData;
|
|
2326
3402
|
}];
|
|
@@ -2492,6 +3568,61 @@ type ObserverEvents = {
|
|
|
2492
3568
|
}];
|
|
2493
3569
|
};
|
|
2494
3570
|
|
|
3571
|
+
/**
|
|
3572
|
+
* Something that wants to be **handed** open client issues rather than to go looking for them.
|
|
3573
|
+
*
|
|
3574
|
+
* Register one with `activeIssuesRegistry.addIssueTracker(type, tracker)` and it receives every
|
|
3575
|
+
* issue of that type when it opens ({@link add}) and when it closes ({@link delete}). A detector
|
|
3576
|
+
* implementing this pays only for the issues it actually consumes.
|
|
3577
|
+
*
|
|
3578
|
+
* `ActiveIssuesRegistry` implements it too, which is how a call's registry feeds the observer's.
|
|
3579
|
+
*/
|
|
3580
|
+
interface ActiveIssueTracker {
|
|
3581
|
+
/** An issue of a subscribed type opened. */
|
|
3582
|
+
add(issue: ActiveClientIssue): void;
|
|
3583
|
+
/**
|
|
3584
|
+
* An issue this tracker was given has closed.
|
|
3585
|
+
*
|
|
3586
|
+
* Return `true` if it was actually held. Returning `false` is legitimate and not an error — a
|
|
3587
|
+
* tracker that counts *occurrences* (see `SfuCongestionDetector`) deliberately ignores
|
|
3588
|
+
* resolutions, because when a symptom ended says nothing about how many endpoints reported it.
|
|
3589
|
+
*/
|
|
3590
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3591
|
+
/** How many issues this tracker currently holds. */
|
|
3592
|
+
size: number;
|
|
3593
|
+
/** Drop everything. Called when the owning scope closes. */
|
|
3594
|
+
clear(): void;
|
|
3595
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3596
|
+
}
|
|
3597
|
+
|
|
3598
|
+
/**
|
|
3599
|
+
* One client's currently **open** stateful issues, keyed by `ClientIssue.key`.
|
|
3600
|
+
*
|
|
3601
|
+
* The server-side mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0):
|
|
3602
|
+
* a raise opens an entry, the matching `<type>-resolved` closes it, and the client's own close
|
|
3603
|
+
* force-resolves whatever is left. That turns point-in-time symptom reports into **intervals**,
|
|
3604
|
+
* which is what lets detectors ask "are these clients broken *at the same time*" rather than "did
|
|
3605
|
+
* they both report something recently".
|
|
3606
|
+
*
|
|
3607
|
+
* Keyed by `key` rather than by type on purpose: one client can have several issues of the same type
|
|
3608
|
+
* open at once (one per track), and they resolve independently.
|
|
3609
|
+
*
|
|
3610
|
+
* Every change is forwarded to the call's `ActiveIssuesRegistry`, which forwards to the observer's —
|
|
3611
|
+
* so the client owns the storage and the wider scopes get their views maintained as it happens.
|
|
3612
|
+
*/
|
|
3613
|
+
declare class ObservedClientIssueRegistry {
|
|
3614
|
+
private readonly registry?;
|
|
3615
|
+
private readonly issues;
|
|
3616
|
+
constructor(registry?: ActiveIssueTracker | undefined);
|
|
3617
|
+
get size(): number;
|
|
3618
|
+
keys(): IterableIterator<string>;
|
|
3619
|
+
values(): IterableIterator<ActiveClientIssue>;
|
|
3620
|
+
get(key: string): ActiveClientIssue | undefined;
|
|
3621
|
+
add(issue: ActiveClientIssue): this;
|
|
3622
|
+
remove(key: string): ActiveClientIssue | undefined;
|
|
3623
|
+
clear(): void;
|
|
3624
|
+
}
|
|
3625
|
+
|
|
2495
3626
|
type ObservedClientSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
2496
3627
|
clientId: string;
|
|
2497
3628
|
appData?: AppData;
|
|
@@ -2570,10 +3701,20 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
|
|
|
2570
3701
|
totalScoreSum: number;
|
|
2571
3702
|
numberOfScoreMeasurements: number;
|
|
2572
3703
|
readonly mediaDevices: MediaDeviceInfo[];
|
|
2573
|
-
|
|
2574
|
-
|
|
3704
|
+
/**
|
|
3705
|
+
* The client's currently **open** stateful issues, keyed by `ClientIssue.key` — the server-side
|
|
3706
|
+
* mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0).
|
|
3707
|
+
*
|
|
3708
|
+
* This turns point-in-time symptom reports into intervals, which is what lets detectors ask
|
|
3709
|
+
* "are these clients broken *at the same time*" instead of "did they both report something
|
|
3710
|
+
* recently". Entries are opened by a raise, closed by the matching `<type>-resolved` entry, and
|
|
3711
|
+
* force-closed when the client closes.
|
|
3712
|
+
*/
|
|
3713
|
+
readonly activeIssues: ObservedClientIssueRegistry;
|
|
3714
|
+
private _pendingInjections;
|
|
3715
|
+
private _activeSample?;
|
|
2575
3716
|
private closeTimer?;
|
|
2576
|
-
constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall);
|
|
3717
|
+
constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall, activeIssues: ObservedClientIssueRegistry);
|
|
2577
3718
|
get numberOfPeerConnections(): number;
|
|
2578
3719
|
get score(): number | undefined;
|
|
2579
3720
|
close(): void;
|
|
@@ -2582,13 +3723,28 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
|
|
|
2582
3723
|
injectEvent(event: ClientEvent): void;
|
|
2583
3724
|
injectIssue(issue: ClientIssue): void;
|
|
2584
3725
|
injectExtensionStat(stat: ExtensionStat): void;
|
|
2585
|
-
injectAttachment(
|
|
3726
|
+
injectAttachment(attachments: Record<string, unknown>): void;
|
|
2586
3727
|
addMetadata(metadata: ClientMetaData): void;
|
|
3728
|
+
/**
|
|
3729
|
+
* Process one `clientIssues[]` entry.
|
|
3730
|
+
*
|
|
3731
|
+
* Entries come in two flavours (client-monitor-js >= 4.6.0):
|
|
3732
|
+
*
|
|
3733
|
+
* - a **raise** — opens an {@link ActiveClientIssue} under `issue.key` and emits `client-issue`;
|
|
3734
|
+
* - a **resolution** — `type` ends in `-resolved` and carries the same `key`; it closes the
|
|
3735
|
+
* matching active issue and emits `client-issue-resolved`.
|
|
3736
|
+
*
|
|
3737
|
+
* Keyless entries are one-shot: reported, never tracked. A re-raise of a key already active
|
|
3738
|
+
* refreshes the payload rather than opening a second interval.
|
|
3739
|
+
*/
|
|
2587
3740
|
addIssue(issue: ClientIssue): void;
|
|
2588
3741
|
addExtensionStats(stats: ExtensionStat): void;
|
|
3742
|
+
/** Close the active issue a `<type>-resolved` entry refers to, and announce the finished interval. */
|
|
3743
|
+
private _resolveIssue;
|
|
2589
3744
|
private _processClientEvent;
|
|
2590
3745
|
private _updatePeerConnection;
|
|
2591
|
-
private
|
|
3746
|
+
private _mergePendingInjections;
|
|
3747
|
+
private _flushPendingInjections;
|
|
2592
3748
|
/** Emit an Observer-bus event scoped to this client (or a peer connection under it). */
|
|
2593
3749
|
private _notify;
|
|
2594
3750
|
}
|
|
@@ -2629,7 +3785,29 @@ declare class RemoteTrackResolver {
|
|
|
2629
3785
|
private readonly resolvers;
|
|
2630
3786
|
private readonly _publisherIdToOutboundTrack;
|
|
2631
3787
|
private readonly _subscriberIdToInboundTrack;
|
|
3788
|
+
/**
|
|
3789
|
+
* Tracks whose publisher id the strategy could not resolve **yet**.
|
|
3790
|
+
*
|
|
3791
|
+
* A track announces itself once, but its `attachments` are replaced on every sample, so a key that
|
|
3792
|
+
* is missing from the first sample can appear on the second — and a strategy backed by an
|
|
3793
|
+
* application's own mapping (a server-side `ssrc -> producerId` table, say) is inherently racy
|
|
3794
|
+
* against sample arrival. Resolving only at `*-track-added` meant losing those tracks for their
|
|
3795
|
+
* entire life, silently: an unresolvable outbound track never even reaches
|
|
3796
|
+
* `unconsumedOutboundTracks`, so it is invisible to `UnconsumedTrackDetector` too.
|
|
3797
|
+
*
|
|
3798
|
+
* So they wait here and are retried on their own `*-track-updated`, i.e. exactly when new stats
|
|
3799
|
+
* arrived for them. A track leaves on its first successful resolution or on removal, which makes
|
|
3800
|
+
* a linked track cost one `Set.has` per update and bounds these sets by the unresolved tracks
|
|
3801
|
+
* alive right now.
|
|
3802
|
+
*/
|
|
3803
|
+
private readonly _pendingInboundTracks;
|
|
3804
|
+
private readonly _pendingOutboundTracks;
|
|
2632
3805
|
constructor(observedCall: ObservedCall, resolvers: RemoteTrackResolvers);
|
|
3806
|
+
/** Tracks still waiting for a resolvable publisher id. Diagnostics; normally both are empty. */
|
|
3807
|
+
get pendingTrackCounts(): {
|
|
3808
|
+
inbound: number;
|
|
3809
|
+
outbound: number;
|
|
3810
|
+
};
|
|
2633
3811
|
/** The published (outbound) track for a publisher id, if any. */
|
|
2634
3812
|
getOutboundTrackByPublisherId(publisherId: string): ObservedOutboundTrack | undefined;
|
|
2635
3813
|
/** The subscribed (inbound) track for a subscriber id, if the strategy resolves subscriber ids. */
|
|
@@ -2642,184 +3820,2398 @@ declare class RemoteTrackResolver {
|
|
|
2642
3820
|
private _removeOutboundTrack;
|
|
2643
3821
|
}
|
|
2644
3822
|
|
|
2645
|
-
interface Updater {
|
|
2646
|
-
readonly name: string;
|
|
2647
|
-
readonly description?: string;
|
|
2648
|
-
close(): void;
|
|
2649
|
-
}
|
|
2650
|
-
|
|
2651
3823
|
interface Detector {
|
|
2652
3824
|
readonly name: string;
|
|
2653
3825
|
/** Called on every entity update; may raise issues via the entity it observes. */
|
|
2654
3826
|
update(): void;
|
|
3827
|
+
/**
|
|
3828
|
+
* Optional teardown, called when the detector is removed from its registry (or the registry is
|
|
3829
|
+
* cleared, which happens when the owning call/observer closes). Implement it when the detector
|
|
3830
|
+
* subscribes to events or holds timers, so it doesn't leak.
|
|
3831
|
+
*/
|
|
3832
|
+
close?(): void;
|
|
2655
3833
|
}
|
|
2656
3834
|
|
|
2657
|
-
declare
|
|
2658
|
-
|
|
2659
|
-
|
|
2660
|
-
|
|
2661
|
-
|
|
2662
|
-
remove(detector: Detector): void;
|
|
2663
|
-
update(): void;
|
|
2664
|
-
clear(): void;
|
|
2665
|
-
}
|
|
2666
|
-
|
|
2667
|
-
type ObservedCallUpdateConfig = {
|
|
2668
|
-
updatePolicy?: 'update-on-any-client-updated' | 'update-when-all-client-updated' | 'none';
|
|
3835
|
+
declare const CallConcurrentIssueTypes: {
|
|
3836
|
+
/** Several participants of this call have the same issue open **at the same time**. */
|
|
3837
|
+
readonly concurrentClientIssues: "CONCURRENT_CLIENT_ISSUES";
|
|
3838
|
+
/** Those issues also *began* together — the signature of one shared event. */
|
|
3839
|
+
readonly issueOnsetBurst: "ISSUE_ONSET_BURST";
|
|
2669
3840
|
};
|
|
2670
|
-
type
|
|
2671
|
-
|
|
2672
|
-
|
|
2673
|
-
|
|
3841
|
+
type CallConcurrentIssueDetectorConfig = {
|
|
3842
|
+
/**
|
|
3843
|
+
* The issue types to watch. **Required, and must not be empty** — the detector subscribes to
|
|
3844
|
+
* exactly these and sees nothing else.
|
|
3845
|
+
*/
|
|
3846
|
+
issueTypes: string[];
|
|
3847
|
+
/**
|
|
3848
|
+
* Participants the call needs before a ratio means anything. Default `3`.
|
|
3849
|
+
*
|
|
3850
|
+
* In a 1:1 call "half the participants" is one person, which is a client problem and not a call
|
|
3851
|
+
* problem — `2` effectively disables the ratio gate. Raise it for large-meeting products where you
|
|
3852
|
+
* only care once a handful are affected.
|
|
3853
|
+
*/
|
|
3854
|
+
minClients: number;
|
|
3855
|
+
/**
|
|
3856
|
+
* Distinct clients that must share the issue. Default `3`.
|
|
3857
|
+
*
|
|
3858
|
+
* The absolute floor under `affectedRatioThreshold`, so a small call cannot clear a ratio with two
|
|
3859
|
+
* unlucky people. Sensible range `2`–`5`; `2` is the lowest that can still mean "more than one
|
|
3860
|
+
* participant", which is the whole premise.
|
|
3861
|
+
*/
|
|
3862
|
+
minAffectedClients: number;
|
|
3863
|
+
/**
|
|
3864
|
+
* Fraction of the call's participants that must share it, `0`–`1`. Default `0.5`.
|
|
3865
|
+
*
|
|
3866
|
+
* Typical `0.3`–`0.7`. Lower catches partial events — a subset on one SFU worker — at the cost of
|
|
3867
|
+
* firing on a few coincidentally unhappy participants; `1` demands literally everyone, which real
|
|
3868
|
+
* incidents rarely produce because someone always reconnects first.
|
|
3869
|
+
*/
|
|
3870
|
+
affectedRatioThreshold: number;
|
|
3871
|
+
/**
|
|
3872
|
+
* Onsets falling within this span (ms) escalate the finding to `ISSUE_ONSET_BURST` — they did not
|
|
3873
|
+
* just overlap, they started together. Default `2_000`.
|
|
3874
|
+
*
|
|
3875
|
+
* Bound this by your sampling period, not below it: onsets are only known as accurately as clients
|
|
3876
|
+
* report them, so a window shorter than one sampling period can only fire by luck. Typical
|
|
3877
|
+
* `1_000`–`5_000`. Wider makes the escalation meaningless, since unrelated issues drift into the
|
|
3878
|
+
* same window.
|
|
3879
|
+
*/
|
|
3880
|
+
onsetBurstWindowInMs: number;
|
|
3881
|
+
/**
|
|
3882
|
+
* Re-arm time per issue type (ms). Default `60_000`.
|
|
3883
|
+
*
|
|
3884
|
+
* A shared event is one incident, not one per tick. Too low and a persistent problem raises an
|
|
3885
|
+
* issue every tick for as long as it lasts; too high and a genuinely new occurrence is swallowed
|
|
3886
|
+
* by the previous one's cooldown. Typical `30_000`–`300_000`.
|
|
3887
|
+
*/
|
|
3888
|
+
cooldownMs: number;
|
|
2674
3889
|
};
|
|
2675
|
-
type
|
|
2676
|
-
|
|
2677
|
-
|
|
2678
|
-
|
|
2679
|
-
|
|
2680
|
-
|
|
3890
|
+
/** What the detector currently knows about one issue type in this call. */
|
|
3891
|
+
type CallConcurrentIssueGroup = {
|
|
3892
|
+
type: string;
|
|
3893
|
+
issues: ActiveClientIssue[];
|
|
3894
|
+
clientIds: string[];
|
|
3895
|
+
affectedRatio: number;
|
|
3896
|
+
totalClients: number;
|
|
3897
|
+
/**
|
|
3898
|
+
* Spread of the onsets, in **observer** time (ms) — `max(observedAt) - min(observedAt)`.
|
|
3899
|
+
*
|
|
3900
|
+
* Measured on the observer clock on purpose: `raisedAt` comes from each client's own clock, and
|
|
3901
|
+
* comparing those across machines makes clock skew look like a shared event.
|
|
3902
|
+
*/
|
|
3903
|
+
onsetSpreadInMs: number;
|
|
3904
|
+
firstObservedAt: number;
|
|
2681
3905
|
};
|
|
2682
|
-
|
|
2683
|
-
|
|
2684
|
-
|
|
2685
|
-
|
|
2686
|
-
|
|
2687
|
-
|
|
2688
|
-
|
|
2689
|
-
|
|
2690
|
-
|
|
2691
|
-
|
|
2692
|
-
|
|
2693
|
-
|
|
2694
|
-
|
|
2695
|
-
|
|
2696
|
-
|
|
2697
|
-
|
|
2698
|
-
|
|
2699
|
-
|
|
2700
|
-
|
|
2701
|
-
|
|
2702
|
-
|
|
2703
|
-
|
|
2704
|
-
|
|
2705
|
-
|
|
2706
|
-
|
|
2707
|
-
|
|
2708
|
-
|
|
2709
|
-
|
|
2710
|
-
|
|
2711
|
-
|
|
2712
|
-
readonly
|
|
2713
|
-
|
|
2714
|
-
|
|
2715
|
-
|
|
2716
|
-
constructor(
|
|
2717
|
-
get
|
|
2718
|
-
|
|
2719
|
-
|
|
2720
|
-
|
|
3906
|
+
/**
|
|
3907
|
+
* Answers **"is this meeting in trouble?"** — several participants of one call with the same issue
|
|
3908
|
+
* open simultaneously.
|
|
3909
|
+
*
|
|
3910
|
+
* The client already decides *what* is wrong for itself — `congestion`, `ice-disconnected`,
|
|
3911
|
+
* `audio-concealment`, `video-decoder-overloaded` — with hysteresis and multi-signal confirmation
|
|
3912
|
+
* behind each verdict. Re-deriving those server-side from raw counters would be strictly worse. What
|
|
3913
|
+
* the server uniquely knows is *how many other participants of the same call are in that state right
|
|
3914
|
+
* now*, which is the difference between "one person's Wi-Fi" and "this room is broken".
|
|
3915
|
+
*
|
|
3916
|
+
* Concurrency is judged from the **open interval set**, not a window of recent reports. A window has
|
|
3917
|
+
* to guess whether a symptom is still happening; an interval is closed by the client when the episode
|
|
3918
|
+
* actually ends (client-monitor-js >= 4.6.0 ships the `<type>-resolved` companion for exactly this).
|
|
3919
|
+
*
|
|
3920
|
+
* ```ts
|
|
3921
|
+
* observedCall.addDetector('call-concurrent-issue-detector', {
|
|
3922
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
3923
|
+
* });
|
|
3924
|
+
* ```
|
|
3925
|
+
*
|
|
3926
|
+
* For the cross-call version of this question — which is a different question, not this one with a
|
|
3927
|
+
* bigger denominator — see `ObserverConcurrentIssueDetector`.
|
|
3928
|
+
*/
|
|
3929
|
+
declare class CallConcurrentIssueDetector implements Detector, ActiveIssueTracker {
|
|
3930
|
+
private readonly _call;
|
|
3931
|
+
static readonly NAME: "call-concurrent-issue-detector";
|
|
3932
|
+
readonly name: "call-concurrent-issue-detector";
|
|
3933
|
+
private readonly _config;
|
|
3934
|
+
private readonly _lastRaisedAt;
|
|
3935
|
+
/** issue type -> the issues of that type currently open in this call. */
|
|
3936
|
+
private readonly _byType;
|
|
3937
|
+
private _size;
|
|
3938
|
+
/** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
|
|
3939
|
+
lastGroups: CallConcurrentIssueGroup[];
|
|
3940
|
+
constructor(_call: ObservedCall, config?: Partial<CallConcurrentIssueDetectorConfig>);
|
|
3941
|
+
get size(): number;
|
|
3942
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3943
|
+
add(issue: ActiveClientIssue): void;
|
|
3944
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3945
|
+
clear(): void;
|
|
3946
|
+
update(): void;
|
|
2721
3947
|
close(): void;
|
|
2722
|
-
|
|
2723
|
-
createObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>): ObservedClient<ClientAppData> | undefined;
|
|
2724
|
-
getOrCreateObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>): ObservedClient<ClientAppData> | undefined;
|
|
2725
|
-
update(context?: AcceptContext): void;
|
|
2726
|
-
private _onClientUpdate;
|
|
2727
|
-
private _clientJoined;
|
|
2728
|
-
private _clientLeft;
|
|
2729
|
-
/** Emit an Observer-bus event scoped to this call. */
|
|
2730
|
-
private _notify;
|
|
2731
|
-
}
|
|
2732
|
-
|
|
2733
|
-
type Middleware<T> = (input: T, next: (nextInput: T) => void) => void;
|
|
2734
|
-
interface Processor<T> {
|
|
2735
|
-
finalCallback?: Callback<T>;
|
|
2736
|
-
process(value: T): void;
|
|
2737
|
-
addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
2738
|
-
removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
2739
|
-
}
|
|
2740
|
-
type Callback<T> = (input: T) => void;
|
|
2741
|
-
declare class MiddlewareProcessor<T> implements Processor<T> {
|
|
2742
|
-
private stack;
|
|
2743
|
-
finalCallback?: Callback<T>;
|
|
2744
|
-
addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
2745
|
-
removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
2746
|
-
process(value: T): void;
|
|
3948
|
+
private _groupOf;
|
|
2747
3949
|
}
|
|
2748
3950
|
|
|
2749
|
-
|
|
2750
|
-
|
|
3951
|
+
/** A point on the earth, as an application reports it for one client. */
|
|
3952
|
+
type ClientLocation = {
|
|
3953
|
+
latitude: number;
|
|
3954
|
+
longitude: number;
|
|
3955
|
+
};
|
|
2751
3956
|
/**
|
|
2752
|
-
*
|
|
2753
|
-
*
|
|
2754
|
-
*
|
|
3957
|
+
* Approximate cell width at each geohash length, for choosing a precision.
|
|
3958
|
+
*
|
|
3959
|
+
* Index is the character count; the value is the rough cell size at the equator. Cells are taller
|
|
3960
|
+
* than they are wide at high latitudes, so treat these as an order of magnitude, not a radius.
|
|
2755
3961
|
*/
|
|
2756
|
-
|
|
2757
|
-
/** The payload threaded through `accept()` middlewares: the sample and its optional context. */
|
|
2758
|
-
type AcceptMiddlewarePayload = {
|
|
2759
|
-
sample: ClientSample;
|
|
2760
|
-
context?: AcceptContext;
|
|
2761
|
-
};
|
|
3962
|
+
declare const GEOHASH_CELL_SIZES: readonly ["", "~5000 km", "~1250 km", "~156 km", "~39 km", "~5 km", "~1.2 km", "~150 m"];
|
|
2762
3963
|
/**
|
|
2763
|
-
*
|
|
2764
|
-
*
|
|
2765
|
-
*
|
|
2766
|
-
*
|
|
3964
|
+
* Encode a point as a geohash of `precision` characters — a **grid cell key**, not a cluster.
|
|
3965
|
+
*
|
|
3966
|
+
* ### Why cells rather than "within N kilometres"
|
|
3967
|
+
*
|
|
3968
|
+
* Grouping clients "within a radius" sounds like the natural thing and is a much worse fit. It is a
|
|
3969
|
+
* clustering problem, not a keying one: the groups depend on which client you start from, two
|
|
3970
|
+
* clients can each be within the radius of a third but not of each other, group identity is not
|
|
3971
|
+
* stable as participants join and leave, and maintaining it costs pairwise distance work. None of
|
|
3972
|
+
* that survives contact with a detector that has to produce the *same* group name on every tick so
|
|
3973
|
+
* a cooldown and a control group mean anything.
|
|
3974
|
+
*
|
|
3975
|
+
* A geohash prefix is a plain function of the coordinates: O(1), stable for the life of the client,
|
|
3976
|
+
* and usable directly as a population label. The honest cost is that a cell boundary can separate
|
|
3977
|
+
* two clients who are physically adjacent, which splits a real group into two smaller ones. That
|
|
3978
|
+
* biases towards **missing** a finding rather than inventing one, which is the right direction for
|
|
3979
|
+
* something that raises issues.
|
|
3980
|
+
*
|
|
3981
|
+
* Returns `undefined` for coordinates that are not finite or not on the earth, rather than encoding
|
|
3982
|
+
* nonsense into a plausible-looking cell key.
|
|
2767
3983
|
*/
|
|
2768
|
-
|
|
2769
|
-
|
|
2770
|
-
|
|
3984
|
+
declare function geohash(location: ClientLocation, precision?: number): string | undefined;
|
|
3985
|
+
|
|
3986
|
+
declare const ClientPopulationIssueTypes: {
|
|
3987
|
+
/** One issue type is concentrated on one client population while the rest of the fleet is fine. */
|
|
3988
|
+
readonly clientPopulationIssue: "CLIENT_POPULATION_ISSUE";
|
|
2771
3989
|
};
|
|
2772
|
-
/**
|
|
2773
|
-
type
|
|
2774
|
-
|
|
2775
|
-
|
|
2776
|
-
|
|
2777
|
-
|
|
2778
|
-
|
|
2779
|
-
|
|
2780
|
-
|
|
2781
|
-
|
|
2782
|
-
|
|
2783
|
-
|
|
2784
|
-
|
|
2785
|
-
|
|
2786
|
-
|
|
3990
|
+
/** The client attribute to group by. One axis per detector — see the class description. */
|
|
3991
|
+
type ClientPopulationAxis = 'browser' | 'engine' | 'platform' | 'operationSystem' | 'location';
|
|
3992
|
+
/**
|
|
3993
|
+
* Reads a client's coordinates, for `groupBy: 'location'`. **Required for that axis.**
|
|
3994
|
+
*
|
|
3995
|
+
* There is no coordinate field in `ClientSample`, so the shape is yours: read it off
|
|
3996
|
+
* `client.attachments`, off a custom meta item, or from an `appData` field your accept middleware
|
|
3997
|
+
* filled in. Return `undefined` for clients whose location you do not know — they are then excluded
|
|
3998
|
+
* from both the population and the control group, exactly like a client that never reported its
|
|
3999
|
+
* browser.
|
|
4000
|
+
*/
|
|
4001
|
+
type ClientLocationResolver = (client: ObservedClient) => ClientLocation | undefined;
|
|
4002
|
+
type ClientPopulationIssueDetectorConfig = {
|
|
4003
|
+
/**
|
|
4004
|
+
* The issue types to watch. **Required, and must not be empty.**
|
|
4005
|
+
*
|
|
4006
|
+
* **Match the issue family to the axis.** On the endpoint axes (`browser`, `engine`, `platform`,
|
|
4007
|
+
* `operationSystem`) the types worth grouping are the ones an endpoint owns: `cpulimitation`,
|
|
4008
|
+
* `encoder-bottleneck`, `capture-bottleneck`, `stuck-decoder`, `video-decoder-overloaded`.
|
|
4009
|
+
* Grouping a *network* symptom by browser is a category error — `congestion` clusters by ISP and
|
|
4010
|
+
* geography, not by build, and the detector would happily report a browser correlation that is
|
|
4011
|
+
* really a "most of our users are on Chrome" artefact.
|
|
4012
|
+
*
|
|
4013
|
+
* On the `location` axis it is the other way round: group the network symptoms — `congestion`,
|
|
4014
|
+
* `ice-disconnected`, `unstable-ice-path` — and not the endpoint ones, since there is no reason a
|
|
4015
|
+
* decoder should stall by geography.
|
|
4016
|
+
*/
|
|
4017
|
+
issueTypes: string[];
|
|
4018
|
+
/** Which client attribute to group by. Default `'browser'`. */
|
|
4019
|
+
groupBy: ClientPopulationAxis;
|
|
2787
4020
|
/**
|
|
2788
|
-
*
|
|
2789
|
-
*
|
|
2790
|
-
* entity. `appData` is application-owned; it is never modified by the `accept()` context.
|
|
4021
|
+
* Where to read a client's coordinates. **Required when `groupBy` is `'location'`**, ignored
|
|
4022
|
+
* otherwise. See {@link ClientLocationResolver}.
|
|
2791
4023
|
*/
|
|
2792
|
-
|
|
2793
|
-
/** Same as `createCallAppData`, for clients. Receives the (already-created) parent call. */
|
|
2794
|
-
createClientAppData?: ClientAppDataFactory;
|
|
4024
|
+
resolveClientLocation?: ClientLocationResolver;
|
|
2795
4025
|
/**
|
|
2796
|
-
*
|
|
2797
|
-
*
|
|
2798
|
-
*
|
|
4026
|
+
* Geohash characters to group locations by, i.e. how coarse a "place" is. Default `3` (~156 km).
|
|
4027
|
+
*
|
|
4028
|
+
* `2` ~1250 km, `3` ~156 km, `4` ~39 km, `5` ~5 km. Coarser cells hold more clients, which is what
|
|
4029
|
+
* makes a rate mean anything, so start coarse: a city-sized cell rarely has `minPopulationSize`
|
|
4030
|
+
* participants in it. See `utils/geohash` for why this is a grid cell and not a radius.
|
|
2799
4031
|
*/
|
|
2800
|
-
|
|
4032
|
+
locationPrecision: number;
|
|
2801
4033
|
/**
|
|
2802
|
-
*
|
|
2803
|
-
*
|
|
2804
|
-
*
|
|
2805
|
-
*
|
|
4034
|
+
* Group by `name` only, or by `name + version`. Default `true` (include version).
|
|
4035
|
+
*
|
|
4036
|
+
* Version is usually the point: "Chrome" is not actionable, "Chrome 141" is, because it names a
|
|
4037
|
+
* thing that changed on a date. Set to `false` when comparing whole engines.
|
|
4038
|
+
*/
|
|
4039
|
+
includeVersion: boolean;
|
|
4040
|
+
/**
|
|
4041
|
+
* Clients in a population before its rate means anything. Default `20`.
|
|
4042
|
+
*
|
|
4043
|
+
* Higher than the other detectors' minimums on purpose: this one compares *rates*, and a rate over
|
|
4044
|
+
* five clients is not a rate. Sensible range `20`–`100`. On the `location` axis this is the field
|
|
4045
|
+
* most likely to silence the detector — a city-sized cell rarely holds twenty concurrent
|
|
4046
|
+
* participants, so reach for a coarser `locationPrecision` before lowering this.
|
|
4047
|
+
*/
|
|
4048
|
+
minPopulationSize: number;
|
|
4049
|
+
/**
|
|
4050
|
+
* Affected clients required within the population. Default `5`.
|
|
4051
|
+
*
|
|
4052
|
+
* Checked independently of `affectedRatioThreshold`, so one unlucky user on a rare browser cannot
|
|
4053
|
+
* page anyone however striking the ratio looks. Sensible range `5`–`20`.
|
|
4054
|
+
*/
|
|
4055
|
+
minAffectedClients: number;
|
|
4056
|
+
/**
|
|
4057
|
+
* Share of the population that must be affected, `0`–`1`. Default `0.3`.
|
|
4058
|
+
*
|
|
4059
|
+
* Lower than the per-call thresholds deliberately: an issue hitting 30% of one browser version while
|
|
4060
|
+
* the rest of the fleet is clean is already a strong signal, and endpoint faults rarely affect
|
|
4061
|
+
* *everyone* on a build. Typical `0.2`–`0.5`. This is the weakest of the gates —
|
|
4062
|
+
* `minRelativeRisk` is what makes the finding mean anything.
|
|
4063
|
+
*/
|
|
4064
|
+
affectedRatioThreshold: number;
|
|
4065
|
+
/**
|
|
4066
|
+
* How many times worse the suspect population must be than the rest of the fleet. Default `3`.
|
|
4067
|
+
*
|
|
4068
|
+
* **This is the gate that makes the finding mean anything** — see the class description. Typical
|
|
4069
|
+
* `2`–`10`. At `2` you will see populations that are merely somewhat worse, which is often just a
|
|
4070
|
+
* different usage pattern; at `10` only stark, unambiguous concentrations survive. A spotless
|
|
4071
|
+
* control group yields `Infinity`, which clears any threshold, so the minimum-count gates above are
|
|
4072
|
+
* what stop that from being trivial.
|
|
4073
|
+
*/
|
|
4074
|
+
minRelativeRisk: number;
|
|
4075
|
+
/**
|
|
4076
|
+
* Clients **outside** the suspect population before a comparison is possible. Default `20`.
|
|
4077
|
+
*
|
|
4078
|
+
* "Worse than everyone else" needs an everyone else. Sensible range `20`–`100`. Note the practical
|
|
4079
|
+
* consequence: a fleet that is overwhelmingly one browser can never have that browser reported,
|
|
4080
|
+
* because there is no control group left — which is honest, since at 95% Chrome you cannot separate a
|
|
4081
|
+
* Chrome fault from a fleet-wide one.
|
|
2806
4082
|
*/
|
|
2807
|
-
|
|
4083
|
+
minControlSize: number;
|
|
4084
|
+
/**
|
|
4085
|
+
* Re-arm time per (population, issue type) in ms. Default `300_000`.
|
|
4086
|
+
*
|
|
4087
|
+
* Long by design: a bad client build is a condition lasting days, not an event, and the action it
|
|
4088
|
+
* prompts — ship a fix, roll back a version — is not one you take twice an hour. Typical
|
|
4089
|
+
* `300_000`–`3_600_000`.
|
|
4090
|
+
*/
|
|
4091
|
+
cooldownMs: number;
|
|
2808
4092
|
};
|
|
2809
|
-
|
|
2810
|
-
|
|
2811
|
-
|
|
2812
|
-
|
|
2813
|
-
|
|
4093
|
+
/** The rollup for one population on one issue type. */
|
|
4094
|
+
type ClientPopulation = {
|
|
4095
|
+
/**
|
|
4096
|
+
* e.g. `'Chrome 141'`, or `'Chrome'` when `includeVersion` is off. For the `'location'` axis this
|
|
4097
|
+
* is the geohash cell — never the coordinates themselves, so an archived payload carries a place
|
|
4098
|
+
* at the configured resolution and not a person's position.
|
|
4099
|
+
*/
|
|
4100
|
+
population: string;
|
|
4101
|
+
axis: ClientPopulationAxis;
|
|
4102
|
+
issueType: string;
|
|
4103
|
+
clients: number;
|
|
4104
|
+
affectedClients: number;
|
|
4105
|
+
affectedRatio: number;
|
|
4106
|
+
affectedClientIds: string[];
|
|
4107
|
+
/** Everyone not in this population. */
|
|
4108
|
+
controlClients: number;
|
|
4109
|
+
controlAffectedClients: number;
|
|
4110
|
+
controlAffectedRatio: number;
|
|
4111
|
+
/** `affectedRatio / controlAffectedRatio`. `Infinity` when the control group is completely clean. */
|
|
4112
|
+
relativeRisk: number;
|
|
4113
|
+
};
|
|
4114
|
+
/**
|
|
4115
|
+
* Finds an issue that is concentrated on **one kind of client** — one browser, one browser version,
|
|
4116
|
+
* one OS — rather than on anything the servers own.
|
|
4117
|
+
*
|
|
4118
|
+
* ### Why this exists
|
|
4119
|
+
*
|
|
4120
|
+
* The other observer-scoped detectors all answer "who else has this open, and what do they share?"
|
|
4121
|
+
* with the answer *the infrastructure*, because clients in unrelated calls share nothing else. That
|
|
4122
|
+
* inference is right for network symptoms and **wrong for endpoint symptoms**, and the difference
|
|
4123
|
+
* matters at 3am. `cpulimitation` opening across six unrelated calls is not an SFU event: CPU is
|
|
4124
|
+
* owned by the endpoint, so what those endpoints have in common is a client release, a browser
|
|
4125
|
+
* update, or a fleet of identical VDI hosts. `IssueConclusion` already says exactly this — it maps
|
|
4126
|
+
* the endpoint-capacity family to a `client-population` fault domain instead of `infrastructure` —
|
|
4127
|
+
* but until now nothing in the library actually computed the grouping that claim refers to. This
|
|
4128
|
+
* detector is that computation.
|
|
4129
|
+
*
|
|
4130
|
+
* It is the one correlation in this library that is neither per-call nor per-server. A client knows
|
|
4131
|
+
* its own browser and nothing about anyone else's; only something sitting above the whole fleet can
|
|
4132
|
+
* notice that every complaint is coming from the same build.
|
|
4133
|
+
*
|
|
4134
|
+
* ### The control group is the whole point
|
|
4135
|
+
*
|
|
4136
|
+
* "30% of Chrome 141 users report encoder-bottleneck" is not a finding on its own. If 30% of
|
|
4137
|
+
* *everyone* reports it, Chrome 141 is not the story — you have a fleet-wide problem and this
|
|
4138
|
+
* detector would be pointing at the largest population rather than at a cause. Naive share-based
|
|
4139
|
+
* grouping always indicts whichever browser is most popular, which is why the gate here is
|
|
4140
|
+
* **relative risk**: the suspect population's rate divided by the rate among everyone else. A
|
|
4141
|
+
* population only qualifies when it is `minRelativeRisk` times worse than the rest of the fleet, and
|
|
4142
|
+
* only when the rest of the fleet is large enough (`minControlSize`) for "the rest of the fleet" to
|
|
4143
|
+
* be a real measurement.
|
|
4144
|
+
*
|
|
4145
|
+
* A completely clean control group gives `Infinity`, which is honest — nobody outside this
|
|
4146
|
+
* population has the problem at all — and is exactly why `minAffectedClients` and
|
|
4147
|
+
* `minPopulationSize` are checked independently, so a single unlucky user on a rare browser cannot
|
|
4148
|
+
* page anyone.
|
|
4149
|
+
*
|
|
4150
|
+
* ### One axis per detector
|
|
4151
|
+
*
|
|
4152
|
+
* `groupBy` takes a single attribute. Add a second instance if you want a second axis:
|
|
4153
|
+
*
|
|
4154
|
+
* ```ts
|
|
4155
|
+
* observer.addObserverDetector('client-population-issue-detector', {
|
|
4156
|
+
* issueTypes: [ 'cpulimitation', 'encoder-bottleneck', 'stuck-decoder' ],
|
|
4157
|
+
* groupBy: 'browser',
|
|
4158
|
+
* });
|
|
4159
|
+
*
|
|
4160
|
+
* observer.on('observer-issue', ({ issue }) => {
|
|
4161
|
+
* if (issue.type !== ClientPopulationIssueTypes.clientPopulationIssue) return;
|
|
4162
|
+
* // → { population: 'Chrome 141', issueType: 'encoder-bottleneck',
|
|
4163
|
+
* // affectedRatio: 0.34, controlAffectedRatio: 0.02, relativeRisk: 17 }
|
|
4164
|
+
* });
|
|
4165
|
+
* ```
|
|
4166
|
+
*
|
|
4167
|
+
* Deliberately not a cross-product of every axis at once: an issue that clusters on macOS *and* on
|
|
4168
|
+
* Safari is usually one fact reported twice, and a detector that emits both leaves the reader to
|
|
4169
|
+
* work out which one is causal. Pick the axis you want to reason about.
|
|
4170
|
+
*
|
|
4171
|
+
* ### The `location` axis
|
|
4172
|
+
*
|
|
4173
|
+
* With `groupBy: 'location'` the population is a **geohash cell** rather than a client attribute, so
|
|
4174
|
+
* the same machinery answers a different question: *is this symptom concentrated in one place?* That
|
|
4175
|
+
* is the grouping the note above says browsers cannot give you, and it is the one that matters for
|
|
4176
|
+
* network symptoms.
|
|
4177
|
+
*
|
|
4178
|
+
* The observer does not derive "this client's RTT jumped" — `client-monitor`'s `CongestionDetector`
|
|
4179
|
+
* already owns that verdict, comparing each peer connection's RTT against its own EWMA baseline and
|
|
4180
|
+
* requiring a bandwidth-limitation corroboration before it raises `congestion`. Absolute RTT is not
|
|
4181
|
+
* comparable between clients anyway: someone 200 ms away is *always* 200 ms away, so the only signal
|
|
4182
|
+
* is deviation from that client's own baseline, which is exactly what the client already measures.
|
|
4183
|
+
* This detector's contribution is the part no endpoint can see — that many of the affected clients
|
|
4184
|
+
* are in the same place at the same time.
|
|
4185
|
+
*
|
|
4186
|
+
* ```ts
|
|
4187
|
+
* observer.addObserverDetector('client-population-issue-detector', {
|
|
4188
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
4189
|
+
* groupBy: 'location',
|
|
4190
|
+
* locationPrecision: 3, // ~156 km cells
|
|
4191
|
+
* resolveClientLocation: (client) => client.attachments?.geo as { latitude: number, longitude: number },
|
|
4192
|
+
* });
|
|
4193
|
+
* ```
|
|
4194
|
+
*
|
|
4195
|
+
* Coordinates are not in `ClientSample`, so `resolveClientLocation` is required — see
|
|
4196
|
+
* {@link ClientLocationResolver}. Only the cell key reaches the issue payload, never the
|
|
4197
|
+
* coordinates, which matters because these payloads are archived into call summaries.
|
|
4198
|
+
*
|
|
4199
|
+
* **The limitation to state plainly: geography is confounded with your topology.** The control group
|
|
4200
|
+
* is "everyone outside this cell", which cannot separate *"the path into this region degraded"* from
|
|
4201
|
+
* *"the SFU that happens to serve this region degraded"*. If a region maps largely onto one
|
|
4202
|
+
* deployment, both hypotheses fit the same evidence. The discriminator is whether clients elsewhere
|
|
4203
|
+
* on the same server also degraded, which is what `SfuCongestionDetector` and
|
|
4204
|
+
* `TurnServerHealthDetector` answer — so the conclusion here points at them rather than claiming an
|
|
4205
|
+
* attribution it cannot support.
|
|
4206
|
+
*
|
|
4207
|
+
* ### Clients that never reported their metadata
|
|
4208
|
+
*
|
|
4209
|
+
* `browser` / `engine` / `platform` / `operationSystem` arrive as client metadata and may be absent —
|
|
4210
|
+
* a client that closed before sending them, or an application that does not collect them. Those
|
|
4211
|
+
* clients are excluded from **both** the population and the control group rather than bucketed as
|
|
4212
|
+
* `'unknown'`. The same applies to a client whose location `resolveClientLocation` cannot supply. A synthetic `'unknown'` population would be a mixture of every real one, so any rate
|
|
4213
|
+
* computed for it means nothing, and leaving those clients in the control group would dilute the
|
|
4214
|
+
* comparison with clients whose kind we cannot verify.
|
|
4215
|
+
*/
|
|
4216
|
+
declare class ClientPopulationIssueDetector implements Detector, ActiveIssueTracker {
|
|
4217
|
+
private readonly _observer;
|
|
4218
|
+
static readonly NAME: "client-population-issue-detector";
|
|
4219
|
+
readonly name: "client-population-issue-detector";
|
|
4220
|
+
private readonly _config;
|
|
4221
|
+
private readonly _lastRaisedAt;
|
|
4222
|
+
private readonly _issues;
|
|
4223
|
+
/** The populations that qualified on the most recent `update()`. Exposed for tests/dashboards. */
|
|
4224
|
+
lastPopulations: ClientPopulation[];
|
|
4225
|
+
constructor(_observer: Observer, config?: Partial<ClientPopulationIssueDetectorConfig>);
|
|
4226
|
+
get size(): number;
|
|
4227
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4228
|
+
add(issue: ActiveClientIssue): void;
|
|
4229
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4230
|
+
clear(): void;
|
|
4231
|
+
update(): void;
|
|
4232
|
+
close(): void;
|
|
4233
|
+
/** `undefined` when the client never reported this attribute — see the class description. */
|
|
4234
|
+
private _populationOf;
|
|
4235
|
+
private _rollupOf;
|
|
4236
|
+
private _riskText;
|
|
4237
|
+
/** Bigger populations and starker contrasts are harder to produce by chance. */
|
|
4238
|
+
private _confidenceOf;
|
|
2814
4239
|
}
|
|
2815
|
-
|
|
2816
|
-
|
|
4240
|
+
|
|
4241
|
+
declare const PublisherFaultTypes: {
|
|
4242
|
+
/**
|
|
4243
|
+
* A publisher is reporting trouble on its own send path **while** its subscribers report trouble
|
|
4244
|
+
* receiving it. Both ends agree, so the source is implicated rather than inferred.
|
|
4245
|
+
*/
|
|
4246
|
+
readonly corroboratedPublisherFault: "CORROBORATED_PUBLISHER_FAULT";
|
|
4247
|
+
};
|
|
4248
|
+
type PublisherFaultCorroborationDetectorConfig = {
|
|
4249
|
+
/**
|
|
4250
|
+
* Issue types raised by the **publishing** client about its own outbound path. **Required.**
|
|
4251
|
+
*
|
|
4252
|
+
* The natural set from client-monitor-js: `encoder-bottleneck`, `capture-bottleneck`,
|
|
4253
|
+
* `dry-outbound-track`. All three mean "I am failing to produce or send this properly", which is
|
|
4254
|
+
* the half of the story the receivers cannot see.
|
|
4255
|
+
*/
|
|
4256
|
+
publisherIssueTypes: string[];
|
|
4257
|
+
/**
|
|
4258
|
+
* Issue types raised by the **subscribing** clients about the track they receive. **Required.**
|
|
4259
|
+
*
|
|
4260
|
+
* The natural set: `freezed-video-track`, `dry-inbound-track`, `video-recovery-failed`. These say
|
|
4261
|
+
* "I am not getting this properly", which is the half the publisher cannot see.
|
|
4262
|
+
*/
|
|
4263
|
+
receiverIssueTypes: string[];
|
|
4264
|
+
/**
|
|
4265
|
+
* Subscribers of the track that must be complaining at the same time. Default `2`.
|
|
4266
|
+
*
|
|
4267
|
+
* `1` still yields a genuine two-sided corroboration — publisher and one receiver agreeing is
|
|
4268
|
+
* already more than either says alone — but `2` rules out the case where a single receiver's own
|
|
4269
|
+
* downlink is at fault and merely coincides with the publisher's complaint. Sensible range `1`–`3`;
|
|
4270
|
+
* higher mostly costs you findings in small calls, where a track may only have two subscribers.
|
|
4271
|
+
*/
|
|
4272
|
+
minAffectedReceivers: number;
|
|
4273
|
+
/**
|
|
4274
|
+
* Re-arm time per published track (ms). Default `60_000`.
|
|
4275
|
+
*
|
|
4276
|
+
* Typical `30_000`–`300_000`. This detector raises the highest-confidence finding in the library,
|
|
4277
|
+
* so it is the one you least want repeating every tick.
|
|
4278
|
+
*/
|
|
4279
|
+
cooldownMs: number;
|
|
4280
|
+
};
|
|
4281
|
+
/** The two-sided evidence behind one finding. */
|
|
4282
|
+
type CorroboratedPublisherFault = {
|
|
4283
|
+
trackId: string;
|
|
4284
|
+
kind: string;
|
|
4285
|
+
publisherClientId: string;
|
|
4286
|
+
/** The publisher's own open issue types on this track. */
|
|
4287
|
+
publisherIssueTypes: string[];
|
|
4288
|
+
/** The receiver-side open issue types across this track's subscribers. */
|
|
4289
|
+
receiverIssueTypes: string[];
|
|
4290
|
+
receivers: number;
|
|
4291
|
+
affectedReceivers: number;
|
|
4292
|
+
affectedClientIds: string[];
|
|
4293
|
+
publisherBitrate?: number;
|
|
4294
|
+
};
|
|
4295
|
+
/**
|
|
4296
|
+
* Fires only when **both ends of one published track are complaining at the same time**: the
|
|
4297
|
+
* publisher about its own send path, and its subscribers about receiving it.
|
|
4298
|
+
*
|
|
4299
|
+
* ### How this differs from `IssueFanOutDetector`
|
|
4300
|
+
*
|
|
4301
|
+
* Fan-out sees one end. It observes that most of Alice's subscribers are unhappy and *infers* that
|
|
4302
|
+
* the fault is on Alice's side, because the affected clients share a publisher and nothing else. That
|
|
4303
|
+
* inference is sound, and it is still a inference: the same observation is produced by the SFU
|
|
4304
|
+
* mangling Alice's stream on the way out, with Alice herself perfectly healthy.
|
|
4305
|
+
*
|
|
4306
|
+
* This detector removes the inference. When Alice reports `encoder-bottleneck` *and* four of her six
|
|
4307
|
+
* subscribers report `freezed-video-track` in the same window, there is nothing left to deduce — the
|
|
4308
|
+
* source said it was struggling and the receivers confirmed the consequence. That is the strongest
|
|
4309
|
+
* statement this library can make about where a fault sits, and it is only available to something
|
|
4310
|
+
* holding both ends at once. Neither the publisher nor any receiver can reach this conclusion alone.
|
|
4311
|
+
*
|
|
4312
|
+
* Run both: fan-out is broader and catches the SFU-forwarding case where the publisher is fine;
|
|
4313
|
+
* this one is narrower and, when it fires, needs no interpretation.
|
|
4314
|
+
*
|
|
4315
|
+
* ### Silence here is not health
|
|
4316
|
+
*
|
|
4317
|
+
* A quiet detector means only that the two halves have not coincided — most commonly because the
|
|
4318
|
+
* publisher is genuinely fine and the fault is in forwarding, which is exactly the case `fan-out`
|
|
4319
|
+
* exists to report. Do not read "no corroborated fault" as "no publisher-side problem".
|
|
4320
|
+
*
|
|
4321
|
+
* ```ts
|
|
4322
|
+
* observedCall.addDetector('publisher-fault-corroboration-detector', {
|
|
4323
|
+
* publisherIssueTypes: [ 'encoder-bottleneck', 'capture-bottleneck', 'dry-outbound-track' ],
|
|
4324
|
+
* receiverIssueTypes: [ 'freezed-video-track', 'dry-inbound-track' ],
|
|
4325
|
+
* });
|
|
4326
|
+
* ```
|
|
4327
|
+
*
|
|
4328
|
+
* ### Requires a `RemoteTrackResolver`
|
|
4329
|
+
*
|
|
4330
|
+
* Matching a publisher's issue to its subscribers' issues needs the publisher↔subscriber links. With
|
|
4331
|
+
* no resolver the detector does nothing rather than guessing.
|
|
4332
|
+
*/
|
|
4333
|
+
declare class PublisherFaultCorroborationDetector implements Detector, ActiveIssueTracker {
|
|
4334
|
+
private readonly _call;
|
|
4335
|
+
static readonly NAME: "publisher-fault-corroboration-detector";
|
|
4336
|
+
readonly name: "publisher-fault-corroboration-detector";
|
|
4337
|
+
private readonly _config;
|
|
4338
|
+
private readonly _lastRaisedAt;
|
|
4339
|
+
/** Open publisher-side issues that name a track. */
|
|
4340
|
+
private readonly _publisherIssues;
|
|
4341
|
+
/** Open receiver-side issues that name a track. */
|
|
4342
|
+
private readonly _receiverIssues;
|
|
4343
|
+
/** The faults corroborated on the most recent `update()`. Exposed for tests/dashboards. */
|
|
4344
|
+
lastFaults: CorroboratedPublisherFault[];
|
|
4345
|
+
constructor(_call: ObservedCall, config?: Partial<PublisherFaultCorroborationDetectorConfig>);
|
|
4346
|
+
get size(): number;
|
|
4347
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4348
|
+
add(issue: ActiveClientIssue): void;
|
|
4349
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4350
|
+
clear(): void;
|
|
4351
|
+
update(): void;
|
|
4352
|
+
close(): void;
|
|
4353
|
+
/**
|
|
4354
|
+
* Resolve a publisher-side issue's `trackId` to the outbound track it is about.
|
|
4355
|
+
*
|
|
4356
|
+
* Looked up through the reporting client's own peer connections: the issue names its client, so
|
|
4357
|
+
* the search is bounded by that client's transports rather than by the size of the call.
|
|
4358
|
+
*/
|
|
4359
|
+
private _outboundTrackOf;
|
|
4360
|
+
}
|
|
4361
|
+
|
|
4362
|
+
declare const ObserverConcurrentIssueTypes: {
|
|
4363
|
+
/**
|
|
4364
|
+
* The same issue is open in **several unrelated calls at once**. Those clients share no meeting,
|
|
4365
|
+
* no publisher and no room — only the infrastructure serving them.
|
|
4366
|
+
*/
|
|
4367
|
+
readonly crossCallConcurrentIssues: "CROSS_CALL_CONCURRENT_ISSUES";
|
|
4368
|
+
/** The cross-call version that also started together. The strongest "it's us" signal available. */
|
|
4369
|
+
readonly crossCallIssueOnsetBurst: "CROSS_CALL_ISSUE_ONSET_BURST";
|
|
4370
|
+
};
|
|
4371
|
+
type ObserverConcurrentIssueDetectorConfig = {
|
|
4372
|
+
/**
|
|
4373
|
+
* The issue types to watch. **Required, and must not be empty** — the detector subscribes to
|
|
4374
|
+
* exactly these and sees nothing else.
|
|
4375
|
+
*/
|
|
4376
|
+
issueTypes: string[];
|
|
4377
|
+
/**
|
|
4378
|
+
* Distinct clients that must share the open issue. Default `3`.
|
|
4379
|
+
*
|
|
4380
|
+
* Absolute, deliberately — see `affectedCallRatioThreshold` for why no client *ratio* exists at
|
|
4381
|
+
* this scope. Sensible range `3`–`10`; scale it with fleet size, since three clients is a real
|
|
4382
|
+
* signal across five calls and background noise across five hundred.
|
|
4383
|
+
*/
|
|
4384
|
+
minAffectedClients: number;
|
|
4385
|
+
/**
|
|
4386
|
+
* Minimum number of *distinct calls* the affected clients must span. Default `2`.
|
|
4387
|
+
*
|
|
4388
|
+
* This is what makes an observer-scoped finding mean something a call-scoped one doesn't. Without
|
|
4389
|
+
* it, one thirty-person meeting with congestion satisfies every client-count threshold and raises
|
|
4390
|
+
* a fleet-wide alert for what is really one bad room — which `CallConcurrentIssueDetector` has
|
|
4391
|
+
* already reported. Requiring two or more independent calls is the difference between a
|
|
4392
|
+
* coincidence and a shared cause.
|
|
4393
|
+
*/
|
|
4394
|
+
minAffectedCalls: number;
|
|
4395
|
+
/**
|
|
4396
|
+
* Fraction of the calls in flight that must be affected. Default `0` — off, because absolute
|
|
4397
|
+
* counts matter more than ratios here: three broken calls out of a thousand is still worth
|
|
4398
|
+
* knowing about, and a ratio threshold would hide it. Raise it if you only care about fleet-wide
|
|
4399
|
+
* events.
|
|
4400
|
+
*
|
|
4401
|
+
* Note there is deliberately **no participant ratio** at this scope. Six broken calls out of forty
|
|
4402
|
+
* is a handful of clients against the whole fleet, so any meaningful client ratio would suppress
|
|
4403
|
+
* exactly the finding this detector exists to produce.
|
|
4404
|
+
*/
|
|
4405
|
+
affectedCallRatioThreshold: number;
|
|
4406
|
+
/**
|
|
4407
|
+
* Onsets falling within this span (ms) escalate the finding to `CROSS_CALL_ISSUE_ONSET_BURST`.
|
|
4408
|
+
* Default `2_000`.
|
|
4409
|
+
*
|
|
4410
|
+
* This is the strongest evidence the detector produces: independent calls starting to fail *at the
|
|
4411
|
+
* same instant* has no explanation other than something they share. Keep it at or above your
|
|
4412
|
+
* sampling period — onsets are only as precise as clients report them — and no wider than a few
|
|
4413
|
+
* seconds, or unrelated failures drift into the same window. Typical `1_000`–`5_000`.
|
|
4414
|
+
*/
|
|
4415
|
+
onsetBurstWindowInMs: number;
|
|
4416
|
+
/**
|
|
4417
|
+
* Re-arm time per issue type (ms). Default `60_000`.
|
|
4418
|
+
*
|
|
4419
|
+
* Typical `60_000`–`600_000`. A fleet-wide event is one incident: without a cooldown a sustained
|
|
4420
|
+
* outage would raise an issue on every tick for its whole duration.
|
|
4421
|
+
*/
|
|
4422
|
+
cooldownMs: number;
|
|
4423
|
+
};
|
|
4424
|
+
/** What the detector currently knows about one issue type across the fleet. */
|
|
4425
|
+
type ObserverConcurrentIssueGroup = {
|
|
4426
|
+
type: string;
|
|
4427
|
+
issues: ActiveClientIssue[];
|
|
4428
|
+
clientIds: string[];
|
|
4429
|
+
totalClients: number;
|
|
4430
|
+
affectedRatio: number;
|
|
4431
|
+
callIds: string[];
|
|
4432
|
+
totalCalls: number;
|
|
4433
|
+
affectedCallRatio: number;
|
|
4434
|
+
/** Per-call breakdown, largest first — the first question anyone asks is "which calls, how badly?". */
|
|
4435
|
+
perCall: {
|
|
4436
|
+
callId: string;
|
|
4437
|
+
affectedClients: number;
|
|
4438
|
+
totalClients: number;
|
|
4439
|
+
}[];
|
|
4440
|
+
/**
|
|
4441
|
+
* Spread of the onsets, in **observer** time (ms). Client clocks are never compared across
|
|
4442
|
+
* machines here — skew between them would masquerade as a synchronized event.
|
|
4443
|
+
*/
|
|
4444
|
+
onsetSpreadInMs: number;
|
|
4445
|
+
firstObservedAt: number;
|
|
4446
|
+
};
|
|
4447
|
+
/**
|
|
4448
|
+
* Answers **"is our infrastructure in trouble?"** — the same issue open across several *unrelated*
|
|
4449
|
+
* calls at the same moment.
|
|
4450
|
+
*
|
|
4451
|
+
* This is not the call-scoped question with a bigger denominator, which is why it is a separate
|
|
4452
|
+
* detector with separate gates and its own finding types. Participant count alone is a bad fleet
|
|
4453
|
+
* signal: one thirty-person meeting where everybody is congested clears every client threshold, yet
|
|
4454
|
+
* it has an obvious local explanation. Clients in *different* calls share no room, no publisher and
|
|
4455
|
+
* no host — only the servers and the network. When the same issue opens across several of them at
|
|
4456
|
+
* once, the infrastructure is the only remaining common factor, and that is the finding worth paging
|
|
4457
|
+
* someone about.
|
|
4458
|
+
*
|
|
4459
|
+
* ```ts
|
|
4460
|
+
* observer.addObserverDetector('observer-concurrent-issue-detector', {
|
|
4461
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
4462
|
+
* minAffectedCalls: 3,
|
|
4463
|
+
* });
|
|
4464
|
+
*
|
|
4465
|
+
* observer.on('observer-issue', ({ issue }) => {
|
|
4466
|
+
* if (issue.type === ObserverConcurrentIssueTypes.crossCallIssueOnsetBurst) page(issue);
|
|
4467
|
+
* });
|
|
4468
|
+
* // → CROSS_CALL_ISSUE_ONSET_BURST { issueType: 'congestion', calls: 40, affectedCalls: 6, … }
|
|
4469
|
+
* ```
|
|
4470
|
+
*
|
|
4471
|
+
* Onsets are compared on the **observer clock** (`observedAt`), never on client clocks: participants
|
|
4472
|
+
* degrading together within a couple of seconds is far more likely to be a deploy, a TURN failover or
|
|
4473
|
+
* a link flap than a coincidence — but only if the timestamps being compared came from one clock.
|
|
4474
|
+
*/
|
|
4475
|
+
declare class ObserverConcurrentIssueDetector implements Detector, ActiveIssueTracker {
|
|
4476
|
+
private readonly _observer;
|
|
4477
|
+
static readonly NAME: "observer-concurrent-issue-detector";
|
|
4478
|
+
readonly name: "observer-concurrent-issue-detector";
|
|
4479
|
+
private readonly _config;
|
|
4480
|
+
private readonly _lastRaisedAt;
|
|
4481
|
+
/** issue type -> the issues of that type currently open anywhere in the fleet. */
|
|
4482
|
+
private readonly _byType;
|
|
4483
|
+
private _size;
|
|
4484
|
+
/** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
|
|
4485
|
+
lastGroups: ObserverConcurrentIssueGroup[];
|
|
4486
|
+
constructor(_observer: Observer, config?: Partial<ObserverConcurrentIssueDetectorConfig>);
|
|
4487
|
+
get size(): number;
|
|
4488
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4489
|
+
add(issue: ActiveClientIssue): void;
|
|
4490
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4491
|
+
clear(): void;
|
|
4492
|
+
update(): void;
|
|
4493
|
+
close(): void;
|
|
4494
|
+
private _groupOf;
|
|
4495
|
+
}
|
|
4496
|
+
|
|
4497
|
+
declare const IssueFanOutTypes: {
|
|
4498
|
+
/** Most receivers of one published track have the same issue open → the fault follows the source. */
|
|
4499
|
+
readonly publishedTrackIssueFanOut: "PUBLISHED_TRACK_ISSUE_FAN_OUT";
|
|
4500
|
+
/** Exactly one receiver of a track has it → that receiver's own problem. */
|
|
4501
|
+
readonly singleReceiverIssue: "SINGLE_RECEIVER_ISSUE";
|
|
4502
|
+
};
|
|
4503
|
+
type IssueFanOutDetectorConfig = {
|
|
4504
|
+
/**
|
|
4505
|
+
* The receiver-side issue types to attribute to publishers. **Required, and must not be empty** —
|
|
4506
|
+
* the detector subscribes to exactly these.
|
|
4507
|
+
*
|
|
4508
|
+
* There is no "all types" option. Which of a receiver's complaints are worth blaming a publisher
|
|
4509
|
+
* for is application knowledge: `freezed-video-track` fanning out across a track's subscribers
|
|
4510
|
+
* implicates the source, `cpulimitation` fanning out the same way implicates the receivers'
|
|
4511
|
+
* hardware and would be a false accusation.
|
|
4512
|
+
*/
|
|
4513
|
+
issueTypes: string[];
|
|
4514
|
+
/**
|
|
4515
|
+
* Receivers a track needs before a ratio means anything. Default `3`.
|
|
4516
|
+
*
|
|
4517
|
+
* With two receivers, "60% affected" is one of them — which is the single-receiver case below, not
|
|
4518
|
+
* a fan-out. Sensible range `2`–`5`; in small calls a published track rarely has more than a couple
|
|
4519
|
+
* of subscribers, so raising this can silence the detector entirely.
|
|
4520
|
+
*/
|
|
4521
|
+
minReceivers: number;
|
|
4522
|
+
/**
|
|
4523
|
+
* Fraction of a track's receivers that must share the issue, `0`–`1`. Default `0.6`.
|
|
4524
|
+
*
|
|
4525
|
+
* The higher this is, the more the finding points at the publisher rather than at the network
|
|
4526
|
+
* between: *everyone* receiving this track badly is hard to explain any other way. Typical
|
|
4527
|
+
* `0.5`–`0.8`. Below `0.5` you are reporting "some receivers", which usually means their own
|
|
4528
|
+
* last miles.
|
|
4529
|
+
*/
|
|
4530
|
+
affectedRatioThreshold: number;
|
|
4531
|
+
/**
|
|
4532
|
+
* Also report when exactly one receiver is affected. Default `true`.
|
|
4533
|
+
*
|
|
4534
|
+
* Kept on because the finding is *useful and correctly weaker*: it is raised with a lower
|
|
4535
|
+
* confidence and the opposite conclusion — one unhappy receiver out of eight points at that
|
|
4536
|
+
* receiver, not at the publisher. Turn it off if you only want publisher-blaming findings and
|
|
4537
|
+
* treat single-receiver trouble as the client's own business.
|
|
4538
|
+
*/
|
|
4539
|
+
reportSingleReceiver: boolean;
|
|
4540
|
+
/**
|
|
4541
|
+
* Re-arm time per (track, issue type) in ms. Default `60_000`.
|
|
4542
|
+
*
|
|
4543
|
+
* Per track, so a call with many bad publishers still reports each of them. Typical
|
|
4544
|
+
* `30_000`–`300_000`.
|
|
4545
|
+
*/
|
|
4546
|
+
cooldownMs: number;
|
|
4547
|
+
};
|
|
4548
|
+
/**
|
|
4549
|
+
* Attributes **client-reported issues to the published track they are about**, then asks how far
|
|
4550
|
+
* the problem fans out across that track's receivers.
|
|
4551
|
+
*
|
|
4552
|
+
* The join is what makes this possible: a receiver-side issue payload carries `trackId` (the client
|
|
4553
|
+
* detectors report it for every track-scoped issue), the observer resolves that to an inbound track,
|
|
4554
|
+
* and `RemoteTrackResolver` links the inbound track to the `remoteOutboundTrack` that published it.
|
|
4555
|
+
* With the whole subscriber set of one source in hand, the verdict is straightforward and is the
|
|
4556
|
+
* single most useful thing a server can say:
|
|
4557
|
+
*
|
|
4558
|
+
* - **most receivers of Alice's track are affected** → the fault is on Alice's path — her uplink, the
|
|
4559
|
+
* SFU's ingress, or its forwarding of that stream.
|
|
4560
|
+
* - **one receiver of Alice's track is affected** → that receiver's downlink. Nothing to do with
|
|
4561
|
+
* Alice, even though the symptom is reported against her stream.
|
|
4562
|
+
*
|
|
4563
|
+
* Deliberately generic over the issue vocabulary: `freezed-video-track`, `keyframe-storm`,
|
|
4564
|
+
* `audio-concealment`, `video-decoder-overloaded`, `stuck-decoder` and anything a custom client
|
|
4565
|
+
* detector invents all fan out the same way, so one mechanism replaces a family of symptom-specific
|
|
4566
|
+
* detectors.
|
|
4567
|
+
*
|
|
4568
|
+
* ### It walks the affected tracks, never all of them
|
|
4569
|
+
*
|
|
4570
|
+
* The detector is fed open issues by the call's registry and keeps only those carrying a `trackId`.
|
|
4571
|
+
* Each tick it resolves *those* tracks to their publishers — never the published tracks of the call,
|
|
4572
|
+
* of which there are many more and almost all of them fine. A call with nothing wrong costs one
|
|
4573
|
+
* `size === 0` check.
|
|
4574
|
+
*
|
|
4575
|
+
* ### Requires a `RemoteTrackResolver`
|
|
4576
|
+
*
|
|
4577
|
+
* Without publisher↔subscriber links there is no way to know which receivers belong to one source,
|
|
4578
|
+
* so the detector does nothing when the call has no resolver. It does not fall back to guessing:
|
|
4579
|
+
* "one receiver of an unknown set" is not a statement worth raising.
|
|
4580
|
+
*/
|
|
4581
|
+
declare class IssueFanOutDetector implements Detector, ActiveIssueTracker {
|
|
4582
|
+
private readonly _call;
|
|
4583
|
+
static readonly NAME = "issue-fan-out-detector";
|
|
4584
|
+
readonly name = "issue-fan-out-detector";
|
|
4585
|
+
private readonly _config;
|
|
4586
|
+
private readonly _lastRaisedAt;
|
|
4587
|
+
/** Open issues that name a track. Issues without a `trackId` cannot be attributed and are dropped. */
|
|
4588
|
+
private readonly _trackIssues;
|
|
4589
|
+
constructor(_call: ObservedCall, config?: Partial<IssueFanOutDetectorConfig>);
|
|
4590
|
+
get size(): number;
|
|
4591
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4592
|
+
add(issue: ActiveClientIssue): void;
|
|
4593
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4594
|
+
clear(): void;
|
|
4595
|
+
update(): void;
|
|
4596
|
+
close(): void;
|
|
4597
|
+
/**
|
|
4598
|
+
* Resolve the issue's `trackId` to the outbound track that published it.
|
|
4599
|
+
*
|
|
4600
|
+
* Looked up through the reporting client's own peer connections rather than by scanning the call:
|
|
4601
|
+
* the issue names its client, so the search is bounded by that client's transports (typically one
|
|
4602
|
+
* or two) instead of by the size of the meeting.
|
|
4603
|
+
*/
|
|
4604
|
+
private _publisherOf;
|
|
4605
|
+
}
|
|
4606
|
+
|
|
4607
|
+
/**
|
|
4608
|
+
* One completed sampling bucket: how many/which clients reported congestion during it.
|
|
4609
|
+
*
|
|
4610
|
+
* `totalClients` (and therefore `congestedClientRatio`) is a snapshot of `observer.numberOfClients`
|
|
4611
|
+
* taken when the bucket closes — an approximation of "how many clients could have been congested",
|
|
4612
|
+
* not a claim that every client sent exactly one sample within the bucket. Good enough for a ratio
|
|
4613
|
+
* that only needs to be comparable bucket-to-bucket.
|
|
4614
|
+
*/
|
|
4615
|
+
type SfuCongestionDetectorBucket = {
|
|
4616
|
+
observedAt: number;
|
|
4617
|
+
totalClients: number;
|
|
4618
|
+
congestedClients: number;
|
|
4619
|
+
congestedClientRatio: number;
|
|
4620
|
+
affectedClientIds: string[];
|
|
4621
|
+
affectedCallIds: string[];
|
|
4622
|
+
};
|
|
4623
|
+
type SfuCongestionDetectorReport = {
|
|
4624
|
+
affectedCallIds: string[];
|
|
4625
|
+
affectedClientIds: string[];
|
|
4626
|
+
congestedClientRatio: number;
|
|
4627
|
+
totalNumberOfClients: number;
|
|
4628
|
+
numberOfCongestedClients: number;
|
|
4629
|
+
historySize: number;
|
|
4630
|
+
baselineCongestedClientRatio: number;
|
|
4631
|
+
robustZ: number;
|
|
4632
|
+
absoluteIncrease: number;
|
|
4633
|
+
relativeIncrease: number;
|
|
4634
|
+
};
|
|
4635
|
+
type SfuCongestionDetectorConfig = {
|
|
4636
|
+
/**
|
|
4637
|
+
* Client issue types counted as "congestion" for this indicator. Default `[ 'congestion' ]`.
|
|
4638
|
+
*
|
|
4639
|
+
* Keep this narrow. Every type you add widens what counts as a congested client, and the whole
|
|
4640
|
+
* method rests on comparing *like with like* across time buckets — mixing in a type that appears
|
|
4641
|
+
* for unrelated reasons raises the baseline and buries the spike you are looking for.
|
|
4642
|
+
*/
|
|
4643
|
+
consumedClientIssueTypes: string[];
|
|
4644
|
+
/** The `observer-issue` type raised when a bucket is judged congested. Default `'sfu-congestion'`. */
|
|
4645
|
+
emittedObserverIssueType: string;
|
|
4646
|
+
/**
|
|
4647
|
+
* How long your clients take to send a sample (ms). Default `10_000`.
|
|
4648
|
+
*
|
|
4649
|
+
* **Set this to your collector's actual sampling period** — it is a description of your clients,
|
|
4650
|
+
* not a tuning knob. The bucket is `samplesSendingTimeInMs * 2`, so every client gets a fair
|
|
4651
|
+
* chance to report at least once inside each bucket. Set it too short and clients that simply had
|
|
4652
|
+
* not reported yet look absent, so the ratio jumps around on sampling noise; too long and the
|
|
4653
|
+
* detector reacts slowly and averages a spike away.
|
|
4654
|
+
*/
|
|
4655
|
+
samplesSendingTimeInMs: number;
|
|
4656
|
+
/**
|
|
4657
|
+
* Completed buckets kept as history — the candidate plus its baseline. Default `30`.
|
|
4658
|
+
*
|
|
4659
|
+
* At the default bucket size this is ~10 minutes of baseline. Sensible range `10`–`60`. Longer is
|
|
4660
|
+
* more robust to a single odd bucket but slower to accept a genuinely changed normal (a growth
|
|
4661
|
+
* spurt, a new region coming online); shorter adapts quickly but lets a sustained problem become
|
|
4662
|
+
* the new baseline and stop being reported.
|
|
4663
|
+
*/
|
|
4664
|
+
historySize: number;
|
|
4665
|
+
/**
|
|
4666
|
+
* Buckets required before any judgement is made. Default `5`.
|
|
4667
|
+
*
|
|
4668
|
+
* Below this the detector is silent, which is the point: a median and MAD over two buckets is not
|
|
4669
|
+
* a baseline. Costs `minHistorySize * samplesSendingTimeInMs * 2` of warm-up after start — about
|
|
4670
|
+
* 100 s at the defaults. Do not lower it to make a test fire faster; shorten the bucket instead.
|
|
4671
|
+
*/
|
|
4672
|
+
minHistorySize: number;
|
|
4673
|
+
/**
|
|
4674
|
+
* Distinct congested clients required in the candidate bucket. Default `3`.
|
|
4675
|
+
*
|
|
4676
|
+
* The absolute floor beneath every ratio below, so that a tiny fleet cannot produce a finding: two
|
|
4677
|
+
* unhappy clients out of four is 50% and means nothing. Raise it on a large fleet where three
|
|
4678
|
+
* clients is always noise.
|
|
4679
|
+
*/
|
|
4680
|
+
minAffectedClients: number;
|
|
4681
|
+
/**
|
|
4682
|
+
* How far the candidate's congested-client **ratio** must exceed the baseline median, in absolute
|
|
4683
|
+
* terms (`0`–`1`). Default `0.05`, i.e. five percentage points.
|
|
4684
|
+
*
|
|
4685
|
+
* This is the practical-significance gate: it stops a statistically striking move from 0.5% to 2%
|
|
4686
|
+
* being reported as an event. Typical `0.03`–`0.15`.
|
|
4687
|
+
*/
|
|
4688
|
+
minAbsoluteRatioIncrease: number;
|
|
4689
|
+
/**
|
|
4690
|
+
* How many times the baseline median the candidate ratio must reach. Default `2`.
|
|
4691
|
+
*
|
|
4692
|
+
* Multiplicative counterpart to the absolute gate — both must pass. Typical `1.5`–`3`. Below
|
|
4693
|
+
* `1.5` ordinary fluctuation qualifies; above ~`4` only near-total events do.
|
|
4694
|
+
*/
|
|
4695
|
+
minRelativeRatioIncrease: number;
|
|
4696
|
+
/**
|
|
4697
|
+
* Robust z-score the candidate must reach against a median+MAD baseline. Default `3`.
|
|
4698
|
+
*
|
|
4699
|
+
* The statistical-significance gate. `3` is the conventional "clearly outside normal variation";
|
|
4700
|
+
* `2` is noticeably chattier, `4`–`5` only for very stable fleets. Median and MAD rather than mean
|
|
4701
|
+
* and standard deviation on purpose — a couple of past incidents in the history would inflate a
|
|
4702
|
+
* standard deviation enough to hide the next one. Note that a perfectly flat baseline gives
|
|
4703
|
+
* `MAD = 0`, where any increase scores `Infinity`; the two ratio gates above are what keep that
|
|
4704
|
+
* honest.
|
|
4705
|
+
*/
|
|
4706
|
+
robustZThreshold: number;
|
|
4707
|
+
};
|
|
4708
|
+
/** The statistical/practical-significance verdict for one candidate bucket against its baseline. */
|
|
4709
|
+
type SfuCongestionDetectorEvaluation = {
|
|
4710
|
+
isCongested: boolean;
|
|
4711
|
+
baselineCongestedClientRatio: number;
|
|
4712
|
+
robustZ: number;
|
|
4713
|
+
absoluteIncrease: number;
|
|
4714
|
+
relativeIncrease: number;
|
|
4715
|
+
};
|
|
4716
|
+
/**
|
|
4717
|
+
* Detects a **shared** congestion event: many clients, across different calls, reporting congestion
|
|
4718
|
+
* inside the same slice of time.
|
|
4719
|
+
*
|
|
4720
|
+
* Only add this when the observer's calls all come from the **same SFU** — the finding's whole
|
|
4721
|
+
* meaning is "these clients have nothing in common except that server", and that is only true if the
|
|
4722
|
+
* server really is the common factor.
|
|
4723
|
+
*
|
|
4724
|
+
* ### Why fixed-interval buckets, and not the update tick
|
|
4725
|
+
*
|
|
4726
|
+
* The obvious implementation counts congested clients on each `update()`. It is wrong here, for two
|
|
4727
|
+
* separate reasons:
|
|
4728
|
+
*
|
|
4729
|
+
* - **The tick is not evenly spaced.** `update()` fires when a client is updated, so its rate is a
|
|
4730
|
+
* function of how many clients are connected and how their sampling happens to interleave. Two
|
|
4731
|
+
* counts taken from windows of different length are not comparable, and this detector's entire
|
|
4732
|
+
* job is to compare a count against earlier counts.
|
|
4733
|
+
* - **Clients report on their own schedule.** A client sends a sample roughly every
|
|
4734
|
+
* `samplesSendingTimeInMs`, unsynchronised with every other client. A window shorter than that
|
|
4735
|
+
* systematically undercounts — half the congested clients simply hadn't spoken yet — and the
|
|
4736
|
+
* undercount varies with arrival phase, which is noise indistinguishable from signal.
|
|
4737
|
+
*
|
|
4738
|
+
* So the detector runs on a wall-clock interval and closes a bucket every
|
|
4739
|
+
* `samplesSendingTimeInMs`, giving every client a fair chance to be heard in each one. Buckets are
|
|
4740
|
+
* equal-length and equally lagged, which is what makes bucket-to-bucket comparison mean something.
|
|
4741
|
+
* {@link update} is deliberately empty: nothing here is driven by the update tick.
|
|
4742
|
+
*
|
|
4743
|
+
* ### Occurrences, not intervals
|
|
4744
|
+
*
|
|
4745
|
+
* Unlike `ConcurrentIssueDetector`, this one ignores resolutions — see {@link delete}. It counts how
|
|
4746
|
+
* many *distinct clients reported* congestion in a bucket, not how many are still congested. A
|
|
4747
|
+
* client that hits congestion and immediately drops its bitrate resolves the issue within seconds
|
|
4748
|
+
* and would vanish from an open-interval view, yet it is exactly the evidence wanted here.
|
|
4749
|
+
*/
|
|
4750
|
+
declare class SfuCongestionDetector implements Detector, ActiveIssueTracker {
|
|
4751
|
+
private readonly _observer;
|
|
4752
|
+
static readonly NAME = "sfu-congestion-detector";
|
|
4753
|
+
readonly name = "sfu-congestion-detector";
|
|
4754
|
+
private readonly _config;
|
|
4755
|
+
private readonly _history;
|
|
4756
|
+
private readonly _trackedIssues;
|
|
4757
|
+
private timer;
|
|
4758
|
+
private _lastEvaluatedBucket;
|
|
4759
|
+
constructor(_observer: Observer, config?: Partial<SfuCongestionDetectorConfig>);
|
|
4760
|
+
get size(): number;
|
|
4761
|
+
/** Record the issue against the bucket currently open. The timer, not this, closes the bucket. */
|
|
4762
|
+
add(issue: ActiveClientIssue): void;
|
|
4763
|
+
/**
|
|
4764
|
+
* Deliberately a no-op returning `false`.
|
|
4765
|
+
*
|
|
4766
|
+
* Resolutions are not interesting here. A congested client typically fixes its own symptom by
|
|
4767
|
+
* dropping bitrate hard, so the issue closes within seconds — but it still *happened*, and it is
|
|
4768
|
+
* evidence that the server was under pressure during this bucket. What matters is how many
|
|
4769
|
+
* distinct clients reported congestion within the bucket and whether that count suddenly jumps,
|
|
4770
|
+
* not how long any one client's issue stayed open.
|
|
4771
|
+
*
|
|
4772
|
+
* Nothing leaks: the tracked set is emptied wholesale every time a bucket closes.
|
|
4773
|
+
*/
|
|
4774
|
+
delete(_issue: ActiveClientIssue): boolean;
|
|
4775
|
+
clear(): void;
|
|
4776
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4777
|
+
close(): void;
|
|
4778
|
+
/** The completed buckets kept so far, oldest first. Read-only — for introspection/tests. */
|
|
4779
|
+
get history(): readonly SfuCongestionDetectorBucket[];
|
|
4780
|
+
/**
|
|
4781
|
+
* Intentionally empty — see the class description.
|
|
4782
|
+
*
|
|
4783
|
+
* Everything here is driven by the bucket timer, because the update tick is neither evenly spaced
|
|
4784
|
+
* nor long enough for every client to have reported. Counting on it would compare windows of
|
|
4785
|
+
* different lengths and call the difference a signal.
|
|
4786
|
+
*/
|
|
4787
|
+
update(): void;
|
|
4788
|
+
private _closeBucket;
|
|
4789
|
+
/**
|
|
4790
|
+
* Evaluate only the latest completed bucket (the candidate) against the buckets before it (the
|
|
4791
|
+
* baseline) — never against itself. Reached once per newly-closed bucket via {@link update}; the
|
|
4792
|
+
* identity check below additionally guards against evaluating the same bucket twice, in case
|
|
4793
|
+
* `update()` is ever called again before the next rotation.
|
|
4794
|
+
*/
|
|
4795
|
+
private _evaluateLatestBucket;
|
|
4796
|
+
/**
|
|
4797
|
+
* Is `candidate` — the latest completed bucket — abnormally high compared with the `baseline`
|
|
4798
|
+
* buckets before it?
|
|
4799
|
+
*
|
|
4800
|
+
* Requires both **statistical** significance (a robust z-score against a median+MAD baseline —
|
|
4801
|
+
* deliberately not Mann-Kendall, which asks "is this a monotonic trend", not "is the latest point
|
|
4802
|
+
* an outlier"; a single sudden spike on an otherwise flat series is exactly what should trigger
|
|
4803
|
+
* here and exactly what a trend test would miss) and **practical** significance (enough affected
|
|
4804
|
+
* clients, and a big enough absolute/relative jump — a statistically significant move in a tiny
|
|
4805
|
+
* or trivial ratio is not worth an alert).
|
|
4806
|
+
*/
|
|
4807
|
+
private _evaluateBucket;
|
|
4808
|
+
}
|
|
4809
|
+
|
|
4810
|
+
declare const TrackDeliveryMismatchTypes: {
|
|
4811
|
+
/**
|
|
4812
|
+
* The source is sending, but **none** of its subscribers are receiving → the media is being lost
|
|
4813
|
+
* between the publisher and the receivers. In an SFU that means the forwarding path.
|
|
4814
|
+
*/
|
|
4815
|
+
readonly publishedTrackNotDelivered: "PUBLISHED_TRACK_NOT_DELIVERED";
|
|
4816
|
+
/**
|
|
4817
|
+
* The source is sending and most subscribers are fine, but **some** are dry → those consumers are
|
|
4818
|
+
* broken individually (in mediasoup, the usual mitigation is recreating the consumer).
|
|
4819
|
+
*/
|
|
4820
|
+
readonly receiverTrackNotDelivered: "RECEIVER_TRACK_NOT_DELIVERED";
|
|
4821
|
+
/**
|
|
4822
|
+
* The source itself stopped producing, so its subscribers being dry is expected and **not** an
|
|
4823
|
+
* SFU fault. Reported so the other two verdicts can be trusted as *not* being this.
|
|
4824
|
+
*/
|
|
4825
|
+
readonly publisherTrackDry: "PUBLISHER_TRACK_DRY";
|
|
4826
|
+
};
|
|
4827
|
+
type TrackDeliveryMismatchDetectorConfig = {
|
|
4828
|
+
/**
|
|
4829
|
+
* The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`, which is
|
|
4830
|
+
* what client-monitor-js raises. Only change it if you raise your own equivalent.
|
|
4831
|
+
*/
|
|
4832
|
+
dryInboundIssueType: string;
|
|
4833
|
+
/**
|
|
4834
|
+
* The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`, as raised
|
|
4835
|
+
* by client-monitor-js. The pairing of these two types is the whole detector: their **disagreement**
|
|
4836
|
+
* is the finding.
|
|
4837
|
+
*/
|
|
4838
|
+
dryOutboundIssueType: string;
|
|
4839
|
+
/**
|
|
4840
|
+
* Subscribers required before "all of them" means anything. Default `2`.
|
|
4841
|
+
*
|
|
4842
|
+
* With one subscriber, "every receiver is dry" is a single client's report and carries no more
|
|
4843
|
+
* weight than the client issue already does. Sensible range `2`–`4`.
|
|
4844
|
+
*/
|
|
4845
|
+
minReceivers: number;
|
|
4846
|
+
/**
|
|
4847
|
+
* Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`.
|
|
4848
|
+
*
|
|
4849
|
+
* `1` — literally all of them — on purpose. The inference here is sharp: the publisher says it is
|
|
4850
|
+
* sending and *every* receiver says nothing arrives, so the fault is between them, in the SFU's
|
|
4851
|
+
* forwarding. Lowering it to `0.8` admits mixed evidence, where some receivers do get the media, and
|
|
4852
|
+
* the conclusion no longer follows: that is a per-receiver problem and `IssueFanOutDetector`'s
|
|
4853
|
+
* question. Do not lower it without deciding what the finding then means.
|
|
4854
|
+
*/
|
|
4855
|
+
allReceiversRatio: number;
|
|
4856
|
+
/** Re-arm time per (track, verdict) in ms. Default `60_000`. Typical `30_000`–`300_000`. */
|
|
4857
|
+
cooldownMs: number;
|
|
4858
|
+
};
|
|
4859
|
+
/**
|
|
4860
|
+
* Answers **"is the media actually getting through?"** by joining the two ends of a published track.
|
|
4861
|
+
*
|
|
4862
|
+
* A dry track is the clearest possible symptom — no bytes are arriving — but on its own it is
|
|
4863
|
+
* ambiguous, and the ambiguity is precisely what a single endpoint cannot resolve. A receiver seeing
|
|
4864
|
+
* silence cannot tell whether the camera was switched off, the SFU stopped forwarding, or its own
|
|
4865
|
+
* consumer wedged. All three look identical from the browser.
|
|
4866
|
+
*
|
|
4867
|
+
* With the publisher↔subscriber links this becomes a three-way decision:
|
|
4868
|
+
*
|
|
4869
|
+
* | publisher | subscribers | verdict |
|
|
4870
|
+
* |---|---|---|
|
|
4871
|
+
* | sending | **all** dry | `PUBLISHED_TRACK_NOT_DELIVERED` — the SFU/forwarding path |
|
|
4872
|
+
* | sending | **some** dry | `RECEIVER_TRACK_NOT_DELIVERED` — those consumers (recreate them) |
|
|
4873
|
+
* | dry | any dry | `PUBLISHER_TRACK_DRY` — the source stopped; not an SFU fault |
|
|
4874
|
+
*
|
|
4875
|
+
* The publisher side is judged from **both** signals available: its own `dry-outbound-track` issue
|
|
4876
|
+
* when the client reports one, and — as the fallback, and the corroboration when it does not — the
|
|
4877
|
+
* observed outbound RTP (`deltaPacketsSent`). That combination is what makes the first row
|
|
4878
|
+
* trustworthy: the server can state that packets demonstrably left the publisher during the same
|
|
4879
|
+
* interval in which every receiver got nothing.
|
|
4880
|
+
*
|
|
4881
|
+
* This is the "SFU forwarding mismatch" check, and notably it needs **no** mediasoup instrumentation
|
|
4882
|
+
* — the client's own dry-track verdicts plus the resolver links are sufficient.
|
|
4883
|
+
*/
|
|
4884
|
+
declare class TrackDeliveryMismatchDetector implements Detector, ActiveIssueTracker {
|
|
4885
|
+
private readonly call;
|
|
4886
|
+
static readonly NAME: "track-delivery-mismatch-detector";
|
|
4887
|
+
readonly name: "track-delivery-mismatch-detector";
|
|
4888
|
+
private readonly _config;
|
|
4889
|
+
private readonly _lastRaisedAt;
|
|
4890
|
+
private readonly dryOutboundTracks;
|
|
4891
|
+
private readonly dryInboundTracks;
|
|
4892
|
+
constructor(call: ObservedCall, config?: Partial<TrackDeliveryMismatchDetectorConfig>);
|
|
4893
|
+
close(): void;
|
|
4894
|
+
add(issue: ActiveClientIssue): void;
|
|
4895
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4896
|
+
get size(): number;
|
|
4897
|
+
clear(): void;
|
|
4898
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4899
|
+
update(): void;
|
|
4900
|
+
}
|
|
4901
|
+
|
|
4902
|
+
declare const TurnServerHealthTypes: {
|
|
4903
|
+
/** One TURN server's clients are in trouble while other servers' clients are fine. */
|
|
4904
|
+
readonly turnServerDegraded: "TURN_SERVER_DEGRADED";
|
|
4905
|
+
};
|
|
4906
|
+
type TurnServerHealthDetectorConfig = {
|
|
4907
|
+
/**
|
|
4908
|
+
* Clients a server must be carrying before its ratio means anything. Default `5`.
|
|
4909
|
+
*
|
|
4910
|
+
* With two relayed clients, "half are degraded" is one person having a bad time. Sensible range
|
|
4911
|
+
* `5`–`20`; raise it if you run many small TURN deployments, since each needs enough traffic to be
|
|
4912
|
+
* measurable on its own.
|
|
4913
|
+
*/
|
|
4914
|
+
minClientsPerServer: number;
|
|
4915
|
+
/**
|
|
4916
|
+
* Fraction of a server's clients that must have an open issue, `0`–`1`. Default `0.5`.
|
|
4917
|
+
*
|
|
4918
|
+
* Typical `0.4`–`0.7`. Remember each finding also carries the *other* servers' ratios, so the
|
|
4919
|
+
* threshold is not doing the comparison on its own — a server at 50% next to peers at 45% reads very
|
|
4920
|
+
* differently from one next to peers at 3%. Below `0.3` you will report servers that are merely
|
|
4921
|
+
* carrying unlucky clients.
|
|
4922
|
+
*/
|
|
4923
|
+
degradedRatioThreshold: number;
|
|
4924
|
+
/**
|
|
4925
|
+
* Which client issue types count as "in trouble". Empty (the default) means **any** open issue.
|
|
4926
|
+
*
|
|
4927
|
+
* The permissive default is deliberate and unusual for this library: the question is not *what* is
|
|
4928
|
+
* wrong with each client but whether trouble clusters on one relay, and a relay problem shows up as
|
|
4929
|
+
* whatever symptom each client happens to notice first. Narrow it to network types
|
|
4930
|
+
* (`congestion`, `ice-disconnected`) if endpoint issues like `cpulimitation` are common enough in
|
|
4931
|
+
* your fleet to blur the comparison between servers.
|
|
4932
|
+
*/
|
|
4933
|
+
issueTypes: string[];
|
|
4934
|
+
/**
|
|
4935
|
+
* Consecutive `observer.update()` ticks the condition must hold before raising. Default `2`.
|
|
4936
|
+
*
|
|
4937
|
+
* The de-bounce. `1` reacts immediately and will fire on a single tick where several clients
|
|
4938
|
+
* happened to be mid-reconnect; `2`–`3` costs a tick or two of delay and removes most of that.
|
|
4939
|
+
* Note this counts *ticks*, not time, so how long it actually waits depends on your sample rate.
|
|
4940
|
+
*/
|
|
4941
|
+
consecutiveTicks: number;
|
|
4942
|
+
/**
|
|
4943
|
+
* Re-arm time per server (ms). Default `60_000`.
|
|
4944
|
+
*
|
|
4945
|
+
* Shorter than the outage detector's, because degradation is a condition you may want re-reported as
|
|
4946
|
+
* it persists or worsens, not a single event. Typical `60_000`–`300_000`.
|
|
4947
|
+
*/
|
|
4948
|
+
cooldownMs: number;
|
|
4949
|
+
};
|
|
4950
|
+
/** The per-server view this detector builds. */
|
|
4951
|
+
type TurnServerHealth = {
|
|
4952
|
+
serverUrl: string;
|
|
4953
|
+
/** Distinct clients whose media is relayed through this server. */
|
|
4954
|
+
clients: number;
|
|
4955
|
+
/** Of those, how many currently have at least one open issue. */
|
|
4956
|
+
degradedClients: number;
|
|
4957
|
+
degradedRatio: number;
|
|
4958
|
+
affectedClientIds: string[];
|
|
4959
|
+
/** The open issue types seen on this server's clients, most common first. */
|
|
4960
|
+
issueTypes: string[];
|
|
4961
|
+
};
|
|
4962
|
+
/**
|
|
4963
|
+
* An **observer-level** detector that groups relayed clients by the TURN server carrying them and
|
|
4964
|
+
* compares the servers against each other.
|
|
4965
|
+
*
|
|
4966
|
+
* Counting TURN usage is not useful on its own; knowing that `turn-eu-1` has 22 of 30 clients in
|
|
4967
|
+
* trouble while `turn-eu-2` has 1 of 34 is. Because the comparison spans calls it lives on
|
|
4968
|
+
* `observer.detectors` and raises `observer-issue` — one actionable alert instead of fifty
|
|
4969
|
+
* per-client ones. Each finding carries the other servers' ratios as context, since "half the
|
|
4970
|
+
* clients here are unhappy" only means something relative to the rest of the fleet.
|
|
4971
|
+
*
|
|
4972
|
+
* Whether a client is in trouble comes from **its own reported issues**, not from thresholds applied
|
|
4973
|
+
* here. The client already decides that far better than a server-side rule could; the value this
|
|
4974
|
+
* adds is the grouping — the dimension no endpoint can see.
|
|
4975
|
+
*
|
|
4976
|
+
* For a relay that has stopped serving entirely, see `TurnServerOutageDetector`: this detector needs
|
|
4977
|
+
* clients *on* the server to ask how many are unhappy, and an outage takes them away.
|
|
4978
|
+
*/
|
|
4979
|
+
declare class TurnServerHealthDetector implements Detector {
|
|
4980
|
+
private readonly _observer;
|
|
4981
|
+
static readonly NAME = "turn-server-health-detector";
|
|
4982
|
+
readonly name = "turn-server-health-detector";
|
|
4983
|
+
private readonly _config;
|
|
4984
|
+
private readonly _streaks;
|
|
4985
|
+
private readonly _lastRaisedAt;
|
|
4986
|
+
/** The per-server rollup computed on the most recent `update()`. */
|
|
4987
|
+
lastServers: TurnServerHealth[];
|
|
4988
|
+
constructor(_observer: Observer, config?: Partial<TurnServerHealthDetectorConfig>);
|
|
4989
|
+
update(): void;
|
|
4990
|
+
close(): void;
|
|
4991
|
+
private _serverHealth;
|
|
4992
|
+
}
|
|
4993
|
+
|
|
4994
|
+
declare const TurnServerOutageTypes: {
|
|
4995
|
+
/** One TURN server's relayed population collapsed while the rest of the fleet is fine. */
|
|
4996
|
+
readonly turnServerOutage: "TURN_SERVER_OUTAGE";
|
|
4997
|
+
};
|
|
4998
|
+
type TurnServerOutageDetectorConfig = {
|
|
4999
|
+
/**
|
|
5000
|
+
* Clients a server must have been carrying at its peak before its collapse means anything. Default
|
|
5001
|
+
* `5`.
|
|
5002
|
+
*
|
|
5003
|
+
* Below this, one or two people leaving looks like an outage. Sensible range `5`–`50`; the higher it
|
|
5004
|
+
* is the more confident the finding, and the more small deployments go unwatched.
|
|
5005
|
+
*/
|
|
5006
|
+
minClientsAtPeak: number;
|
|
5007
|
+
/**
|
|
5008
|
+
* Fraction of the peak population that must be gone or disrupted. Default `0.8` — an outage is
|
|
5009
|
+
* near-total by definition; partial degradation is `TurnServerHealthDetector`'s question.
|
|
5010
|
+
*/
|
|
5011
|
+
lossRatioThreshold: number;
|
|
5012
|
+
/**
|
|
5013
|
+
* Window the peak population is measured over (ms). Default `120_000`.
|
|
5014
|
+
*
|
|
5015
|
+
* Long enough to span a real outage's onset, short enough that yesterday's peak is not held against
|
|
5016
|
+
* today. Typical `60_000`–`600_000`. Too long and the natural end of a busy period reads as a
|
|
5017
|
+
* collapse; too short and a gradual failure never shows a peak to fall from.
|
|
5018
|
+
*/
|
|
5019
|
+
peakWindowMs: number;
|
|
5020
|
+
/**
|
|
5021
|
+
* Require a healthy **control group** — clients not relayed through this server that are still
|
|
5022
|
+
* connected — before blaming the server. Without this, a call ending, a fleet-wide network
|
|
5023
|
+
* event, or the observer shutting down all look exactly like a TURN outage. Default `true`.
|
|
5024
|
+
*/
|
|
5025
|
+
requireControlGroup: boolean;
|
|
5026
|
+
/**
|
|
5027
|
+
* Clients elsewhere before the control group is worth anything. Default `5`.
|
|
5028
|
+
*
|
|
5029
|
+
* If you run a single TURN server there is never a control group, so with `requireControlGroup: true`
|
|
5030
|
+
* this detector can never fire — which is correct rather than unfortunate: with one relay you cannot
|
|
5031
|
+
* distinguish "the relay died" from "everyone went home". Sensible range `5`–`20`.
|
|
5032
|
+
*/
|
|
5033
|
+
minControlGroupClients: number;
|
|
5034
|
+
/**
|
|
5035
|
+
* Fraction of the control group that must still be healthy, `0`–`1`. Default `0.7`.
|
|
5036
|
+
*
|
|
5037
|
+
* The evidence that the rest of the world is fine. Typical `0.6`–`0.9`. Set it too high and a
|
|
5038
|
+
* concurrent unrelated problem elsewhere masks a real outage; too low and a fleet-wide network event
|
|
5039
|
+
* gets blamed on whichever server lost clients first.
|
|
5040
|
+
*/
|
|
5041
|
+
controlGroupHealthyRatio: number;
|
|
5042
|
+
/**
|
|
5043
|
+
* Consecutive `observer.update()` ticks the condition must hold before raising. Default `2`.
|
|
5044
|
+
*
|
|
5045
|
+
* Counts ticks, not time. `1` will fire on a single tick where a batch of clients happened to be
|
|
5046
|
+
* between samples; `2`–`4` is the useful range for something this consequential to declare.
|
|
5047
|
+
*/
|
|
5048
|
+
consecutiveTicks: number;
|
|
5049
|
+
/**
|
|
5050
|
+
* Re-arm time (ms) per server. Long by default (`300_000`) — an outage is one event, not one
|
|
5051
|
+
* per tick, and a server that stays down would otherwise alert forever.
|
|
5052
|
+
*/
|
|
5053
|
+
cooldownMs: number;
|
|
5054
|
+
};
|
|
5055
|
+
/**
|
|
5056
|
+
* Detects a **TURN server outage** — a relay that has stopped serving — by watching its client
|
|
5057
|
+
* population collapse while the rest of the fleet carries on.
|
|
5058
|
+
*
|
|
5059
|
+
* This is the case its sibling `TurnServerHealthDetector` structurally *cannot* see, and the
|
|
5060
|
+
* distinction is worth being precise about. That detector groups clients by the server relaying them
|
|
5061
|
+
* and asks how many are reporting issues. It needs clients on the server to ask the question. When a
|
|
5062
|
+
* TURN server goes down completely, allocation fails: existing sessions drop, and new clients never
|
|
5063
|
+
* obtain a relay candidate through it at all, so they are never attributed to it. The server's
|
|
5064
|
+
* population goes to zero and the health detector falls silent for the worst possible reason — it
|
|
5065
|
+
* has nobody left to ask. Degradation makes clients unhappy; an outage makes them *disappear*.
|
|
5066
|
+
*
|
|
5067
|
+
* So the signal here is absence, measured against the server's own recent peak:
|
|
5068
|
+
*
|
|
5069
|
+
* - clients gone entirely (their relayed peer connections closed, or they re-negotiated onto a
|
|
5070
|
+
* different path), plus
|
|
5071
|
+
* - clients still attributed to the server whose ICE or connection state is `disconnected` /
|
|
5072
|
+
* `failed` / `closed` — the ones mid-collapse, which is what you catch if you look during the
|
|
5073
|
+
* outage rather than after it.
|
|
5074
|
+
*
|
|
5075
|
+
* ### The control group is the whole design
|
|
5076
|
+
*
|
|
5077
|
+
* Absence is a dangerous signal: a call ending, everyone going home at 6pm, a fleet-wide network
|
|
5078
|
+
* event, and the observer itself shutting down all produce exactly the same collapse. The detector
|
|
5079
|
+
* therefore refuses to blame a server unless clients **not** relayed through it are demonstrably
|
|
5080
|
+
* still connected — `requireControlGroup`, on by default. "Everyone on `turn-eu-1` vanished" is
|
|
5081
|
+
* ambiguous; "everyone on `turn-eu-1` vanished while 200 clients elsewhere are fine" is an outage.
|
|
5082
|
+
*
|
|
5083
|
+
* That comparison is only available to something watching every call at once, which is why this is
|
|
5084
|
+
* an observer-level detector raising `observer-issue` — one alert for the fleet, not one per
|
|
5085
|
+
* abandoned call.
|
|
5086
|
+
*
|
|
5087
|
+
* ### Caveats worth knowing before you tune it
|
|
5088
|
+
*
|
|
5089
|
+
* Clients that fail over cleanly to a second TURN server still count as lost here, which is
|
|
5090
|
+
* correct — the server did stop serving them — but it means a well-configured fleet with automatic
|
|
5091
|
+
* failover reports outages that users never felt. That is the intended behaviour: the failover
|
|
5092
|
+
* worked *and* the server is down are both true, and you want to know the second one.
|
|
5093
|
+
*
|
|
5094
|
+
* A genuinely quiet server (last call of the day ends) is suppressed by the control group, not by
|
|
5095
|
+
* the collapse test. If you run a small deployment where the control group is routinely below
|
|
5096
|
+
* `minControlGroupClients`, this detector will stay quiet — prefer alerting on your TURN server's
|
|
5097
|
+
* own health checks there, since a handful of clients cannot distinguish these cases.
|
|
5098
|
+
*/
|
|
5099
|
+
declare class TurnServerOutageDetector implements Detector {
|
|
5100
|
+
private readonly _observer;
|
|
5101
|
+
static readonly NAME = "turn-server-outage-detector";
|
|
5102
|
+
readonly name = "turn-server-outage-detector";
|
|
5103
|
+
private readonly _config;
|
|
5104
|
+
/** serverUrl -> recent population observations, used to derive the windowed peak. */
|
|
5105
|
+
private readonly _peaks;
|
|
5106
|
+
private readonly _streaks;
|
|
5107
|
+
private readonly _lastRaisedAt;
|
|
5108
|
+
constructor(_observer: Observer, config?: Partial<TurnServerOutageDetectorConfig>);
|
|
5109
|
+
update(): void;
|
|
5110
|
+
close(): void;
|
|
5111
|
+
/** Distinct clients on a server, split by whether their relayed transport is actually up. */
|
|
5112
|
+
private _populationOf;
|
|
5113
|
+
/** Record this tick's population and return the peak across `peakWindowMs`. */
|
|
5114
|
+
private _recordAndPeak;
|
|
5115
|
+
/**
|
|
5116
|
+
* Everyone *not* relayed through `serverUrl`: clients on other TURN servers plus every client
|
|
5117
|
+
* the observer knows about that isn't relayed at all. The healthy share of that group is what
|
|
5118
|
+
* separates "this server broke" from "everything broke".
|
|
5119
|
+
*/
|
|
5120
|
+
private _controlGroup;
|
|
5121
|
+
}
|
|
5122
|
+
|
|
5123
|
+
declare const UnconsumedTrackTypes: {
|
|
5124
|
+
/** A track is being published to the SFU that nobody is subscribed to — pure wasted uplink. */
|
|
5125
|
+
readonly unconsumedPublishedTrack: "UNCONSUMED_PUBLISHED_TRACK";
|
|
5126
|
+
};
|
|
5127
|
+
type UnconsumedTrackDetectorConfig = {
|
|
5128
|
+
/**
|
|
5129
|
+
* How long a track must stay unconsumed **while still sending** before it is reported (ms).
|
|
5130
|
+
* Default `30_000`.
|
|
5131
|
+
*
|
|
5132
|
+
* This is the main guard against a false alarm, because a gap between publishing and the first
|
|
5133
|
+
* subscription is completely normal at join time — and again after every renegotiation. Sensible
|
|
5134
|
+
* range `15_000`–`120_000`. Too low and you report every join; too high and you tolerate wasted
|
|
5135
|
+
* uplink for longer than you need to. Waste is not an outage, so err high.
|
|
5136
|
+
*/
|
|
5137
|
+
minUnconsumedDurationInMs: number;
|
|
5138
|
+
/**
|
|
5139
|
+
* Ignore tracks sending below this bitrate (**bits per second**). Default `50_000` (50 kbps).
|
|
5140
|
+
*
|
|
5141
|
+
* The point of the detector is wasted bandwidth, and a track trickling keep-alive packets wastes
|
|
5142
|
+
* none worth an alert. Typical `20_000`–`100_000`: muted or paused tracks sit near zero, a real
|
|
5143
|
+
* video track is hundreds of kbps. Set it to `0` to report every unconsumed track regardless of
|
|
5144
|
+
* cost.
|
|
5145
|
+
*/
|
|
5146
|
+
minBitrate: number;
|
|
5147
|
+
/**
|
|
5148
|
+
* Re-arm time per track (ms). Default `300_000`.
|
|
5149
|
+
*
|
|
5150
|
+
* Long on purpose: an unconsumed track usually *stays* unconsumed, so a short cooldown means a
|
|
5151
|
+
* steady drip of the same finding for the life of the call. Typical `300_000`–`900_000`.
|
|
5152
|
+
*/
|
|
5153
|
+
cooldownMs: number;
|
|
5154
|
+
};
|
|
5155
|
+
/**
|
|
5156
|
+
* Finds tracks that are **published but consumed by nobody** — uplink and SFU ingress spent on media
|
|
5157
|
+
* that is never forwarded anywhere.
|
|
5158
|
+
*
|
|
5159
|
+
* This is the one detector that reads the resolver's *silence* as the signal: an outbound track with
|
|
5160
|
+
* an empty `remoteInboundTracks` set, still pushing packets. It reads `call.unconsumedOutboundTracks`,
|
|
5161
|
+
* which the resolver maintains as tracks gain and lose subscribers, so a healthy call costs one
|
|
5162
|
+
* `size === 0` check rather than a walk over every published track. The usual causes are a participant
|
|
5163
|
+
* publishing while everyone has them hidden or muted-in-UI, a simulcast layer no viewer's bandwidth
|
|
5164
|
+
* ever selects, or an application that forgot to stop a track after the last subscriber left.
|
|
5165
|
+
*
|
|
5166
|
+
* It is deliberately slow to fire: `minUnconsumedDurationInMs` must elapse with the track still
|
|
5167
|
+
* sending, because a brief gap between publishing and the first subscription is completely normal at
|
|
5168
|
+
* join time.
|
|
5169
|
+
*
|
|
5170
|
+
* ### Careful: this detector is only sound with a resolver
|
|
5171
|
+
*
|
|
5172
|
+
* "No subscribers" and "no resolver configured" produce the identical observation — an empty link
|
|
5173
|
+
* set. Without a `RemoteTrackResolver` this would report *every* published track in the call as
|
|
5174
|
+
* unconsumed, so it checks `call.remoteTrackResolver` at runtime and does nothing without one.
|
|
5175
|
+
*/
|
|
5176
|
+
declare class UnconsumedTrackDetector implements Detector {
|
|
5177
|
+
private readonly call;
|
|
5178
|
+
static readonly NAME = "unconsumed-track-detector";
|
|
5179
|
+
readonly name = "unconsumed-track-detector";
|
|
5180
|
+
readonly config: UnconsumedTrackDetectorConfig;
|
|
5181
|
+
/** trackId -> when it was first seen sending with no subscribers. */
|
|
5182
|
+
private readonly _unconsumedSince;
|
|
5183
|
+
private readonly _lastRaisedAt;
|
|
5184
|
+
constructor(call: ObservedCall, config?: Partial<UnconsumedTrackDetectorConfig>);
|
|
5185
|
+
update(): void;
|
|
5186
|
+
close(): void;
|
|
5187
|
+
}
|
|
5188
|
+
|
|
5189
|
+
/**
|
|
5190
|
+
* Detectors that reason **across calls**, created once on the observer.
|
|
5191
|
+
*
|
|
5192
|
+
* Adding one means: give the class a `static readonly NAME`, add its entry here, and add a `case` to
|
|
5193
|
+
* `Observer.addObserverDetector`. This map is what types the call site — the config is checked
|
|
5194
|
+
* against the right detector and an unknown name won't compile.
|
|
5195
|
+
*/
|
|
5196
|
+
type AvailableObserverScopeDetectorsConfigs = {
|
|
5197
|
+
[SfuCongestionDetector.NAME]: SfuCongestionDetectorConfig;
|
|
5198
|
+
[ObserverConcurrentIssueDetector.NAME]: ObserverConcurrentIssueDetectorConfig;
|
|
5199
|
+
[ClientPopulationIssueDetector.NAME]: ClientPopulationIssueDetectorConfig;
|
|
5200
|
+
[TurnServerHealthDetector.NAME]: TurnServerHealthDetectorConfig;
|
|
5201
|
+
[TurnServerOutageDetector.NAME]: TurnServerOutageDetectorConfig;
|
|
5202
|
+
};
|
|
5203
|
+
/**
|
|
5204
|
+
* Detectors that reason **within one call**, created for every call the observer opens.
|
|
5205
|
+
*
|
|
5206
|
+
* Note there is no detector in both maps. "Is this meeting in trouble?" and "is our infrastructure in
|
|
5207
|
+
* trouble?" are different questions with different gates and different findings, so they are separate
|
|
5208
|
+
* classes — `CallConcurrentIssueDetector` and `ObserverConcurrentIssueDetector` — rather than one
|
|
5209
|
+
* class branching on what it was handed.
|
|
5210
|
+
*/
|
|
5211
|
+
type AvailableCallScopeDetectorsConfigs = {
|
|
5212
|
+
[UnconsumedTrackDetector.NAME]: UnconsumedTrackDetectorConfig;
|
|
5213
|
+
[TrackDeliveryMismatchDetector.NAME]: TrackDeliveryMismatchDetectorConfig;
|
|
5214
|
+
[CallConcurrentIssueDetector.NAME]: CallConcurrentIssueDetectorConfig;
|
|
5215
|
+
[IssueFanOutDetector.NAME]: IssueFanOutDetectorConfig;
|
|
5216
|
+
[PublisherFaultCorroborationDetector.NAME]: PublisherFaultCorroborationDetectorConfig;
|
|
5217
|
+
};
|
|
5218
|
+
type AvailableDetectorsConfigs = AvailableObserverScopeDetectorsConfigs | AvailableCallScopeDetectorsConfigs;
|
|
5219
|
+
declare class Detectors {
|
|
5220
|
+
private _detectors;
|
|
5221
|
+
constructor(...detectors: Detector[]);
|
|
5222
|
+
/**
|
|
5223
|
+
* Every registered detector, in registration order.
|
|
5224
|
+
*
|
|
5225
|
+
* This is **the** way to get hold of an instance: `addDetector` / `addObserverDetector` are
|
|
5226
|
+
* chainable and return the owning entity, so the registry is where instances live. Read it to
|
|
5227
|
+
* inspect a detector's state, or to pick one out and {@link remove} it.
|
|
5228
|
+
*
|
|
5229
|
+
* A copy, not the live array — a caller iterating this while removing would otherwise skip
|
|
5230
|
+
* entries, and that is exactly what "remove the ones that look like X" does.
|
|
5231
|
+
*/
|
|
5232
|
+
get instances(): Detector[];
|
|
5233
|
+
/** Iterate the registry directly: `for (const detector of call.detectors)`. */
|
|
5234
|
+
[Symbol.iterator](): IterableIterator<Detector>;
|
|
5235
|
+
/** The names in registration order. Duplicates are meaningful — see {@link getAll}. */
|
|
5236
|
+
get listOfNames(): string[];
|
|
5237
|
+
get size(): number;
|
|
5238
|
+
add(detector: Detector): void;
|
|
5239
|
+
/** The first detector registered under `name`. See {@link getAll} when several can share one. */
|
|
5240
|
+
get(name: string): Detector | undefined;
|
|
5241
|
+
/**
|
|
5242
|
+
* Every detector registered under `name`.
|
|
5243
|
+
*
|
|
5244
|
+
* More than one is legitimate: `ClientPopulationIssueDetector` is meant to be added once per
|
|
5245
|
+
* `groupBy` axis, and two instances of it share a name.
|
|
5246
|
+
*/
|
|
5247
|
+
getAll(name: string): Detector[];
|
|
5248
|
+
has(name: string): boolean;
|
|
5249
|
+
/** Remove one specific instance. Returns `false` if it was not registered here. */
|
|
5250
|
+
remove(detector: Detector): boolean;
|
|
5251
|
+
/**
|
|
5252
|
+
* Remove **every** detector registered under `name`, returning how many were removed.
|
|
5253
|
+
*
|
|
5254
|
+
* All of them rather than the first, because a name can legitimately be registered more than once
|
|
5255
|
+
* (see {@link getAll}) and "remove the `client-population-issue-detector`" cannot sensibly mean
|
|
5256
|
+
* "remove whichever axis happens to be first in the array". Removing all of them is the only
|
|
5257
|
+
* behaviour that leaves the registry in a state the caller can predict from the name alone.
|
|
5258
|
+
*
|
|
5259
|
+
* Each removed detector gets `close()`, so trackers unsubscribe from the issue registry, bus
|
|
5260
|
+
* listeners drop, and timers clear — a detector removed without closing keeps being fed issues
|
|
5261
|
+
* forever.
|
|
5262
|
+
*/
|
|
5263
|
+
removeByName(name: string): number;
|
|
5264
|
+
update(): void;
|
|
5265
|
+
clear(): void;
|
|
5266
|
+
private _close;
|
|
5267
|
+
}
|
|
5268
|
+
|
|
5269
|
+
/**
|
|
5270
|
+
* The set of client issues currently believed to be **open**, plus the fan-out that pushes them to
|
|
5271
|
+
* whoever asked for them.
|
|
5272
|
+
*
|
|
5273
|
+
* ### Push, not poll
|
|
5274
|
+
*
|
|
5275
|
+
* A detector does not scan for the issues it cares about; it registers as an
|
|
5276
|
+
* {@link ActiveIssueTracker} for the types it consumes and is handed them as they open and close.
|
|
5277
|
+
* The cost of a detector is then proportional to the issues it actually receives, not to the number
|
|
5278
|
+
* of participants — a healthy 500-client fleet does no work per tick.
|
|
5279
|
+
*
|
|
5280
|
+
* ```ts
|
|
5281
|
+
* observer.activeIssuesRegistry.addIssueTracker('congestion', detector);
|
|
5282
|
+
* ```
|
|
5283
|
+
*
|
|
5284
|
+
* There is **no wildcard**. A tracker names the types it consumes, and nothing else reaches it. "Feed
|
|
5285
|
+
* me everything and I'll work out what matters" pushes the decision from the application — which
|
|
5286
|
+
* knows its client build and its issue vocabulary — onto a detector that has to guess, and it makes
|
|
5287
|
+
* the cost of a subscription unbounded and invisible. If a detector should watch five issue types,
|
|
5288
|
+
* the caller lists five issue types.
|
|
5289
|
+
*
|
|
5290
|
+
* ### Two levels
|
|
5291
|
+
*
|
|
5292
|
+
* Every call owns a registry constructed with the observer's as its `parent`. An add or delete
|
|
5293
|
+
* touches both, so a call-scoped tracker sees only that call's issues while an observer-scoped one
|
|
5294
|
+
* sees the fleet — without either side iterating the other. The child keeps **its own** storage:
|
|
5295
|
+
* `size` is this scope's count, and {@link clear} (called when the call closes) removes only this
|
|
5296
|
+
* scope's issues from the parent and never touches the parent's tracker registrations.
|
|
5297
|
+
*
|
|
5298
|
+
* ### Only keyed issues arrive here
|
|
5299
|
+
*
|
|
5300
|
+
* An issue without a `key` has no lifecycle — nothing can ever close it — so treating it as "active"
|
|
5301
|
+
* would mean holding a symptom that may have ended long ago. Keyless issues stay one-shot: emitted
|
|
5302
|
+
* as `client-issue`, never registered. `client-monitor-js` >= 4.6.0 sends `key` on everything
|
|
5303
|
+
* stateful.
|
|
5304
|
+
*/
|
|
5305
|
+
declare class ActiveIssuesRegistry implements ActiveIssueTracker {
|
|
5306
|
+
private readonly parent?;
|
|
5307
|
+
private readonly issues;
|
|
5308
|
+
private readonly typesToTrackers;
|
|
5309
|
+
constructor(parent?: ActiveIssueTracker | undefined);
|
|
5310
|
+
get size(): number;
|
|
5311
|
+
/**
|
|
5312
|
+
* The open issues in this scope, in insertion order.
|
|
5313
|
+
*
|
|
5314
|
+
* Insertion order is age order (`observedAt` is assigned on insert), which is what lets a consumer
|
|
5315
|
+
* stop at the first entry newer than its cutoff instead of scanning the whole set.
|
|
5316
|
+
*/
|
|
5317
|
+
values(): IterableIterator<ActiveClientIssue>;
|
|
5318
|
+
[Symbol.iterator](): IterableIterator<ActiveClientIssue>;
|
|
5319
|
+
has(issue: ActiveClientIssue): boolean;
|
|
5320
|
+
add(issue: ActiveClientIssue): this;
|
|
5321
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
5322
|
+
/** Feed `tracker` every issue of `type` as it opens and closes. One call per type; no wildcard. */
|
|
5323
|
+
addIssueTracker(type: string, tracker: ActiveIssueTracker): this;
|
|
5324
|
+
removeIssueTracker(tracker: ActiveIssueTracker): this;
|
|
5325
|
+
/**
|
|
5326
|
+
* Drop every issue in this scope, e.g. because the call closed.
|
|
5327
|
+
*
|
|
5328
|
+
* Deletes through {@link delete} so the parent sheds exactly this scope's issues. Tracker
|
|
5329
|
+
* *registrations* survive: a detector subscribed to the observer's registry must keep receiving
|
|
5330
|
+
* issues after any one call ends.
|
|
5331
|
+
*/
|
|
5332
|
+
clear(): void;
|
|
5333
|
+
private _trackIssue;
|
|
5334
|
+
private _untrackIssue;
|
|
5335
|
+
/**
|
|
5336
|
+
* Apply `apply` to every tracker interested in `type`.
|
|
5337
|
+
*
|
|
5338
|
+
* A tracker throwing must not abort the fan-out: the issue has already been added to (or removed
|
|
5339
|
+
* from) this registry, so a partial dispatch would leave the remaining trackers permanently out of
|
|
5340
|
+
* step with it. One broken detector should not desynchronise the others.
|
|
5341
|
+
*/
|
|
5342
|
+
private _trackersOf;
|
|
5343
|
+
private _safely;
|
|
5344
|
+
}
|
|
5345
|
+
|
|
5346
|
+
type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
5347
|
+
callId: string;
|
|
5348
|
+
appData?: AppData;
|
|
5349
|
+
closeCallIfEmptyForMs?: number;
|
|
5350
|
+
/**
|
|
5351
|
+
* When `true`, the call's `update()` is invoked whenever a client accepts a sample. When `false`, it is not.
|
|
5352
|
+
*
|
|
5353
|
+
* DEFAULT: `true` — the call is updated on every client sample, which is the most common use case.
|
|
5354
|
+
*/
|
|
5355
|
+
autoUpdateOnClientUpdate?: boolean;
|
|
5356
|
+
};
|
|
5357
|
+
type ObservedCallEvents = {
|
|
5358
|
+
update: [];
|
|
5359
|
+
newclient: [ObservedClient];
|
|
5360
|
+
empty: [];
|
|
5361
|
+
'not-empty': [];
|
|
5362
|
+
close: [];
|
|
5363
|
+
};
|
|
5364
|
+
declare interface ObservedCall {
|
|
5365
|
+
on<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
|
|
5366
|
+
off<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
|
|
5367
|
+
once<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
|
|
5368
|
+
emit<U extends keyof ObservedCallEvents>(event: U, ...args: ObservedCallEvents[U]): boolean;
|
|
5369
|
+
}
|
|
5370
|
+
declare class ObservedCall<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
|
|
5371
|
+
readonly observer: Observer;
|
|
5372
|
+
readonly activeIssuesRegistry: ActiveIssuesRegistry;
|
|
5373
|
+
scoreCalculator: ScoreCalculator;
|
|
5374
|
+
readonly detectors: Detectors;
|
|
5375
|
+
readonly callId: string;
|
|
5376
|
+
readonly observedClients: Map<string, ObservedClient<Record<string, unknown>>>;
|
|
5377
|
+
readonly clientsUsedTurn: Set<string>;
|
|
5378
|
+
readonly calculatedScore: CalculatedScore;
|
|
5379
|
+
remoteTrackResolver?: RemoteTrackResolver;
|
|
5380
|
+
/**
|
|
5381
|
+
* The accumulating record of this call's life, or `undefined` when no summary was configured.
|
|
5382
|
+
*
|
|
5383
|
+
* Live — read it at any point during the call. It is also delivered once on `call-summary` when
|
|
5384
|
+
* the call closes. See `CallSummary`: an absent section means "not collected", never "nothing
|
|
5385
|
+
* happened".
|
|
5386
|
+
*/
|
|
5387
|
+
summary?: CallSummary;
|
|
5388
|
+
/**
|
|
5389
|
+
* Published tracks that currently have **no** subscriber linked to them.
|
|
5390
|
+
*
|
|
5391
|
+
* Maintained by the `RemoteTrackResolver` at the exact moments a track gains or loses its last
|
|
5392
|
+
* subscriber — the only moments the answer can change. `UnconsumedTrackDetector` reads this
|
|
5393
|
+
* instead of walking every published track in the call, so in a healthy call (where the set is
|
|
5394
|
+
* empty) it does no work at all.
|
|
5395
|
+
*
|
|
5396
|
+
* Empty when no resolver is configured: without links, "no subscribers" is unknowable.
|
|
5397
|
+
*/
|
|
5398
|
+
readonly unconsumedOutboundTracks: Set<ObservedOutboundTrack>;
|
|
5399
|
+
totalAddedClients: number;
|
|
5400
|
+
totalRemovedClients: number;
|
|
5401
|
+
numberOfIssues: number;
|
|
5402
|
+
numberOfPeerConnections: number;
|
|
5403
|
+
numberOfInboundRtpStreams: number;
|
|
5404
|
+
numberOfOutboundRtpStreams: number;
|
|
5405
|
+
numberOfDataChannels: number;
|
|
5406
|
+
maxNumberOfClients: number;
|
|
5407
|
+
deltaNumberOfIssues: number;
|
|
5408
|
+
appData: AppData;
|
|
5409
|
+
closed: boolean;
|
|
5410
|
+
startedAt?: number;
|
|
5411
|
+
endedAt?: number;
|
|
5412
|
+
closedAt?: number;
|
|
5413
|
+
readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs' | 'autoUpdateOnClientUpdate'>;
|
|
5414
|
+
/** Ancestry base shared by all Observer-bus events originating at this call. */
|
|
5415
|
+
readonly eventScope: ObservedCallScope;
|
|
5416
|
+
private closeTimer?;
|
|
5417
|
+
constructor(settings: ObservedCallSettings<AppData>, observer: Observer, activeIssuesRegistry: ActiveIssuesRegistry);
|
|
5418
|
+
get numberOfClients(): number;
|
|
5419
|
+
get score(): number | undefined;
|
|
5420
|
+
/**
|
|
5421
|
+
* Build a call-scoped detector onto this call. Chainable.
|
|
5422
|
+
*
|
|
5423
|
+
* To get a handle on what was built — to inspect it, or to remove that exact instance later — read
|
|
5424
|
+
* it back off the registry: `call.detectors.getAll(name)`, or `call.detectors.instances`.
|
|
5425
|
+
*/
|
|
5426
|
+
addDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
|
|
5427
|
+
/**
|
|
5428
|
+
* Start accumulating this call's summary, if the observer was configured for summaries.
|
|
5429
|
+
*
|
|
5430
|
+
* Called by `createObservedCall`; you should not need it. It takes no configuration of its own on
|
|
5431
|
+
* purpose: the collector subscribes to exactly the events the observer's `include` requires, so a
|
|
5432
|
+
* per-call section outside that set would be created and then never written to — an empty section
|
|
5433
|
+
* that reads as "nothing happened". One shape per observer is the only shape that can be filled.
|
|
5434
|
+
*
|
|
5435
|
+
* The collector builds it rather than this method, so the resolved configuration never has to
|
|
5436
|
+
* leave the one object that owns it. Returns `undefined` when summaries are off, and is
|
|
5437
|
+
* idempotent: an existing summary is kept, not restarted.
|
|
5438
|
+
*/
|
|
5439
|
+
enableSummary(): CallSummary | undefined;
|
|
5440
|
+
/**
|
|
5441
|
+
* Remove a detector from **this call** by name, returning how many were removed.
|
|
5442
|
+
*
|
|
5443
|
+
* **Every** instance under the name goes — a name can legitimately be registered more than once.
|
|
5444
|
+
* When you want one of them specifically, go through the registry, which deals in instances:
|
|
5445
|
+
*
|
|
5446
|
+
* ```ts
|
|
5447
|
+
* const [ first ] = call.detectors.getAll('issue-fan-out-detector');
|
|
5448
|
+
*
|
|
5449
|
+
* call.detectors.remove(first);
|
|
5450
|
+
* ```
|
|
5451
|
+
*
|
|
5452
|
+
* Either route `close()`s the detector, so it unsubscribes from `activeIssuesRegistry` — without
|
|
5453
|
+
* that the registry keeps feeding a detector nobody is running any more, and its tracked set grows
|
|
5454
|
+
* for the life of the call.
|
|
5455
|
+
*
|
|
5456
|
+
* To stop building it on *future* calls too, use `observer.removeCallDetector(name)`.
|
|
5457
|
+
*/
|
|
5458
|
+
removeDetector(name: keyof AvailableCallScopeDetectorsConfigs): number;
|
|
5459
|
+
/**
|
|
5460
|
+
* Raise a call-level (server-side) finding; surfaced on the Observer bus as `call-issue`.
|
|
5461
|
+
*
|
|
5462
|
+
* `payload` is an **object** and holds evidence only — it is delivered to an in-process handler,
|
|
5463
|
+
* so there is nothing to serialise for. `scope` is stamped here, and the `callId` is already on
|
|
5464
|
+
* the event, so neither belongs in the payload. Put the interpretation in `conclusion`.
|
|
5465
|
+
*/
|
|
5466
|
+
addIssue(issue: Omit<CallIssue, 'scope'>): void;
|
|
5467
|
+
close(): void;
|
|
5468
|
+
getObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(clientId: string): ObservedClient<ClientAppData> | undefined;
|
|
5469
|
+
createObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>, acceptCtx?: AcceptContext): ObservedClient<ClientAppData> | undefined;
|
|
5470
|
+
getOrCreateObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>, acceptCtx?: AcceptContext): ObservedClient<ClientAppData> | undefined;
|
|
5471
|
+
update(context?: AcceptContext): void;
|
|
5472
|
+
private _onClientUpdate;
|
|
5473
|
+
private _clientJoined;
|
|
5474
|
+
private _clientLeft;
|
|
5475
|
+
/** Emit an Observer-bus event scoped to this call. */
|
|
5476
|
+
private _notify;
|
|
5477
|
+
}
|
|
5478
|
+
|
|
5479
|
+
type Middleware<T> = (input: T, next: (nextInput: T) => void) => void;
|
|
5480
|
+
interface Processor<T> {
|
|
5481
|
+
finalCallback?: Callback<T>;
|
|
5482
|
+
process(value: T): void;
|
|
5483
|
+
addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
5484
|
+
removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
5485
|
+
}
|
|
5486
|
+
type Callback<T> = (input: T) => void;
|
|
5487
|
+
declare class MiddlewareProcessor<T> implements Processor<T> {
|
|
5488
|
+
private stack;
|
|
5489
|
+
finalCallback?: Callback<T>;
|
|
5490
|
+
addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
5491
|
+
removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
|
|
5492
|
+
process(value: T): void;
|
|
5493
|
+
}
|
|
5494
|
+
|
|
5495
|
+
/** Raised once if one receiver turns out to be dragging a publisher down for everyone. */
|
|
5496
|
+
declare const LOWEST_COMMON_DENOMINATOR_ISSUE = "WORST_RECEIVER_CONTAGION";
|
|
5497
|
+
/** The measurements behind a decided verdict — everything needed to check the call yourself. */
|
|
5498
|
+
type SimulcastReceiverEvidence = {
|
|
5499
|
+
callId: string;
|
|
5500
|
+
trackId: string;
|
|
5501
|
+
publisherClientId: string;
|
|
5502
|
+
worstReceiverClientId: string;
|
|
5503
|
+
publisherBitrate: number;
|
|
5504
|
+
worstReceiverBitrate: number;
|
|
5505
|
+
medianReceiverBitrate: number;
|
|
5506
|
+
/** How closely the publisher's bitrate followed the **worst** receiver's, `0..1`. */
|
|
5507
|
+
trackingWithWorst: number;
|
|
5508
|
+
/** The same against the **median** receiver — the control. */
|
|
5509
|
+
trackingWithMedian: number;
|
|
5510
|
+
};
|
|
5511
|
+
type SimulcastReceiverReportPayload = ({
|
|
5512
|
+
/** The publisher held up while one receiver lagged: layers are being chosen per consumer. */
|
|
5513
|
+
verdict: 'layer-decided-per-receiver';
|
|
5514
|
+
evidence: SimulcastReceiverEvidence;
|
|
5515
|
+
} | {
|
|
5516
|
+
/** The publisher tracked its worst receiver: everyone is getting the lowest common denominator. */
|
|
5517
|
+
verdict: 'layer-decided-lowest-common-denominator';
|
|
5518
|
+
evidence: SimulcastReceiverEvidence;
|
|
5519
|
+
} | {
|
|
5520
|
+
/** Gave up without the conditions needed to judge. **Not a pass.** */
|
|
5521
|
+
verdict: 'inconclusive';
|
|
5522
|
+
reason: string;
|
|
5523
|
+
}) & {
|
|
5524
|
+
startedAt: number;
|
|
5525
|
+
/** How many times the check actually ran — i.e. how much the verdict is worth. */
|
|
5526
|
+
checks: number;
|
|
5527
|
+
};
|
|
5528
|
+
type SimulcastReceiverValidatorConfig = {
|
|
5529
|
+
/**
|
|
5530
|
+
* Receivers a published track needs before the comparison means anything. Default `3`.
|
|
5531
|
+
*
|
|
5532
|
+
* The question is whether *one* receiver drags *the others* down, which needs at least one other to
|
|
5533
|
+
* be dragged — so `3` gives a worst receiver plus two to compare against. `2` is the technical
|
|
5534
|
+
* minimum but makes "median of the others" a single number. Sensible range `3`–`5`.
|
|
5535
|
+
*/
|
|
5536
|
+
minReceivers: number;
|
|
5537
|
+
/**
|
|
5538
|
+
* How long bitrates are correlated over (ms). Default `10_000`.
|
|
5539
|
+
*
|
|
5540
|
+
* Long enough to contain a real adaptation response — the publisher's encoder reacting to a
|
|
5541
|
+
* bandwidth estimate takes seconds, not milliseconds. Typical `10_000`–`30_000`. Too short and you
|
|
5542
|
+
* catch transient jitter rather than a sustained relationship; too long and a genuine change is
|
|
5543
|
+
* averaged out by the healthy period around it.
|
|
5544
|
+
*/
|
|
5545
|
+
windowMs: number;
|
|
5546
|
+
/**
|
|
5547
|
+
* Samples needed inside the window before it can be judged. Default `5`.
|
|
5548
|
+
*
|
|
5549
|
+
* A correlation over two or three points is meaningless. Combined with `windowMs` this implies a
|
|
5550
|
+
* sampling period: 5 samples in 10 s needs clients reporting at least every ~2 s. If your collector
|
|
5551
|
+
* is slower, widen `windowMs` rather than lowering this.
|
|
5552
|
+
*/
|
|
5553
|
+
minSamples: number;
|
|
5554
|
+
/**
|
|
5555
|
+
* The worst receiver must be at most this share of the median receiver, `0`–`1`. Default `0.5`.
|
|
5556
|
+
*
|
|
5557
|
+
* The precondition, not the finding: unless somebody is genuinely doing much worse than the rest,
|
|
5558
|
+
* there is nothing for the publisher to be dragged *by* and the check has nothing to look at.
|
|
5559
|
+
* Typical `0.4`–`0.7`. Higher makes the check run more often on weaker evidence; lower means it
|
|
5560
|
+
* rarely finds a qualifying situation at all.
|
|
5561
|
+
*/
|
|
5562
|
+
outlierRatioThreshold: number;
|
|
5563
|
+
/**
|
|
5564
|
+
* How closely the publisher must track the worst receiver to count as dragged, `0`–`1`. Default
|
|
5565
|
+
* `0.8`.
|
|
5566
|
+
*
|
|
5567
|
+
* This is the finding: the publisher sending at ≥80% of the *worst* receiver's rate means it has
|
|
5568
|
+
* collapsed to the lowest common denominator instead of serving everyone else properly. Typical
|
|
5569
|
+
* `0.7`–`0.9`. Toward `1` you only catch total collapse; below ~`0.6` normal encoder behaviour can
|
|
5570
|
+
* look like dragging.
|
|
5571
|
+
*/
|
|
5572
|
+
trackingRatioThreshold: number;
|
|
5573
|
+
/**
|
|
5574
|
+
* Clean checks required before concluding per-receiver adaptation works. Default `3`.
|
|
5575
|
+
*
|
|
5576
|
+
* One clean check could be luck — the qualifying moment might simply not have been bad enough. This
|
|
5577
|
+
* is what stops a lucky sample from being reported as a pass, so raising it strengthens the verdict
|
|
5578
|
+
* at the cost of taking longer to reach one. Typical `3`–`10`.
|
|
5579
|
+
*/
|
|
5580
|
+
minChecks: number;
|
|
5581
|
+
};
|
|
5582
|
+
/**
|
|
5583
|
+
* Answers one question: **does this SFU adapt each receiver on its own, or does one bad receiver
|
|
5584
|
+
* drag the publisher down for everyone?**
|
|
5585
|
+
*
|
|
5586
|
+
* That is what simulcast (or SVC) exists to prevent. With several encodings available the server can
|
|
5587
|
+
* hand the struggling participant a lower layer and leave everyone else alone. Without it — or with
|
|
5588
|
+
* a server that relays RTCP end to end instead of terminating it, so the publisher's bandwidth
|
|
5589
|
+
* estimate collapses to the minimum across all receivers — the only way to serve the slowest
|
|
5590
|
+
* participant is to make the source send less, and everybody gets the lowest common denominator.
|
|
5591
|
+
*
|
|
5592
|
+
* The two causes are worth naming because the *observation* cannot separate them: the publisher's
|
|
5593
|
+
* bitrate tracking its worst receiver looks identical either way. What the check establishes is
|
|
5594
|
+
* whether per-receiver adaptation is happening at all. If the verdict is
|
|
5595
|
+
* `layer-decided-lowest-common-denominator`, look at both — is simulcast/SVC actually enabled with
|
|
5596
|
+
* layers selected per consumer, and is the SFU terminating receiver reports rather than forwarding
|
|
5597
|
+
* them?
|
|
5598
|
+
*
|
|
5599
|
+
* ### The control matters more than the correlation
|
|
5600
|
+
*
|
|
5601
|
+
* "Publisher follows worst receiver" alone proves nothing: when the whole call degrades together,
|
|
5602
|
+
* the publisher follows *everyone*, and that is ordinary adaptation working correctly. The verdict
|
|
5603
|
+
* only goes against the deployment when the publisher tracks the worst receiver **more closely than
|
|
5604
|
+
* it tracks the median** — the worst receiver is leading, not merely coinciding.
|
|
5605
|
+
*
|
|
5606
|
+
* Likewise, a window with no outlier in it is not evidence of health, it is an untested SFU: if
|
|
5607
|
+
* nobody is struggling, there is nothing for per-receiver adaptation to do. Those windows are
|
|
5608
|
+
* skipped and never counted in `checks`.
|
|
5609
|
+
*
|
|
5610
|
+
* ### Why a validator, not a detector
|
|
5611
|
+
*
|
|
5612
|
+
* This is a property of the SFU build and configuration, not of this moment: a server doing
|
|
5613
|
+
* per-receiver layer selection at 09:00 still is at 17:00. Re-deriving it every tick cannot produce
|
|
5614
|
+
* new information — it would only keep a sliding window alive per published track for the life of
|
|
5615
|
+
* every call. So it decides once, reports, and releases that state.
|
|
5616
|
+
*
|
|
5617
|
+
* ```ts
|
|
5618
|
+
* observer.on('validation-ready', ({ validator, report }) => {
|
|
5619
|
+
* if (validator !== 'simulcast-receivers' || !report.ready) return;
|
|
5620
|
+
* console.log(report.verdict); // 'layer-decided-per-receiver' | ... | 'inconclusive'
|
|
5621
|
+
* });
|
|
5622
|
+
*
|
|
5623
|
+
* observer.addValidator('simulcast-receivers');
|
|
5624
|
+
* onDeploy(() => observer.addValidator('simulcast-receivers')); // check again
|
|
5625
|
+
* ```
|
|
5626
|
+
*
|
|
5627
|
+
* ### `inconclusive` is not a pass
|
|
5628
|
+
*
|
|
5629
|
+
* The check only runs when a publisher has several receivers and one of them is far behind the
|
|
5630
|
+
* median; plenty of healthy deployments never present that. Concluding from the absence of a failure
|
|
5631
|
+
* would verify nothing, so `checks` counts the times the check genuinely ran, and a validator that
|
|
5632
|
+
* is cancelled (or whose observer closes) finishes `inconclusive` with the reason why.
|
|
5633
|
+
*/
|
|
5634
|
+
declare class SimulcastReceiverValidator implements Validator<SimulcastReceiverReportPayload> {
|
|
5635
|
+
private readonly _observer;
|
|
5636
|
+
readonly onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void;
|
|
5637
|
+
static readonly NAME: "simulcast-receivers";
|
|
5638
|
+
readonly name: "simulcast-receivers";
|
|
5639
|
+
readonly startedAt: number;
|
|
5640
|
+
report: ValidationReport<SimulcastReceiverReportPayload>;
|
|
5641
|
+
private readonly _config;
|
|
5642
|
+
private readonly _windows;
|
|
5643
|
+
private _checks;
|
|
5644
|
+
private _done;
|
|
5645
|
+
constructor(_observer: Observer, onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void, config?: Partial<SimulcastReceiverValidatorConfig>);
|
|
5646
|
+
/** How many times the comparison actually ran. `0` means nothing was established. */
|
|
5647
|
+
get checks(): number;
|
|
5648
|
+
/** Give up without a verdict, freeing anything waiting on this validator. */
|
|
5649
|
+
cancel(reason?: string): void;
|
|
5650
|
+
update(): void;
|
|
5651
|
+
/** Returns `true` when a verdict was reached and the caller should stop iterating. */
|
|
5652
|
+
private _inspect;
|
|
5653
|
+
/**
|
|
5654
|
+
* Settle on a verdict, exactly once.
|
|
5655
|
+
*
|
|
5656
|
+
* The guard is not paranoia: `onDone` removes this validator from the observer, and a second call
|
|
5657
|
+
* would emit a second `validation-ready` for a validator that is no longer registered — e.g. when
|
|
5658
|
+
* `observer.close()` cancels a validator that decided earlier in the same tick.
|
|
5659
|
+
*/
|
|
5660
|
+
private _finish;
|
|
5661
|
+
private _windowOf;
|
|
5662
|
+
}
|
|
5663
|
+
|
|
5664
|
+
/** Raised once if the resolver turns out never to link anything. */
|
|
5665
|
+
declare const UNRESOLVED_TRACK_LINKS_ISSUE = "REMOTE_TRACK_LINKS_UNRESOLVED";
|
|
5666
|
+
/** What the check actually saw, whichever way it went. */
|
|
5667
|
+
type RemoteTrackLinkEvidence = {
|
|
5668
|
+
/** Calls that presented the conditions for linking: a resolver, ≥2 clients, and inbound tracks. */
|
|
5669
|
+
eligibleCalls: number;
|
|
5670
|
+
/** Inbound tracks seen across those calls. */
|
|
5671
|
+
inboundTracks: number;
|
|
5672
|
+
/** Of those, how many were linked to the outbound track that published them. */
|
|
5673
|
+
linkedInboundTracks: number;
|
|
5674
|
+
/** `linkedInboundTracks / inboundTracks`. */
|
|
5675
|
+
linkedRatio: number;
|
|
5676
|
+
/** A call that presented the conditions, for the reader to go and look at. */
|
|
5677
|
+
exampleCallId?: string;
|
|
5678
|
+
};
|
|
5679
|
+
type RemoteTrackResolverReportPayload = ({
|
|
5680
|
+
/** The resolver is linking subscribers to publishers. The detectors that need links will work. */
|
|
5681
|
+
verdict: 'links-resolved';
|
|
5682
|
+
evidence: RemoteTrackLinkEvidence;
|
|
5683
|
+
} | {
|
|
5684
|
+
/** Every condition for linking was met, repeatedly, and nothing was ever linked. */
|
|
5685
|
+
verdict: 'no-links-resolved';
|
|
5686
|
+
evidence: RemoteTrackLinkEvidence;
|
|
5687
|
+
} | {
|
|
5688
|
+
/** Never saw a call that could have been linked. **Not a pass.** */
|
|
5689
|
+
verdict: 'inconclusive';
|
|
5690
|
+
reason: string;
|
|
5691
|
+
}) & {
|
|
5692
|
+
startedAt: number;
|
|
5693
|
+
/** How many times the check genuinely ran — i.e. how much the verdict is worth. */
|
|
5694
|
+
checks: number;
|
|
5695
|
+
};
|
|
5696
|
+
type RemoteTrackResolverValidatorConfig = {
|
|
5697
|
+
/**
|
|
5698
|
+
* Participants a call needs before it can plausibly have publisher↔subscriber links. Default `2`.
|
|
5699
|
+
*
|
|
5700
|
+
* A one-person call has nobody to subscribe to anyone, so counting it would dilute the ratio with
|
|
5701
|
+
* calls that *could not* have produced a link. `2` is the true minimum here and there is little
|
|
5702
|
+
* reason to raise it.
|
|
5703
|
+
*/
|
|
5704
|
+
minClients: number;
|
|
5705
|
+
/**
|
|
5706
|
+
* Inbound tracks a call must have before it counts as a check. Default `2`.
|
|
5707
|
+
*
|
|
5708
|
+
* Same idea: no subscribed tracks means nothing to link. Sensible range `2`–`5`.
|
|
5709
|
+
*/
|
|
5710
|
+
minInboundTracks: number;
|
|
5711
|
+
/**
|
|
5712
|
+
* Share of inbound tracks that must be linked to conclude the resolver works, `0`–`1`. Default
|
|
5713
|
+
* `0.5`.
|
|
5714
|
+
*
|
|
5715
|
+
* Deliberately lenient, because a partially-linked call is normal: tracks arrive before their
|
|
5716
|
+
* publisher is known, and simulcast layers or probing streams may have no publisher at all. The
|
|
5717
|
+
* question is "is this resolver wired up", not "is every track linked". Typical `0.3`–`0.7`. Raising
|
|
5718
|
+
* it toward `1` turns the check into a strictness audit and it will report failure on healthy
|
|
5719
|
+
* systems.
|
|
5720
|
+
*/
|
|
5721
|
+
linkedRatioThreshold: number;
|
|
5722
|
+
/**
|
|
5723
|
+
* Eligible calls to observe before concluding either way. Default `3`.
|
|
5724
|
+
*
|
|
5725
|
+
* One call could be a race — every track happening to arrive before its publisher. Typical `3`–`10`.
|
|
5726
|
+
* Note that a low value makes a *pass* less trustworthy than a failure: linking nothing repeatedly is
|
|
5727
|
+
* conclusive, linking things once might be luck.
|
|
5728
|
+
*/
|
|
5729
|
+
minChecks: number;
|
|
5730
|
+
};
|
|
5731
|
+
/**
|
|
5732
|
+
* Answers one question: **is the `RemoteTrackResolver` actually linking anything?**
|
|
5733
|
+
*
|
|
5734
|
+
* ### Why this is worth a validator
|
|
5735
|
+
*
|
|
5736
|
+
* Four things in this library are built on publisher↔subscriber links —
|
|
5737
|
+
* `IssueFanOutDetector`, `TrackDeliveryMismatchDetector`, `UnconsumedTrackDetector` and
|
|
5738
|
+
* `SimulcastReceiverValidator`. Every one of them checks `call.remoteTrackResolver` and, finding no
|
|
5739
|
+
* links, correctly does nothing rather than guessing.
|
|
5740
|
+
*
|
|
5741
|
+
* That is the right behaviour and it produces a nasty failure mode: a resolver wired to the wrong id
|
|
5742
|
+
* field, or a mediasoup `producerId` the application never attaches, leaves all four permanently
|
|
5743
|
+
* silent — and **silence is what a healthy deployment looks like too**. You would conclude your
|
|
5744
|
+
* calls were clean when in fact nothing was ever examined. This check exists to make that specific
|
|
5745
|
+
* mistake loud.
|
|
5746
|
+
*
|
|
5747
|
+
* ### `inconclusive` is not a pass
|
|
5748
|
+
*
|
|
5749
|
+
* A verdict is only reached from calls that *could* have been linked: a resolver configured, at
|
|
5750
|
+
* least `minClients` participants, and at least `minInboundTracks` inbound tracks present. A
|
|
5751
|
+
* one-to-one deployment, a lobby full of audio-only listeners, or a quiet period never presents
|
|
5752
|
+
* those conditions — and concluding "resolver works" from calls that had nothing to resolve would be
|
|
5753
|
+
* the very mistake this validator is here to catch. `checks` counts the eligible calls actually
|
|
5754
|
+
* seen; a validator cancelled before reaching `minChecks` finishes `inconclusive` and says so.
|
|
5755
|
+
*
|
|
5756
|
+
* ```ts
|
|
5757
|
+
* observer.on('validation-ready', ({ validator, report }) => {
|
|
5758
|
+
* if (validator !== 'remote-track-resolver' || !report.ready) return;
|
|
5759
|
+
* if (report.verdict === 'no-links-resolved') alert('resolver misconfigured — 4 detectors are inert');
|
|
5760
|
+
* });
|
|
5761
|
+
*
|
|
5762
|
+
* observer.addValidator('remote-track-resolver');
|
|
5763
|
+
* ```
|
|
5764
|
+
*
|
|
5765
|
+
* Run it once at start-up, and again after changing the resolver or the SFU's id scheme. Like every
|
|
5766
|
+
* validator it is one-shot: the answer is a property of the wiring, not of this moment.
|
|
5767
|
+
*/
|
|
5768
|
+
declare class RemoteTrackResolverValidator implements Validator<RemoteTrackResolverReportPayload> {
|
|
5769
|
+
private readonly _observer;
|
|
5770
|
+
readonly onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void;
|
|
5771
|
+
static readonly NAME: "remote-track-resolver";
|
|
5772
|
+
readonly name: "remote-track-resolver";
|
|
5773
|
+
readonly startedAt: number;
|
|
5774
|
+
report: ValidationReport<RemoteTrackResolverReportPayload>;
|
|
5775
|
+
private readonly _config;
|
|
5776
|
+
/** Accumulated across every eligible call seen, so one small call cannot decide alone. */
|
|
5777
|
+
private _eligibleCalls;
|
|
5778
|
+
private _inboundTracks;
|
|
5779
|
+
private _linkedInboundTracks;
|
|
5780
|
+
private _exampleCallId?;
|
|
5781
|
+
private _checks;
|
|
5782
|
+
private _done;
|
|
5783
|
+
constructor(_observer: Observer, onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void, config?: Partial<RemoteTrackResolverValidatorConfig>);
|
|
5784
|
+
/** How many eligible calls were actually examined. `0` means nothing was established. */
|
|
5785
|
+
get checks(): number;
|
|
5786
|
+
cancel(reason?: string): void;
|
|
5787
|
+
update(): void;
|
|
5788
|
+
private _evidence;
|
|
5789
|
+
private _finish;
|
|
5790
|
+
}
|
|
5791
|
+
|
|
5792
|
+
/** Raised once if the deployment is not actually delivering the codec it thinks it is. */
|
|
5793
|
+
declare const CODEC_MISMATCH_ISSUE = "CODEC_INCONSISTENCY";
|
|
5794
|
+
/** What the check saw across a call's participants. */
|
|
5795
|
+
type CodecEvidence = {
|
|
5796
|
+
callId: string;
|
|
5797
|
+
kind: 'audio' | 'video';
|
|
5798
|
+
/** Every mime type in use in that call, most common first — e.g. `[ 'video/VP8', 'video/H264' ]`. */
|
|
5799
|
+
mimeTypes: string[];
|
|
5800
|
+
/** How many clients used each, in the same order as {@link mimeTypes}. */
|
|
5801
|
+
clientsPerMimeType: number[];
|
|
5802
|
+
/** Clients considered — those that reported at least one codec of this kind. */
|
|
5803
|
+
clients: number;
|
|
5804
|
+
/** The codec the check was told to expect, when it was given one. */
|
|
5805
|
+
expected?: string;
|
|
5806
|
+
};
|
|
5807
|
+
type CodecConsistencyReportPayload = ({
|
|
5808
|
+
/** One codec per media kind, and it is the expected one if an expectation was given. */
|
|
5809
|
+
verdict: 'codec-consistent';
|
|
5810
|
+
evidence: CodecEvidence[];
|
|
5811
|
+
} | {
|
|
5812
|
+
/** Participants of one call are split across different codecs. */
|
|
5813
|
+
verdict: 'codec-split';
|
|
5814
|
+
evidence: CodecEvidence[];
|
|
5815
|
+
} | {
|
|
5816
|
+
/** Consistent, but not what the deployment believes it negotiated. */
|
|
5817
|
+
verdict: 'unexpected-codec';
|
|
5818
|
+
evidence: CodecEvidence[];
|
|
5819
|
+
} | {
|
|
5820
|
+
/** Never saw a call with enough participants reporting codecs. **Not a pass.** */
|
|
5821
|
+
verdict: 'inconclusive';
|
|
5822
|
+
reason: string;
|
|
5823
|
+
}) & {
|
|
5824
|
+
startedAt: number;
|
|
5825
|
+
checks: number;
|
|
5826
|
+
};
|
|
5827
|
+
type CodecConsistencyValidatorConfig = {
|
|
5828
|
+
/**
|
|
5829
|
+
* The mime type you believe you are delivering, per kind — e.g.
|
|
5830
|
+
* `{ video: 'video/VP8', audio: 'audio/opus' }`.
|
|
5831
|
+
*
|
|
5832
|
+
* Optional. Without it the check still reports a *split* (participants disagreeing with each
|
|
5833
|
+
* other), which needs no expectation to be a fact. With it, the check can additionally catch the
|
|
5834
|
+
* case where everyone agrees on the wrong thing — a silent fallback that nothing else notices.
|
|
5835
|
+
*/
|
|
5836
|
+
expected?: Partial<Record<'audio' | 'video', string>>;
|
|
5837
|
+
/**
|
|
5838
|
+
* Which kinds to inspect. Default `[ 'audio', 'video' ]`.
|
|
5839
|
+
*
|
|
5840
|
+
* Narrow it when only one matters: audio codec splits are the ones that usually cost transcoding,
|
|
5841
|
+
* while video splits are more often a deliberate per-client decision.
|
|
5842
|
+
*/
|
|
5843
|
+
kinds: ('audio' | 'video')[];
|
|
5844
|
+
/**
|
|
5845
|
+
* Participants a call needs before disagreement is meaningful. Default `3`.
|
|
5846
|
+
*
|
|
5847
|
+
* In a 1:1 call "the participants disagree" is two clients differing, which can be a legitimate
|
|
5848
|
+
* negotiation outcome rather than a fault. Sensible range `3`–`5`.
|
|
5849
|
+
*/
|
|
5850
|
+
minClients: number;
|
|
5851
|
+
/**
|
|
5852
|
+
* Calls to inspect before concluding. Default `3`.
|
|
5853
|
+
*
|
|
5854
|
+
* A structural property of your negotiation, so a handful of calls is plenty — but one call could be
|
|
5855
|
+
* an unusual mix of participants. Typical `3`–`10`. Higher delays the verdict without adding much,
|
|
5856
|
+
* since the answer does not vary call to call.
|
|
5857
|
+
*/
|
|
5858
|
+
minChecks: number;
|
|
5859
|
+
};
|
|
5860
|
+
/**
|
|
5861
|
+
* Answers: **is every participant of a call actually using the same codec — and is it the one you
|
|
5862
|
+
* think you negotiated?**
|
|
5863
|
+
*
|
|
5864
|
+
* ### Why the server has to answer this
|
|
5865
|
+
*
|
|
5866
|
+
* A client knows only its own codec. It cannot tell whether it is the odd one out, and an SFU that
|
|
5867
|
+
* forwards without transcoding cannot serve a call where participants disagree — so a split is a
|
|
5868
|
+
* real fault with a very confusing symptom: some pairs of participants see each other and some do
|
|
5869
|
+
* not, with no error anywhere. Only something holding every participant of a call at once can see
|
|
5870
|
+
* the split at all.
|
|
5871
|
+
*
|
|
5872
|
+
* The second half is the quieter failure. A deployment configured for VP9 or AV1 will fall back to
|
|
5873
|
+
* VP8 whenever one endpoint cannot negotiate the preferred codec, and nothing reports that — the
|
|
5874
|
+
* call works, the bitrate is higher than it should be, and the team believes it shipped AV1 months
|
|
5875
|
+
* ago. Give the check an `expected` mime type and it will say so.
|
|
5876
|
+
*
|
|
5877
|
+
* ### Why a validator and not a detector
|
|
5878
|
+
*
|
|
5879
|
+
* The answer is a property of the deployment — SDP munging, codec preferences, the SFU build — not
|
|
5880
|
+
* of this moment. A deployment that negotiates VP8 at 09:00 negotiates VP8 at 17:00. Re-deriving it
|
|
5881
|
+
* every tick would walk every codec of every peer connection of every call, forever, to re-learn a
|
|
5882
|
+
* constant. So it decides once and stops.
|
|
5883
|
+
*
|
|
5884
|
+
* Start it again after a deploy, or after changing codec preferences:
|
|
5885
|
+
*
|
|
5886
|
+
* ```ts
|
|
5887
|
+
* observer.addValidator('codec-consistency', {
|
|
5888
|
+
* expected: { video: 'video/VP9', audio: 'audio/opus' },
|
|
5889
|
+
* });
|
|
5890
|
+
*
|
|
5891
|
+
* observer.on('validation-ready', ({ validator, report }) => {
|
|
5892
|
+
* if (validator !== 'codec-consistency' || !report.ready) return;
|
|
5893
|
+
* // 'codec-consistent' | 'codec-split' | 'unexpected-codec' | 'inconclusive'
|
|
5894
|
+
* });
|
|
5895
|
+
* ```
|
|
5896
|
+
*
|
|
5897
|
+
* ### `inconclusive` is not a pass
|
|
5898
|
+
*
|
|
5899
|
+
* Only calls with at least `minClients` participants *reporting codecs of that kind* count as a
|
|
5900
|
+
* check. An audio-only deployment will never say anything about video, and concluding "video codecs
|
|
5901
|
+
* are consistent" from calls that carried no video would verify nothing.
|
|
5902
|
+
*
|
|
5903
|
+
* ### Comparison is by mime type only
|
|
5904
|
+
*
|
|
5905
|
+
* `sdpFmtpLine` carries profile and level — `profile-level-id` for H.264, `profile-id` for VP9 — and
|
|
5906
|
+
* two clients on the same mime type with different profiles are not truly interchangeable. That is
|
|
5907
|
+
* deliberately out of scope: fmtp differences are common, usually benign, and would make this check
|
|
5908
|
+
* noisy enough to ignore. It answers the coarse question, which is the one that is actually wrong in
|
|
5909
|
+
* practice.
|
|
5910
|
+
*/
|
|
5911
|
+
declare class CodecConsistencyValidator implements Validator<CodecConsistencyReportPayload> {
|
|
5912
|
+
private readonly _observer;
|
|
5913
|
+
readonly onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void;
|
|
5914
|
+
static readonly NAME: "codec-consistency";
|
|
5915
|
+
readonly name: "codec-consistency";
|
|
5916
|
+
readonly startedAt: number;
|
|
5917
|
+
report: ValidationReport<CodecConsistencyReportPayload>;
|
|
5918
|
+
private readonly _config;
|
|
5919
|
+
private readonly _inspectedCallIds;
|
|
5920
|
+
private _evidence;
|
|
5921
|
+
private _checks;
|
|
5922
|
+
private _done;
|
|
5923
|
+
constructor(_observer: Observer, onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void, config?: Partial<CodecConsistencyValidatorConfig>);
|
|
5924
|
+
get checks(): number;
|
|
5925
|
+
cancel(reason?: string): void;
|
|
5926
|
+
update(): void;
|
|
5927
|
+
/** One evidence entry per configured kind that the call actually carried. */
|
|
5928
|
+
private _tallyOf;
|
|
5929
|
+
private _finish;
|
|
5930
|
+
}
|
|
5931
|
+
|
|
5932
|
+
/**
|
|
5933
|
+
* The validators `observer.addValidator(name, config)` knows how to build, and the config each takes.
|
|
5934
|
+
*
|
|
5935
|
+
* Adding one means: write the class with a `static readonly NAME`, add its entry here, and add a
|
|
5936
|
+
* `case` to `addValidator`. The map is what gives the call site its types —
|
|
5937
|
+
* `addValidator('simulcast-receivers', { … })` type-checks the config against the right validator,
|
|
5938
|
+
* and an unknown name won't compile.
|
|
5939
|
+
*/
|
|
5940
|
+
type AvailableValidatorConfigs = {
|
|
5941
|
+
[SimulcastReceiverValidator.NAME]: SimulcastReceiverValidatorConfig;
|
|
5942
|
+
[RemoteTrackResolverValidator.NAME]: RemoteTrackResolverValidatorConfig;
|
|
5943
|
+
[CodecConsistencyValidator.NAME]: CodecConsistencyValidatorConfig;
|
|
5944
|
+
};
|
|
5945
|
+
/** A validator name that can be started. */
|
|
5946
|
+
type ValidatorName = keyof AvailableValidatorConfigs;
|
|
5947
|
+
|
|
5948
|
+
/**
|
|
5949
|
+
* Keeps every configured `CallSummary` up to date, from **observer-level** bus subscriptions.
|
|
5950
|
+
*
|
|
5951
|
+
* ### Why one collector and not one per call
|
|
5952
|
+
*
|
|
5953
|
+
* The obvious implementation subscribes each call's summary to the events it needs. But the bus is
|
|
5954
|
+
* observer-wide: a listener attached for call A is invoked for every event of every call, so that
|
|
5955
|
+
* design costs `calls × events` listeners *and* `calls` invocations per event — quadratic in the
|
|
5956
|
+
* thing most likely to be large. At 500 concurrent calls and eight subscribed events that is 4 000
|
|
5957
|
+
* listeners doing 500 no-op calls each, per event.
|
|
5958
|
+
*
|
|
5959
|
+
* So the collector attaches **one listener per event type, once**, and routes each event to the
|
|
5960
|
+
* summary of the call it names. Cost is O(subscribed event types), independent of how many calls are
|
|
5961
|
+
* in flight, and an event for a call with no summary costs one `undefined` check.
|
|
5962
|
+
*
|
|
5963
|
+
* ### Only call-scoped events
|
|
5964
|
+
*
|
|
5965
|
+
* Routing needs `observedCall` on the payload, which is exactly what `CallScopedEventName` selects.
|
|
5966
|
+
* Observer-scoped events have no single call to attribute to; see that type for why fanning them out
|
|
5967
|
+
* to every open summary would be worse than refusing.
|
|
5968
|
+
*/
|
|
5969
|
+
declare class CallSummaryCollector {
|
|
5970
|
+
private readonly _observer;
|
|
5971
|
+
private readonly _config;
|
|
5972
|
+
private readonly _scratch;
|
|
5973
|
+
private readonly _listeners;
|
|
5974
|
+
private _closed;
|
|
5975
|
+
constructor(_observer: Observer, _config: CallSummaryConfig);
|
|
5976
|
+
/**
|
|
5977
|
+
* Build a summary for `callId` and start tracking it.
|
|
5978
|
+
*
|
|
5979
|
+
* Creating it here, rather than letting the call create one and hand it over, keeps the resolved
|
|
5980
|
+
* configuration inside the single object that owns it — and makes it impossible to end up with a
|
|
5981
|
+
* summary whose sections nobody subscribed to fill.
|
|
5982
|
+
*/
|
|
5983
|
+
createSummary(callId: string): CallSummary;
|
|
5984
|
+
/**
|
|
5985
|
+
* Finalise `call`'s summary: fold in what only makes sense once, and stamp the closing times.
|
|
5986
|
+
*
|
|
5987
|
+
* Percentiles are computed here rather than on every update — a median recomputed per tick over a
|
|
5988
|
+
* growing array is quadratic work to produce a number nobody reads until the end.
|
|
5989
|
+
*/
|
|
5990
|
+
finalise(call: ObservedCall): void;
|
|
5991
|
+
/** Drop every bus subscription. Called when the observer closes. */
|
|
5992
|
+
close(): void;
|
|
5993
|
+
/**
|
|
5994
|
+
* Subscribe `listener` to `event`, routed to the summary of the call the event names.
|
|
5995
|
+
*
|
|
5996
|
+
* The `observedCall` is read off the payload rather than closed over, which is what lets one
|
|
5997
|
+
* subscription serve every call.
|
|
5998
|
+
*/
|
|
5999
|
+
private _on;
|
|
6000
|
+
private _subscribeBuiltIns;
|
|
6001
|
+
private _subscribeEnrichers;
|
|
6002
|
+
}
|
|
6003
|
+
|
|
6004
|
+
type SampleRejectedReason = 'observer-closed' | 'missing-callId' | 'missing-clientId';
|
|
6005
|
+
|
|
6006
|
+
/**
|
|
6007
|
+
* Optional, free-form context supplied to `accept()`. A single context object is
|
|
6008
|
+
* threaded down the accept chain (Observer -> Client -> PeerConnection) and is
|
|
6009
|
+
* merged into the `appData` of entities created during the accept pass.
|
|
6010
|
+
*/
|
|
6011
|
+
type AcceptContext = Record<string, unknown>;
|
|
6012
|
+
/** The payload threaded through `accept()` middlewares: the sample and its optional context. */
|
|
6013
|
+
type AcceptMiddlewarePayload = {
|
|
6014
|
+
sample: ClientSample;
|
|
6015
|
+
context?: AcceptContext;
|
|
6016
|
+
};
|
|
6017
|
+
/**
|
|
6018
|
+
* A global middleware run on every sample passed to `observer.accept()`, in registration order,
|
|
6019
|
+
* **before** the sample is dispatched to any call/client. It may inspect or mutate the sample
|
|
6020
|
+
* (e.g. set/normalize `callId`/`clientId`, enrich, redact) or the context, then call
|
|
6021
|
+
* `next(payload)` to continue the chain. Not calling `next` **drops** the sample.
|
|
6022
|
+
*/
|
|
6023
|
+
type AcceptMiddleware = Middleware<AcceptMiddlewarePayload>;
|
|
6024
|
+
/** Produces the initial `appData` for a call created without an explicit `appData`. */
|
|
6025
|
+
type CallAppDataFactory = (params: {
|
|
6026
|
+
callId: string;
|
|
6027
|
+
observer: Observer;
|
|
6028
|
+
acceptCtx?: AcceptContext;
|
|
6029
|
+
}) => Record<string, unknown>;
|
|
6030
|
+
/** Produces the initial `appData` for a client created without an explicit `appData`. */
|
|
6031
|
+
type ClientAppDataFactory = (params: {
|
|
6032
|
+
clientId: string;
|
|
6033
|
+
observedCall: ObservedCall;
|
|
6034
|
+
acceptCtx?: AcceptContext;
|
|
6035
|
+
}) => Record<string, unknown>;
|
|
6036
|
+
type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
6037
|
+
/**
|
|
6038
|
+
* Application-owned data for the observer itself. Never read or modified by the library.
|
|
6039
|
+
*
|
|
6040
|
+
* For per-call / per-client data prefer `createCallAppData` / `createClientAppData`, which run at
|
|
6041
|
+
* creation time and can see the `accept()` context.
|
|
6042
|
+
*/
|
|
6043
|
+
appData?: AppData;
|
|
6044
|
+
/**
|
|
6045
|
+
* Close a client that has not produced a sample for this long (ms). Default `60_000`.
|
|
6046
|
+
*
|
|
6047
|
+
* This is the **liveness timeout for a participant**, so set it from your client's sampling
|
|
6048
|
+
* period, not from taste: a client sampling every 5 s needs several missed samples to look gone.
|
|
6049
|
+
* Sensible range `3x`–`10x` the sampling period; `60_000` suits the usual 2–10 s collectors.
|
|
6050
|
+
*
|
|
6051
|
+
* Too low and a client that merely paused (tab backgrounded, brief network drop) is closed and
|
|
6052
|
+
* then re-created as a *new* client, which restarts its detectors and splits one participant into
|
|
6053
|
+
* two in any summary. Too high and left participants linger, inflating `peak`, the denominators of
|
|
6054
|
+
* every ratio-based detector, and memory. `undefined` disables the timeout — then nothing closes
|
|
6055
|
+
* an abandoned client but your own `client.close()`.
|
|
6056
|
+
*/
|
|
6057
|
+
closeClientIfIdleForMs?: number;
|
|
6058
|
+
/**
|
|
6059
|
+
* Close a call once it has had zero clients for this long (ms). Default `60_000`.
|
|
6060
|
+
*
|
|
6061
|
+
* The grace period exists so a brief empty moment — everyone reconnecting after a network blip,
|
|
6062
|
+
* the last participant refreshing — does not end the call and start a new one under the same
|
|
6063
|
+
* `callId`. Sensible range `10_000`–`300_000`.
|
|
6064
|
+
*
|
|
6065
|
+
* Too low splits one meeting into several calls, and each split emits its own `call-summary`. Too
|
|
6066
|
+
* high keeps dead calls in `observedCalls`, holding their detectors and summaries in memory.
|
|
6067
|
+
* `undefined` disables it: the call then lives until you call `call.close()`.
|
|
6068
|
+
*/
|
|
6069
|
+
closeCallIfEmptyForMs?: number;
|
|
6070
|
+
/**
|
|
6071
|
+
* When `true` (the default), every call update triggers an observer-wide `update()` pass.
|
|
6072
|
+
*
|
|
6073
|
+
* There is deliberately no timer and no separate policy object: a call is updated when any of its
|
|
6074
|
+
* clients is, and the observer is updated when any of its calls is — so the observer is updated
|
|
6075
|
+
* exactly when any client anywhere is. Set to `false` only if you drive `observer.update()`
|
|
6076
|
+
* yourself, and note that observer-scoped detectors and validators run *nowhere else*.
|
|
6077
|
+
*/
|
|
6078
|
+
autoUpdateOnCallUpdate?: boolean;
|
|
6079
|
+
/**
|
|
6080
|
+
* Accumulate a {@link CallSummary} on every call this observer creates.
|
|
6081
|
+
*
|
|
6082
|
+
* **Absent or `null` means no summaries at all** — no accumulation, and not one bus subscription.
|
|
6083
|
+
* Pass an object (`{}` is valid) to switch it on; anything you leave out takes its default from
|
|
6084
|
+
* `defaultCallSummaryConfig`, including `include: []`, which collects *no* built-in section. A
|
|
6085
|
+
* summary that only runs `enrich` is a perfectly good summary.
|
|
6086
|
+
*
|
|
6087
|
+
* ```ts
|
|
6088
|
+
* const observer = new Observer({
|
|
6089
|
+
* callSummary: {
|
|
6090
|
+
* include: [ 'clients', 'issues' ],
|
|
6091
|
+
* enrich: {
|
|
6092
|
+
* 'client-joined': (summary, { observedClient }) => {
|
|
6093
|
+
* ((summary.attachments.regions ??= []) as string[]).push(String(observedClient.appData.region));
|
|
6094
|
+
* },
|
|
6095
|
+
* },
|
|
6096
|
+
* },
|
|
6097
|
+
* });
|
|
6098
|
+
*
|
|
6099
|
+
* observer.on('call-summary', ({ summary }) => archive(summary));
|
|
6100
|
+
* ```
|
|
6101
|
+
*
|
|
6102
|
+
* This is construction-time and fixed for the observer's life, unlike detectors, which are added
|
|
6103
|
+
* per call as an application decides what to watch. A summary is a record of what happened, and a
|
|
6104
|
+
* record you can turn on halfway through is a record with a hole in it — calls that started
|
|
6105
|
+
* earlier would carry different sections from calls that started later, with nothing on either to
|
|
6106
|
+
* say which. One shape for every call, or none.
|
|
6107
|
+
*/
|
|
6108
|
+
callSummary?: Partial<CallSummaryConfig> | null;
|
|
6109
|
+
/**
|
|
6110
|
+
* Thresholds that mark a **received** track as degraded. Omitted (the default) means no track is
|
|
6111
|
+
* ever marked degraded — `inboundTrack.degraded` stays `false` and `degradationReasons` stays
|
|
6112
|
+
* empty, so this is opt-in and absence is *not* a clean bill of health.
|
|
6113
|
+
*
|
|
6114
|
+
* Each field is an **exclusive upper bound**: the reason is added when the measured value is
|
|
6115
|
+
* strictly greater. All are evaluated on every update and any number can fire at once;
|
|
6116
|
+
* `degradationReasons` lists the ones that did.
|
|
6117
|
+
*
|
|
6118
|
+
* These feed `CallHealthAggregator` and `outboundTrack.degradedRatio` (how many of a publisher's
|
|
6119
|
+
* receivers are unhappy), which is what `TrackDeliveryMismatchDetector` and
|
|
6120
|
+
* `PublisherFaultCorroborationDetector` read. They are **not** a substitute for client issues:
|
|
6121
|
+
* client-monitor already decides "this endpoint is in trouble" with hysteresis and multi-signal
|
|
6122
|
+
* confirmation. Treat these as a coarse per-track flag for cross-participant comparison, and keep
|
|
6123
|
+
* them loose enough that a single bad tick does not trip them.
|
|
6124
|
+
*/
|
|
6125
|
+
inboundTrackDegradationThresholds?: {
|
|
6126
|
+
/**
|
|
6127
|
+
* Freezes counted **in one sampling period**, not since the start of the call. `1` means "any
|
|
6128
|
+
* freeze at all this tick", which is strict; `2`–`3` tolerates the odd frame hiccup.
|
|
6129
|
+
* Reason: `'freezes'`.
|
|
6130
|
+
*/
|
|
6131
|
+
deltaFreezeCount: number;
|
|
6132
|
+
/**
|
|
6133
|
+
* Fraction of frames dropped, `0`–`1`. Typical `0.05`–`0.2`; below ~`0.02` you are inside
|
|
6134
|
+
* normal jitter for most decoders. Reason: `'frames-dropped'`.
|
|
6135
|
+
*/
|
|
6136
|
+
framesDroppedRatio: number;
|
|
6137
|
+
/**
|
|
6138
|
+
* Jitter buffer delay in **ms**. Typical `150`–`500`: audio stays intelligible well past
|
|
6139
|
+
* `200`, while conversation turn-taking suffers beyond ~`400`. Reason:
|
|
6140
|
+
* `'jitter-buffer-delay'`.
|
|
6141
|
+
*/
|
|
6142
|
+
jitterBufferDelayInMs: number;
|
|
6143
|
+
/**
|
|
6144
|
+
* Concealed (synthesised) audio samples as a fraction, `0`–`1`. Typical `0.05`–`0.15`; above
|
|
6145
|
+
* ~`0.1` is usually audible as robotic or clipped speech. Reason: `'concealment'`.
|
|
6146
|
+
*/
|
|
6147
|
+
concealmentRatio: number;
|
|
6148
|
+
/**
|
|
6149
|
+
* Round-trip time in **ms**, as reported for this inbound stream. Typical `250`–`500`.
|
|
6150
|
+
* Remember this is absolute, not relative to the participant's own baseline — a genuinely
|
|
6151
|
+
* distant participant will sit permanently above any fixed bound, which is why "RTT got
|
|
6152
|
+
* worse" belongs to client-monitor and not here. Reason: `'rtt'`.
|
|
6153
|
+
*/
|
|
6154
|
+
rttInMs: number;
|
|
6155
|
+
};
|
|
6156
|
+
/**
|
|
6157
|
+
* Thresholds that mark a **published** track as degraded, from what the receivers report back.
|
|
6158
|
+
* Omitted (the default) means neither threshold-based reason can fire.
|
|
6159
|
+
*
|
|
6160
|
+
* Note two reasons are added regardless of this setting, because they need no threshold:
|
|
6161
|
+
* `quality-limited-<reason>` when the encoder reports a `qualityLimitationReason`, and
|
|
6162
|
+
* `no-packets-sent` when an unmuted track sent nothing in a sampling period.
|
|
6163
|
+
*
|
|
6164
|
+
* Also an **exclusive upper bound** per field, evaluated on every update.
|
|
6165
|
+
*/
|
|
6166
|
+
outboundTrackDegradationThresholds?: {
|
|
6167
|
+
/**
|
|
6168
|
+
* Fraction of packets lost as reported by the remote end, `0`–`1`. Typical `0.02`–`0.1`;
|
|
6169
|
+
* anything under ~`0.01` is normal on the open internet and will fire constantly. Reason:
|
|
6170
|
+
* `'remote-fraction-lost'`.
|
|
6171
|
+
*/
|
|
6172
|
+
fractionLost: number;
|
|
6173
|
+
/**
|
|
6174
|
+
* Round-trip time in **ms** as reported by the remote end. Typical `250`–`500`, and absolute
|
|
6175
|
+
* rather than baseline-relative — see the note on the inbound `rttInMs`. Reason:
|
|
6176
|
+
* `'remote-rtt'`.
|
|
6177
|
+
*/
|
|
6178
|
+
rttInMs: number;
|
|
6179
|
+
};
|
|
6180
|
+
/**
|
|
6181
|
+
* Optional factory invoked when a call is created without an explicit `appData`
|
|
6182
|
+
* (e.g. lazily by `accept()`), so apps can enrich appData without pre-creating the
|
|
6183
|
+
* entity. `appData` is application-owned; it is never modified by the `accept()` context.
|
|
6184
|
+
*/
|
|
6185
|
+
createCallAppData?: CallAppDataFactory;
|
|
6186
|
+
/** Same as `createCallAppData`, for clients. Receives the (already-created) parent call. */
|
|
6187
|
+
createClientAppData?: ClientAppDataFactory;
|
|
6188
|
+
/**
|
|
6189
|
+
* Optional factory invoked when a client is created, producing a per-client sink that
|
|
6190
|
+
* receives every sample the client accepts (or `undefined` for no sink). The destination
|
|
6191
|
+
* can be derived from `callId` / `clientId`.
|
|
6192
|
+
*/
|
|
6193
|
+
createClientSink?: ClientSampleSinkFactory;
|
|
6194
|
+
/**
|
|
6195
|
+
* Optional factory invoked when a call is created, producing the call's `RemoteTrackResolver`
|
|
6196
|
+
* (or `undefined` for none). Use the built-ins
|
|
6197
|
+
* (`createDefaultMediasoupRemoteTrackResolverFactory()` / `createP2pRemoteTrackResolverFactory()`)
|
|
6198
|
+
* or build a `RemoteTrackResolver` with custom publisher/subscriber id resolvers.
|
|
6199
|
+
*/
|
|
6200
|
+
createRemoteTrackResolver?: RemoteTrackResolverFactory;
|
|
6201
|
+
};
|
|
6202
|
+
declare interface Observer {
|
|
6203
|
+
on<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
|
|
6204
|
+
off<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
|
|
6205
|
+
once<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
|
|
6206
|
+
emit<U extends keyof ObserverEvents>(event: U, ...args: ObserverEvents[U]): boolean;
|
|
6207
|
+
}
|
|
6208
|
+
declare class Observer<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
|
|
2817
6209
|
readonly observedTURN: ObservedTURN;
|
|
2818
6210
|
readonly observedCalls: Map<string, ObservedCall<Record<string, unknown>>>;
|
|
2819
6211
|
readonly observedMediasoupRouters: Map<string, ObservedMediasoupRouter<Record<string, unknown>>>;
|
|
2820
|
-
updater?: Updater;
|
|
2821
6212
|
/** Ancestry base shared by all Observer-bus events originating at the observer. */
|
|
2822
6213
|
readonly eventScope: ObserverEventBase;
|
|
6214
|
+
readonly config: ObserverConfig<AppData>;
|
|
2823
6215
|
closed: boolean;
|
|
2824
6216
|
totalAddedCall: number;
|
|
2825
6217
|
totalRemovedCall: number;
|
|
@@ -2831,19 +6223,151 @@ declare class Observer<AppData extends Record<string, unknown> = Record<string,
|
|
|
2831
6223
|
numberOfPeerConnections: number;
|
|
2832
6224
|
/** Global, pre-dispatch middleware chain run on every accepted sample. */
|
|
2833
6225
|
readonly acceptMiddlewares: MiddlewareProcessor<AcceptMiddlewarePayload>;
|
|
2834
|
-
|
|
6226
|
+
/**
|
|
6227
|
+
* Fleet-wide index of every open client issue, maintained incrementally as issues open and close.
|
|
6228
|
+
* Each call's index propagates into this one, so cross-call queries cost O(matching issues) rather
|
|
6229
|
+
* than a walk over every call and client. This is what observer-scoped detectors read.
|
|
6230
|
+
*/
|
|
6231
|
+
readonly activeIssuesRegistry: ActiveIssuesRegistry;
|
|
6232
|
+
/**
|
|
6233
|
+
* Validators currently running. Each removes itself when it finishes, so this is normally empty —
|
|
6234
|
+
* a validator is a one-shot check, not a permanent fixture. Start one with {@link addValidator}.
|
|
6235
|
+
*/
|
|
6236
|
+
readonly validators: Set<RunningValidator>;
|
|
6237
|
+
/**
|
|
6238
|
+
* Observer-scoped detector registry, run on every `observer.update()` — the place for findings
|
|
6239
|
+
* that span **calls**, e.g. "many calls on the same SFU degraded at once". Detectors raise
|
|
6240
|
+
* findings with `observer.addIssue(...)`, surfaced on the bus as `observer-issue`.
|
|
6241
|
+
* (For findings within a single call use `observedCall.detectors`.)
|
|
6242
|
+
*
|
|
6243
|
+
* Starts **empty**. Populate it with {@link addObserverDetector}, or `detectors.add(...)` for an
|
|
6244
|
+
* instance you built yourself.
|
|
6245
|
+
*/
|
|
6246
|
+
readonly detectors: Detectors;
|
|
6247
|
+
/**
|
|
6248
|
+
* The call-scoped detectors to build on **every** call this observer creates, in registration
|
|
6249
|
+
* order. Written by {@link addCallDetector}; read by `createObservedCall`.
|
|
6250
|
+
*
|
|
6251
|
+
* Nothing is created implicitly. There is no detector configuration in `ObserverConfig` and no
|
|
6252
|
+
* default set, because a detector that nobody asked for is a detector nobody will act on: it costs
|
|
6253
|
+
* time on every tick and raises findings into a handler that was not written to expect them. An
|
|
6254
|
+
* application says what it wants to watch, or it watches nothing.
|
|
6255
|
+
*
|
|
6256
|
+
* ```ts
|
|
6257
|
+
* observer.addCallDetector('call-concurrent-issue-detector', {
|
|
6258
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
6259
|
+
* });
|
|
6260
|
+
* ```
|
|
6261
|
+
*/
|
|
6262
|
+
readonly callDetectorConfigs: Map<keyof AvailableCallScopeDetectorsConfigs, Partial<UnconsumedTrackDetectorConfig | TrackDeliveryMismatchDetectorConfig | CallConcurrentIssueDetectorConfig | IssueFanOutDetectorConfig | PublisherFaultCorroborationDetectorConfig>>;
|
|
6263
|
+
/**
|
|
6264
|
+
* Owns every call's summary: the resolved `config.callSummary`, the bus subscriptions that keep
|
|
6265
|
+
* the summaries current (one per event type, not one per call), and the summaries themselves.
|
|
6266
|
+
*
|
|
6267
|
+
* `undefined` when `config.callSummary` was absent or `null` — so its presence *is* the answer to
|
|
6268
|
+
* "are summaries on", and nothing is subscribed to anything.
|
|
6269
|
+
*/
|
|
6270
|
+
readonly callSummaryCollector?: CallSummaryCollector;
|
|
6271
|
+
constructor(config?: Partial<ObserverConfig<AppData>>);
|
|
2835
6272
|
get numberOfCalls(): number;
|
|
2836
6273
|
get appData(): AppData | undefined;
|
|
6274
|
+
/**
|
|
6275
|
+
* Build a cross-call detector onto `observer.detectors`. Chainable.
|
|
6276
|
+
*
|
|
6277
|
+
* To get a handle on what was built — to inspect it, or to remove that exact instance later — read
|
|
6278
|
+
* it back off the registry: `observer.detectors.getAll(name)`, or `observer.detectors.instances`.
|
|
6279
|
+
*/
|
|
6280
|
+
addObserverDetector<K extends keyof AvailableObserverScopeDetectorsConfigs>(name: K, config?: Partial<AvailableObserverScopeDetectorsConfigs[K]>): this;
|
|
6281
|
+
/**
|
|
6282
|
+
* Enable a call-scoped detector for calls created **from now on**.
|
|
6283
|
+
*
|
|
6284
|
+
* This edits the config, not the live calls: calls already open keep the detector set they were
|
|
6285
|
+
* built with. To add one to an existing call, use `observedCall.addDetector(...)` directly.
|
|
6286
|
+
*/
|
|
6287
|
+
addCallDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
|
|
6288
|
+
/**
|
|
6289
|
+
* Remove an observer-scoped detector by name, returning how many were removed.
|
|
6290
|
+
*
|
|
6291
|
+
* **Every** instance registered under the name goes, since a name can legitimately be registered
|
|
6292
|
+
* more than once (`ClientPopulationIssueDetector` is meant to be added once per `groupBy` axis).
|
|
6293
|
+
* When you want one of them specifically, go through the registry, which deals in instances:
|
|
6294
|
+
*
|
|
6295
|
+
* ```ts
|
|
6296
|
+
* const [ byBrowser, byOs ] = observer.detectors.getAll('client-population-issue-detector');
|
|
6297
|
+
*
|
|
6298
|
+
* observer.detectors.remove(byOs); // keeps the browser axis running
|
|
6299
|
+
* ```
|
|
6300
|
+
*
|
|
6301
|
+
* Either route `close()`s the detector, so it unsubscribes from the issue registry and drops any
|
|
6302
|
+
* timers or bus listeners it held.
|
|
6303
|
+
*/
|
|
6304
|
+
removeObserverDetector(name: keyof AvailableObserverScopeDetectorsConfigs): number;
|
|
6305
|
+
/**
|
|
6306
|
+
* Stop building `name` on calls created from now on.
|
|
6307
|
+
*
|
|
6308
|
+
* By default this also removes it from the calls **already open**, so that "remove this detector"
|
|
6309
|
+
* means the same thing whether you say it before or after a call started — the alternative leaves
|
|
6310
|
+
* a fleet where the detector is live on some calls and not others, decided by join time. Pass
|
|
6311
|
+
* `{ includeOpenCalls: false }` to change only what future calls are built with.
|
|
6312
|
+
*
|
|
6313
|
+
* Returns the number of live detector instances removed (`0` when only the config changed).
|
|
6314
|
+
*/
|
|
6315
|
+
removeCallDetector(name: keyof AvailableCallScopeDetectorsConfigs, { includeOpenCalls }?: {
|
|
6316
|
+
includeOpenCalls?: boolean;
|
|
6317
|
+
}): number;
|
|
6318
|
+
/**
|
|
6319
|
+
* Start a structural check. It runs on each `observer.update()` until it can decide, reports once
|
|
6320
|
+
* on `validation-ready`, and removes itself.
|
|
6321
|
+
*
|
|
6322
|
+
* ```ts
|
|
6323
|
+
* observer.validate('simulcast-receiver-validator', { minChecks: 5 });
|
|
6324
|
+
* ```
|
|
6325
|
+
*
|
|
6326
|
+
* Config keys are optional and merged over that validator's defaults. Call it again — after a
|
|
6327
|
+
* deploy, say — to check again; there is no revalidation timer, because a deploy rather than
|
|
6328
|
+
* elapsed time is what makes a structural verdict stale.
|
|
6329
|
+
*/
|
|
6330
|
+
addValidator<K extends keyof AvailableValidatorConfigs>(name: K, config?: Partial<AvailableValidatorConfigs[K]>): this;
|
|
6331
|
+
/**
|
|
6332
|
+
* Stop a running validation, by name or by instance. Returns how many were cancelled.
|
|
6333
|
+
*
|
|
6334
|
+
* Cancelling is **not** silent discarding. The validator finishes with `inconclusive` and the given
|
|
6335
|
+
* `reason`, emits `validation-ready` like any other completion, and removes itself. That matters
|
|
6336
|
+
* because anything waiting on the verdict — a deploy gate, a dashboard, a promise — would otherwise
|
|
6337
|
+
* wait forever, and because "we stopped asking" is a materially different outcome from "we asked
|
|
6338
|
+
* and learned nothing", which is exactly what `inconclusive` with a reason records.
|
|
6339
|
+
*
|
|
6340
|
+
* ```ts
|
|
6341
|
+
* observer.cancelValidator('simulcast-receivers', 'sfu redeployed');
|
|
6342
|
+
*
|
|
6343
|
+
* // or one specific instance — `observer.validators` holds what is running
|
|
6344
|
+
* for (const validator of observer.validators) observer.cancelValidator(validator, 'shutting down');
|
|
6345
|
+
* ```
|
|
6346
|
+
*
|
|
6347
|
+
* Pass a real reason. The default tells the reader nothing they could not already infer.
|
|
6348
|
+
*/
|
|
6349
|
+
cancelValidator(target: keyof AvailableValidatorConfigs | RunningValidator, reason?: string): number;
|
|
2837
6350
|
getObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(callId: string): ObservedCall<T> | undefined;
|
|
2838
|
-
|
|
2839
|
-
|
|
6351
|
+
/**
|
|
6352
|
+
* @param acceptCtx the `accept()` context, when this call is being created to receive a sample.
|
|
6353
|
+
* Passed on to `ObserverConfig.createCallAppData`, so the factory can read whatever the caller (or
|
|
6354
|
+
* an accept middleware) put there — a tenant, a region, a trace id.
|
|
6355
|
+
*/
|
|
6356
|
+
createObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>, acceptCtx?: AcceptContext): ObservedCall<T> | undefined;
|
|
6357
|
+
getOrCreateObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>, acceptCtx?: AcceptContext): ObservedCall<T> | undefined;
|
|
2840
6358
|
createObservedMediasoupRouter<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedMediasoupRouterSettings<T> & {
|
|
2841
|
-
|
|
2842
|
-
bindCallByWebRtcTransportId?: boolean;
|
|
6359
|
+
matchPeerConnectionByWebRtcTransportId?: boolean;
|
|
2843
6360
|
}): ObservedMediasoupRouter<Record<string, unknown>> | undefined;
|
|
2844
6361
|
close(): void;
|
|
2845
6362
|
accept(sample: ClientSample, context?: AcceptContext): void;
|
|
2846
6363
|
update(): void;
|
|
6364
|
+
/**
|
|
6365
|
+
* Raise an observer-scoped (cross-call / SFU-wide) finding. Emitted on the bus as
|
|
6366
|
+
* `observer-issue`. Intended for `observer.detectors`, but the application may call it too.
|
|
6367
|
+
*
|
|
6368
|
+
* `payload` takes an **object**; see `ObserverIssue`.
|
|
6369
|
+
*/
|
|
6370
|
+
addIssue(issue: Omit<ObserverIssue, 'scope'>): void;
|
|
2847
6371
|
/** Emit an Observer-bus event. */
|
|
2848
6372
|
private _notify;
|
|
2849
6373
|
}
|
|
@@ -2882,6 +6406,286 @@ declare enum ClientEventTypes {
|
|
|
2882
6406
|
DATA_CONSUMER_CLOSED = "DATA_CONSUMER_CLOSED"
|
|
2883
6407
|
}
|
|
2884
6408
|
|
|
6409
|
+
/** Thresholds deciding when a client counts as degraded on the receiving / sending side. */
|
|
6410
|
+
type ClientHealthThresholds = {
|
|
6411
|
+
/** Inbound loss fraction across the client's received streams (0..1). */
|
|
6412
|
+
inboundFractionLost: number;
|
|
6413
|
+
/** Loss fraction reported back about the client's sent streams via RTCP (0..1). */
|
|
6414
|
+
outboundFractionLost: number;
|
|
6415
|
+
/** Round-trip time (ms). */
|
|
6416
|
+
rttInMs: number;
|
|
6417
|
+
/** Freezes observed across the client's inbound video in the tick. */
|
|
6418
|
+
freezeCount: number;
|
|
6419
|
+
/** Concealment fraction across the client's inbound audio (0..1). */
|
|
6420
|
+
concealmentRatio: number;
|
|
6421
|
+
};
|
|
6422
|
+
declare const defaultClientHealthThresholds: ClientHealthThresholds;
|
|
6423
|
+
/** The per-client health view, split by direction. */
|
|
6424
|
+
type ClientHealth = {
|
|
6425
|
+
observedClient: ObservedClient;
|
|
6426
|
+
clientId: string;
|
|
6427
|
+
/** Receiving (download) side is impaired. */
|
|
6428
|
+
inboundDegraded: boolean;
|
|
6429
|
+
/** Sending (upload) side is impaired. */
|
|
6430
|
+
outboundDegraded: boolean;
|
|
6431
|
+
/** `inboundDegraded || outboundDegraded`. */
|
|
6432
|
+
degraded: boolean;
|
|
6433
|
+
reasons: string[];
|
|
6434
|
+
inboundFractionLost?: number;
|
|
6435
|
+
outboundFractionLost?: number;
|
|
6436
|
+
rttInMs?: number;
|
|
6437
|
+
deltaFreezeCount: number;
|
|
6438
|
+
concealmentRatio?: number;
|
|
6439
|
+
/** Quality-limitation reasons seen on this client's outbound video ('cpu' | 'bandwidth' | …). */
|
|
6440
|
+
qualityLimitationReasons: string[];
|
|
6441
|
+
usingTURN: boolean;
|
|
6442
|
+
usingTCP: boolean;
|
|
6443
|
+
};
|
|
6444
|
+
/** Call-level rollup of the per-client health, using percentiles rather than means. */
|
|
6445
|
+
type CallHealth = {
|
|
6446
|
+
callId: string;
|
|
6447
|
+
clients: ClientHealth[];
|
|
6448
|
+
numberOfClients: number;
|
|
6449
|
+
numberOfDegradedClients: number;
|
|
6450
|
+
numberOfInboundDegradedClients: number;
|
|
6451
|
+
numberOfOutboundDegradedClients: number;
|
|
6452
|
+
/** degradedClients / clients (0..1). */
|
|
6453
|
+
degradedRatio: number;
|
|
6454
|
+
inboundDegradedRatio: number;
|
|
6455
|
+
outboundDegradedRatio: number;
|
|
6456
|
+
rttInMs?: StatsSummary;
|
|
6457
|
+
inboundFractionLost?: StatsSummary;
|
|
6458
|
+
concealmentRatio?: StatsSummary;
|
|
6459
|
+
/** How many clients reported each quality-limitation reason on their outbound video. */
|
|
6460
|
+
qualityLimitation: {
|
|
6461
|
+
cpu: number;
|
|
6462
|
+
bandwidth: number;
|
|
6463
|
+
other: number;
|
|
6464
|
+
};
|
|
6465
|
+
freezes: {
|
|
6466
|
+
affectedClients: number;
|
|
6467
|
+
total: number;
|
|
6468
|
+
};
|
|
6469
|
+
};
|
|
6470
|
+
/**
|
|
6471
|
+
* Aggregates a call along the **client** axis (as `TrackDistributionAggregator` does along the
|
|
6472
|
+
* publisher→subscriber axis): per-client health split into sending vs receiving, plus percentile
|
|
6473
|
+
* rollups and "affected ratio" counts for the whole call.
|
|
6474
|
+
*
|
|
6475
|
+
* Build once per call and call `aggregate()` on each `call.update()`.
|
|
6476
|
+
*/
|
|
6477
|
+
declare class CallHealthAggregator {
|
|
6478
|
+
private readonly _call;
|
|
6479
|
+
readonly thresholds: ClientHealthThresholds;
|
|
6480
|
+
constructor(_call: ObservedCall, thresholds?: ClientHealthThresholds);
|
|
6481
|
+
aggregate(): CallHealth;
|
|
6482
|
+
private _clientHealth;
|
|
6483
|
+
}
|
|
6484
|
+
|
|
6485
|
+
/** An entry retained by {@link SlidingWindow}. */
|
|
6486
|
+
type SlidingWindowEntry<T> = {
|
|
6487
|
+
timestamp: number;
|
|
6488
|
+
value: T;
|
|
6489
|
+
};
|
|
6490
|
+
/**
|
|
6491
|
+
* A time-bounded buffer used by detectors that reason over a window ("N of M clients degraded within
|
|
6492
|
+
* 10 s"). Entries older than `windowMs` are evicted on write and on read.
|
|
6493
|
+
*
|
|
6494
|
+
* ### Ordering is enforced, not assumed
|
|
6495
|
+
*
|
|
6496
|
+
* Eviction walks from the front and stops at the first entry still inside the window, which is only
|
|
6497
|
+
* correct if entries are ordered by timestamp. Callers mostly pass `Date.now()` and are ordered by
|
|
6498
|
+
* construction — but not always: a timestamp taken from a client sample, or two calls inside the
|
|
6499
|
+
* same millisecond, can arrive out of order, and one such entry would park itself at the head and
|
|
6500
|
+
* stop eviction *permanently*, so the window would grow without bound and keep reporting symptoms
|
|
6501
|
+
* from hours ago.
|
|
6502
|
+
*
|
|
6503
|
+
* Rather than trust the caller, {@link add} inserts in timestamp order. Appending (the overwhelmingly
|
|
6504
|
+
* common case) stays O(1); an out-of-order insert costs a short backward scan, because such entries
|
|
6505
|
+
* are near the tail in practice.
|
|
6506
|
+
*
|
|
6507
|
+
* ### The window advances on the newest observation
|
|
6508
|
+
*
|
|
6509
|
+
* Eviction is relative to the largest timestamp seen, not the one just passed. A caller that reads
|
|
6510
|
+
* with a `now` behind the newest entry (a replayed sample, a clock that stepped back) would
|
|
6511
|
+
* otherwise un-evict nothing and, worse, a caller passing an old `now` to {@link add} would evict
|
|
6512
|
+
* everything newer.
|
|
6513
|
+
*/
|
|
6514
|
+
declare class SlidingWindow<T> {
|
|
6515
|
+
readonly windowMs: number;
|
|
6516
|
+
/** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
|
|
6517
|
+
readonly maxEntries: number;
|
|
6518
|
+
private _entries;
|
|
6519
|
+
private _latest;
|
|
6520
|
+
constructor(windowMs: number,
|
|
6521
|
+
/** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
|
|
6522
|
+
maxEntries?: number);
|
|
6523
|
+
/** Add an entry (defaults to `Date.now()`), then evict anything outside the window. */
|
|
6524
|
+
add(value: T, timestamp?: number): void;
|
|
6525
|
+
/**
|
|
6526
|
+
* The entries still inside the window, oldest first.
|
|
6527
|
+
*
|
|
6528
|
+
* A **copy** — callers routinely map/sort what they get back, and handing out the live array let
|
|
6529
|
+
* them mutate the window from the outside.
|
|
6530
|
+
*/
|
|
6531
|
+
entries(now?: number): SlidingWindowEntry<T>[];
|
|
6532
|
+
/** The values still inside the window, oldest first. */
|
|
6533
|
+
values(now?: number): T[];
|
|
6534
|
+
/**
|
|
6535
|
+
* How many entries are inside the window as of `now`.
|
|
6536
|
+
*
|
|
6537
|
+
* Prefer this to `values(now).length`: counting through {@link values} allocates an array of every
|
|
6538
|
+
* entry only to read its length, which on a hot path is the whole cost of the call.
|
|
6539
|
+
*/
|
|
6540
|
+
count(now?: number): number;
|
|
6541
|
+
/** Retained entries, without evicting first. See {@link count} for the windowed answer. */
|
|
6542
|
+
get size(): number;
|
|
6543
|
+
clear(): void;
|
|
6544
|
+
private _evict;
|
|
6545
|
+
}
|
|
6546
|
+
|
|
6547
|
+
type TrendTesterConfig = {
|
|
6548
|
+
/**
|
|
6549
|
+
* How many of the most recent values to keep. Default `30`, floored at `2`.
|
|
6550
|
+
*
|
|
6551
|
+
* The one knob controlling how far back "trend" looks, for both tests — there is deliberately no
|
|
6552
|
+
* separate window per test. This counts *samples*, so the time it spans depends on how often you
|
|
6553
|
+
* push. Mann-Kendall needs roughly 8–10 points before its significance test is worth anything, so
|
|
6554
|
+
* below ~`10` it will mostly answer `no-trend`. Typical `20`–`60`: long enough for a stable
|
|
6555
|
+
* baseline, short enough that a sustained change eventually becomes the new normal instead of being
|
|
6556
|
+
* flagged forever.
|
|
6557
|
+
*/
|
|
6558
|
+
size?: number;
|
|
6559
|
+
/**
|
|
6560
|
+
* Page-Hinkley's **drift tolerance** — change smaller than this is treated as noise and never
|
|
6561
|
+
* accumulated. Default `0`. See {@link pageHinkley}.
|
|
6562
|
+
*
|
|
6563
|
+
* Expressed in the units of whatever you push, so there is no universally good value: for RTT in ms
|
|
6564
|
+
* a few ms is a reasonable tolerance. The default `0` accumulates *every* deviation, which is the
|
|
6565
|
+
* most sensitive setting. Too low and ordinary fluctuation accumulates into a false step change; too
|
|
6566
|
+
* high and a real but gradual shift never accumulates at all.
|
|
6567
|
+
*/
|
|
6568
|
+
pageHinkleyDelta?: number;
|
|
6569
|
+
/**
|
|
6570
|
+
* Page-Hinkley's **detection threshold** — how much accumulated drift counts as a step change.
|
|
6571
|
+
* Default `50`. See {@link pageHinkley}.
|
|
6572
|
+
*
|
|
6573
|
+
* Also in your units, and the direct sensitivity control: lower detects smaller or earlier steps and
|
|
6574
|
+
* false-positives more; higher waits for unmistakable ones. Worth tuning against a recorded series
|
|
6575
|
+
* rather than by intuition, because the right value depends entirely on the scale and noisiness of
|
|
6576
|
+
* the metric you feed it.
|
|
6577
|
+
*/
|
|
6578
|
+
pageHinkleyLambda?: number;
|
|
6579
|
+
/**
|
|
6580
|
+
* Mann-Kendall's significance level. Default `0.05`. See {@link mannKendallVerdict}.
|
|
6581
|
+
*
|
|
6582
|
+
* Conventional values are `0.01`, `0.05` and `0.1`. This is the probability of claiming a monotonic
|
|
6583
|
+
* trend that is not really there: `0.01` is stricter and slower to call a trend, `0.1` more
|
|
6584
|
+
* sensitive and noisier.
|
|
6585
|
+
*/
|
|
6586
|
+
mannKendallAlpha?: number;
|
|
6587
|
+
};
|
|
6588
|
+
/**
|
|
6589
|
+
* Streaming home for `stats.ts`'s two trend tests: feed it one value at a time via {@link add}
|
|
6590
|
+
* instead of re-running the batch functions over an array you manage yourself.
|
|
6591
|
+
*
|
|
6592
|
+
* ### The two tests answer different questions
|
|
6593
|
+
*
|
|
6594
|
+
* Take a client's RTT, sampled every couple of seconds. Two things can go wrong with it, and only
|
|
6595
|
+
* one of them looks like a spike:
|
|
6596
|
+
*
|
|
6597
|
+
* - **Mann-Kendall** asks *"is this drifting?"* — a monotonic trend, regardless of shape or scale.
|
|
6598
|
+
* `40, 45, 52, 61, 70, 84 ms` is a rising path with no single dramatic step; every jump is small
|
|
6599
|
+
* and plausible on its own. Mann-Kendall counts how many later samples exceed earlier ones and
|
|
6600
|
+
* reports whether that lopsidedness could plausibly be chance. It is rank-based, so one absurd
|
|
6601
|
+
* reading (a 4000 ms outlier from a stalled event loop) moves it by exactly one pair, not by the
|
|
6602
|
+
* 4000.
|
|
6603
|
+
* - **Page-Hinkley** asks *"did it change, and when?"* — a step. `40, 42, 39, 41, 180, 176, 182 ms`
|
|
6604
|
+
* is not a trend at all; it is one level followed by a different level, which is what a route
|
|
6605
|
+
* change or a TURN failover looks like. It accumulates the deviation from the running mean and
|
|
6606
|
+
* fires when the cumulative excess passes `lambda`.
|
|
6607
|
+
*
|
|
6608
|
+
* Neither subsumes the other, which is why both live here on one window. A slow climb toward
|
|
6609
|
+
* unusability shows up in Mann-Kendall and never trips Page-Hinkley; a hard failover trips
|
|
6610
|
+
* Page-Hinkley immediately while Mann-Kendall may read `no-trend`, because after the step the series
|
|
6611
|
+
* is flat again. `tests/trendTester.spec.ts` builds both RTT series and shows exactly this.
|
|
6612
|
+
*
|
|
6613
|
+
* ```ts
|
|
6614
|
+
* const rtt = new TrendTester({ size: 30, mannKendallAlpha: 0.05, pageHinkleyLambda: 50 });
|
|
6615
|
+
*
|
|
6616
|
+
* peerConnection.on('update', () => {
|
|
6617
|
+
* if (peerConnection.currentRttInMs === undefined) return; // no measurement is not a measurement
|
|
6618
|
+
* rtt.add(peerConnection.currentRttInMs);
|
|
6619
|
+
*
|
|
6620
|
+
* if (rtt.mannKendall().trend === 'increasing') warn('RTT is drifting up');
|
|
6621
|
+
* if (rtt.pageHinkley()?.changeDetected) {
|
|
6622
|
+
* warn('RTT stepped');
|
|
6623
|
+
* rtt.clear(); // the old level is no longer the baseline — judge the new one on its own
|
|
6624
|
+
* }
|
|
6625
|
+
* });
|
|
6626
|
+
* ```
|
|
6627
|
+
*
|
|
6628
|
+
* ### Both read the same window
|
|
6629
|
+
*
|
|
6630
|
+
* `size` is the one knob controlling how far back either test looks. They are kept incremental
|
|
6631
|
+
* differently, because they don't tolerate an evicted point the same way:
|
|
6632
|
+
*
|
|
6633
|
+
* - **Mann-Kendall**'s statistic is a sum over *pairs*, so evicting the oldest value only touches
|
|
6634
|
+
* the pairs it was part of — one pass over the (bounded) window corrects it in O(size) instead of
|
|
6635
|
+
* the O(size²) a full recompute costs.
|
|
6636
|
+
* - **Page-Hinkley**'s statistic is a running minimum of a cumulative sum, which has no cheap
|
|
6637
|
+
* correction for "forget this one old point" — the minimum may have depended on it. It is
|
|
6638
|
+
* recomputed from the window on every {@link add} rather than hand-rolling an incremental version
|
|
6639
|
+
* that would be easy to get subtly wrong. That recompute is O(size), the same order as above.
|
|
6640
|
+
*
|
|
6641
|
+
* ### Non-finite input is rejected, not absorbed
|
|
6642
|
+
*
|
|
6643
|
+
* See {@link add}. A single `NaN` would otherwise destroy the instance permanently.
|
|
6644
|
+
*/
|
|
6645
|
+
declare class TrendTester {
|
|
6646
|
+
private readonly _size;
|
|
6647
|
+
private readonly _values;
|
|
6648
|
+
private readonly _tieCounts;
|
|
6649
|
+
private readonly _pageHinkleyDelta;
|
|
6650
|
+
private readonly _pageHinkleyLambda;
|
|
6651
|
+
private readonly _mannKendallAlpha;
|
|
6652
|
+
private _s;
|
|
6653
|
+
private _rejected;
|
|
6654
|
+
private _pageHinkleyResult?;
|
|
6655
|
+
constructor(config?: TrendTesterConfig);
|
|
6656
|
+
/** Values rejected by {@link add} for being non-finite. Non-zero means the caller has a bug. */
|
|
6657
|
+
get rejected(): number;
|
|
6658
|
+
/** How many values are currently in the window (`<= size`). */
|
|
6659
|
+
get length(): number;
|
|
6660
|
+
/** The configured window length. */
|
|
6661
|
+
get size(): number;
|
|
6662
|
+
/**
|
|
6663
|
+
* Add the next value in the stream, evicting the oldest once the window is full.
|
|
6664
|
+
*
|
|
6665
|
+
* **Non-finite values are rejected** rather than stored, and the rejection is counted in
|
|
6666
|
+
* {@link rejected}. This is not defensive noise — it is the difference between a bad reading and
|
|
6667
|
+
* a bad instance. `Math.sign(NaN)` is `NaN`, so a single `NaN` would poison the incremental
|
|
6668
|
+
* Mann-Kendall sum `_s` **permanently**: every later `add` and `_evictOldest` adds or subtracts
|
|
6669
|
+
* `NaN`, the z-score is `NaN`, every comparison against it is `false`, and the tester silently
|
|
6670
|
+
* reports `no-trend` forever after. It would also take a `NaN` key in `_tieCounts` that can never
|
|
6671
|
+
* be matched on eviction, since `NaN !== NaN`.
|
|
6672
|
+
*
|
|
6673
|
+
* `undefined` RTT (no measurement this tick) must not be coerced to `0` and passed in either —
|
|
6674
|
+
* "we didn't measure" is not "the trip took no time", and feeding zeros manufactures a downward
|
|
6675
|
+
* trend. Skip the sample instead.
|
|
6676
|
+
*/
|
|
6677
|
+
add(value: number): void;
|
|
6678
|
+
/** The current Page-Hinkley read-out over the window. `undefined` before the first value. */
|
|
6679
|
+
pageHinkley(): PageHinkleyResult | undefined;
|
|
6680
|
+
/** The current Mann-Kendall read-out over the window. */
|
|
6681
|
+
mannKendall(): MannKendallResult;
|
|
6682
|
+
/** Drop everything, e.g. after a detected change point, to start judging the trend fresh. */
|
|
6683
|
+
clear(): void;
|
|
6684
|
+
/** Remove the oldest value from the window and correct `_s` for the pairs it was part of. */
|
|
6685
|
+
private _evictOldest;
|
|
6686
|
+
private _bumpTie;
|
|
6687
|
+
}
|
|
6688
|
+
|
|
2885
6689
|
interface Logger {
|
|
2886
6690
|
trace(...args: any[]): void;
|
|
2887
6691
|
debug(...args: any[]): void;
|
|
@@ -2953,4 +6757,4 @@ declare function createInMemorySink(samples?: ClientSample[]): InMemorySink;
|
|
|
2953
6757
|
declare function createDefaultMediasoupRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
|
|
2954
6758
|
declare function createP2pRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
|
|
2955
6759
|
|
|
2956
|
-
export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type CallAppDataFactory, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type Detector, Detectors, InMemorySink, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, type Logger, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, type ObserverEventBase, type ObserverEvents, type ObserverLogger, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolvers, type SampleRejectedReason, type ScoreCalculator, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, setObserverLogger };
|
|
6760
|
+
export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, type ActiveIssueTracker, ActiveIssuesRegistry, type AvailableCallScopeDetectorsConfigs, type AvailableDetectorsConfigs, type AvailableObserverScopeDetectorsConfigs, type AvailableValidatorConfigs, CODEC_MISMATCH_ISSUE, type CalculatedScore, type CallAppDataFactory, CallConcurrentIssueDetector, type CallConcurrentIssueDetectorConfig, type CallConcurrentIssueGroup, CallConcurrentIssueTypes, type CallHealth, CallHealthAggregator, type CallIssue, type CallIssueSpread, type CallScopedEventName, type CallSummary, type CallSummaryClients, CallSummaryCollector, type CallSummaryConfig, type CallSummaryEnricher, type CallSummaryEnrichers, type CallSummaryScores, type CallSummarySection, type CallSummaryTruncation, type CallSummaryTurnServers, type CertificateStats, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientLocation, type ClientLocationResolver, type ClientMetaData, ClientMetaTypes, type ClientPopulation, type ClientPopulationAxis, ClientPopulationIssueDetector, type ClientPopulationIssueDetectorConfig, ClientPopulationIssueTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type CodecConsistencyReportPayload, CodecConsistencyValidator, type CodecConsistencyValidatorConfig, type CodecEvidence, type CodecStats, type CorroboratedPublisherFault, type DataChannelStats, type Detector, Detectors, type ExtensionStat, GEOHASH_CELL_SIZES, type IceCandidatePairStats, type IceCandidateStats, type IceTransportStats, InMemorySink, type InboundRtpStats, type InboundTrackSample, type Issue, type IssueBase, type IssueConclusion, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, type IssueFaultDomain, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, LOWEST_COMMON_DENOMINATOR_ISSUE, type Logger, type MannKendallResult, type MediaKind, type MediaPlayoutStats, type MediaSourceStats, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, MiddlewareProcessor, ObservedCall, type ObservedCallScope, type ObservedCallSettings, ObservedCertificate, ObservedClient, ObservedClientIssueRegistry, type ObservedClientScope, type ObservedClientSettings, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, ObservedTURN, type ObservedTURNEventMap, ObservedTurnServer, Observer, ObserverConcurrentIssueDetector, type ObserverConcurrentIssueDetectorConfig, type ObserverConcurrentIssueGroup, ObserverConcurrentIssueTypes, type ObserverEventBase, type ObserverEvents, type ObserverIssue, type ObserverIssueSpread, type ObserverLogger, type OutboundRtpStats, type OutboundTrackSample, type PageHinkleyResult, type PeerConnectionSample, type PeerConnectionTransportStats, type PsnrSum, PublisherFaultCorroborationDetector, type PublisherFaultCorroborationDetectorConfig, PublisherFaultTypes, type QualityLimitationDurations, RESOLVED_ISSUE_SUFFIX, type RemoteInboundRtpStats, type RemoteOutboundRtpStats, type RemoteTrackLinkEvidence, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolverReportPayload, RemoteTrackResolverValidator, type RemoteTrackResolverValidatorConfig, type RemoteTrackResolvers, type ResolvedActiveClientIssue, type RunningValidator, type SampleRejectedReason, type ScoreCalculator, SfuCongestionDetector, type SfuCongestionDetectorBucket, type SfuCongestionDetectorConfig, type SfuCongestionDetectorEvaluation, type SfuCongestionDetectorReport, type SimulcastReceiverEvidence, type SimulcastReceiverReportPayload, SimulcastReceiverValidator, type SimulcastReceiverValidatorConfig, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrendTester, type TrendTesterConfig, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, TurnServerOutageDetector, type TurnServerOutageDetectorConfig, TurnServerOutageTypes, UNRESOLVED_TRACK_LINKS_ISSUE, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, type ValidationReport, type Validator, type ValidatorName, baseIssueType, concludeCallIssue, concludeObserverIssue, correlation, counterDelta, createCallSummary, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultCallSummaryConfig, defaultClientHealthThresholds, geohash, isClientIssueResolutionEntry, issuePayloadAsString, mannKendall, mannKendallVerdict, median, medianAbsoluteDeviation, pageHinkley, percentile, percentileOfSorted, robustZScore, schemaVersion, setObserverLogger, summarize };
|