@observertc/observer-js 1.0.0-beta.9 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.mts CHANGED
@@ -1,6 +1,7 @@
1
1
  import { EventEmitter } from 'events';
2
2
  import { types } from 'mediasoup';
3
3
 
4
+ declare const schemaVersion = "3.7.0";
4
5
  /**
5
6
  * The WebRTC app provided custom stats payload
6
7
  */
@@ -12,7 +13,7 @@ type ExtensionStat = {
12
13
  /**
13
14
  * The payload of the extension stats the custom app provides
14
15
  */
15
- payload?: string;
16
+ payload?: Record<string, unknown>;
16
17
  };
17
18
  /**
18
19
  * A list of additional client events.
@@ -23,9 +24,9 @@ type ClientMetaData = {
23
24
  */
24
25
  type: string;
25
26
  /**
26
- * The value associated with the event, if applicable.
27
+ * The attributes of the meta data entry, if applicable.
27
28
  */
28
- payload?: string;
29
+ payload?: Record<string, unknown>;
29
30
  /**
30
31
  * The unique identifier of the peer connection for which the event was generated.
31
32
  */
@@ -52,9 +53,13 @@ type ClientIssue = {
52
53
  */
53
54
  type: string;
54
55
  /**
55
- * The value associated with the event, if applicable.
56
+ * Identifier of the related issue or resolution when it is provided.
56
57
  */
57
- payload?: string;
58
+ key?: string;
59
+ /**
60
+ * The attributes of the issue, if applicable.
61
+ */
62
+ payload?: Record<string, unknown>;
58
63
  /**
59
64
  * The timestamp in epoch format when the event was generated.
60
65
  */
@@ -69,16 +74,16 @@ type ClientEvent = {
69
74
  */
70
75
  type: string;
71
76
  /**
72
- * The value associated with the event, if applicable.
77
+ * The attributes of the event, if applicable.
73
78
  */
74
- payload?: string;
79
+ payload?: Record<string, unknown>;
75
80
  /**
76
81
  * The timestamp in epoch format when the event was generated.
77
82
  */
78
83
  timestamp?: number;
79
84
  };
80
85
  /**
81
- * Certificates
86
+ * Certificate Stats
82
87
  */
83
88
  type CertificateStats = {
84
89
  /**
@@ -102,7 +107,7 @@ type CertificateStats = {
102
107
  */
103
108
  base64Certificate?: string;
104
109
  /**
105
- * The certificate ID of the issuer (nullable).
110
+ * The certificate ID of the issuer.
106
111
  */
107
112
  issuerCertificateId?: string;
108
113
  /**
@@ -119,7 +124,7 @@ type IceCandidatePairStats = {
119
124
  */
120
125
  id: string;
121
126
  /**
122
- * The timestamp of when the stats were recorded, in seconds.
127
+ * The timestamp of when the stats were recorded, in milliseconds.
123
128
  */
124
129
  timestamp: number;
125
130
  /**
@@ -134,7 +139,10 @@ type IceCandidatePairStats = {
134
139
  * The ID of the remote ICE candidate in this pair.
135
140
  */
136
141
  remoteCandidateId?: string;
137
- state?: 'new' | 'in-progress' | 'waiting' | 'failed' | 'succeeded' | 'cancelled' | 'inprogress';
142
+ /**
143
+ * The checklist state of this candidate pair. Values follow the W3C RTCStatsIceCandidatePairState enum (frozen, waiting, in-progress, failed, succeeded). Two further values are accepted for backward compatibility and are not part of the current spec: `new` (never standardised) and `cancelled` (removed from the spec after 2016).
144
+ */
145
+ state?: "new" | "frozen" | "in-progress" | "waiting" | "failed" | "succeeded" | "cancelled" | "inprogress";
138
146
  /**
139
147
  * Whether this candidate pair has been nominated.
140
148
  */
@@ -229,7 +237,7 @@ type IceCandidateStats = {
229
237
  */
230
238
  transportId?: string;
231
239
  /**
232
- * The IP address of the ICE candidate (nullable).
240
+ * The IP address of the ICE candidate.
233
241
  */
234
242
  address?: string;
235
243
  /**
@@ -358,7 +366,15 @@ type IceTransportStats = {
358
366
  */
359
367
  selectedCandidatePairChanges?: number;
360
368
  /**
361
- * Additional information attached to this stats
369
+ * Number of congestion control feedback (CCFB) messages sent on this transport.
370
+ */
371
+ ccfbMessagesSent?: number;
372
+ /**
373
+ * Number of congestion control feedback (CCFB) messages received on this transport.
374
+ */
375
+ ccfbMessagesReceived?: number;
376
+ /**
377
+ * Additional information attached to this stats.
362
378
  */
363
379
  attachments?: Record<string, unknown>;
364
380
  };
@@ -539,7 +555,7 @@ type MediaSourceStats = {
539
555
  attachments?: Record<string, unknown>;
540
556
  };
541
557
  /**
542
- * Remote Outbound RTPs
558
+ * Remote Outbound RTP Stats
543
559
  */
544
560
  type RemoteOutboundRtpStats = {
545
561
  /**
@@ -625,7 +641,24 @@ type QualityLimitationDurations = {
625
641
  other: number;
626
642
  };
627
643
  /**
628
- * Outbound RTPs
644
+ * Cumulative PSNR measurements for Y, U, V components.
645
+ */
646
+ type PsnrSum = {
647
+ /**
648
+ * PSNR value for the Y (luminance) component.
649
+ */
650
+ y: number;
651
+ /**
652
+ * PSNR value for the U (chrominance) component.
653
+ */
654
+ u: number;
655
+ /**
656
+ * PSNR value for the V (chrominance) component.
657
+ */
658
+ v: number;
659
+ };
660
+ /**
661
+ * Outbound RTP Stats
629
662
  */
630
663
  type OutboundRtpStats = {
631
664
  /**
@@ -677,6 +710,10 @@ type OutboundRtpStats = {
677
710
  */
678
711
  rid?: string;
679
712
  /**
713
+ * Index of the encoding in the encoding array.
714
+ */
715
+ encodingIndex?: number;
716
+ /**
680
717
  * The total number of header bytes sent on this stream.
681
718
  */
682
719
  headerBytesSent?: number;
@@ -733,6 +770,14 @@ type OutboundRtpStats = {
733
770
  */
734
771
  qpSum?: number;
735
772
  /**
773
+ * Cumulative PSNR measurements for Y, U, V components.
774
+ */
775
+ psnrSum?: PsnrSum;
776
+ /**
777
+ * Total number of PSNR measurements collected.
778
+ */
779
+ psnrMeasurements?: number;
780
+ /**
736
781
  * The total time spent encoding frames on this stream in seconds.
737
782
  */
738
783
  totalEncodeTime?: number;
@@ -741,10 +786,14 @@ type OutboundRtpStats = {
741
786
  */
742
787
  totalPacketSendDelay?: number;
743
788
  /**
744
- * The reason for any quality limitation on this stream.
789
+ * The reason for any quality limitation on this stream (e.g., 'cpu', 'bandwidth', 'other').
745
790
  */
746
791
  qualityLimitationReason?: string;
747
792
  /**
793
+ * The duration of quality limitation reasons categorized by type.
794
+ */
795
+ qualityLimitationDurations?: QualityLimitationDurations;
796
+ /**
748
797
  * The number of resolution changes due to quality limitations.
749
798
  */
750
799
  qualityLimitationResolutionChanges?: number;
@@ -765,7 +814,7 @@ type OutboundRtpStats = {
765
814
  */
766
815
  encoderImplementation?: string;
767
816
  /**
768
- * Indicates whether the encoder is power efficient.
817
+ * Indicates whether the encoder is power-efficient.
769
818
  */
770
819
  powerEfficientEncoder?: boolean;
771
820
  /**
@@ -777,16 +826,16 @@ type OutboundRtpStats = {
777
826
  */
778
827
  scalabilityMode?: string;
779
828
  /**
780
- * The duration of quality limitation reasons categorized by type.
829
+ * Number of packets sent with ECT(1) congestion marking.
781
830
  */
782
- qualityLimitationDurations?: QualityLimitationDurations;
831
+ packetsSentWithEct1?: number;
783
832
  /**
784
833
  * Additional information attached to this stats.
785
834
  */
786
835
  attachments?: Record<string, unknown>;
787
836
  };
788
837
  /**
789
- * Remote Inbound RTPs
838
+ * Remote Inbound RTP Stats
790
839
  */
791
840
  type RemoteInboundRtpStats = {
792
841
  /**
@@ -818,6 +867,22 @@ type RemoteInboundRtpStats = {
818
867
  */
819
868
  packetsReceived?: number;
820
869
  /**
870
+ * Total number of RTP packets received for this SSRC marked with the ECT(1) marking.
871
+ */
872
+ packetsReceivedWithEct1?: number;
873
+ /**
874
+ * Total number of RTP packets received for this SSRC marked with the CE marking.
875
+ */
876
+ packetsReceivedWithCe?: number;
877
+ /**
878
+ * Total number of RTP packets for which an RFC8888 report has been sent with a zero R bit.
879
+ */
880
+ packetsReportedAsLost?: number;
881
+ /**
882
+ * Total number of RTP packets reported as lost but later recovered in a subsequent RFC8888 report.
883
+ */
884
+ packetsReportedAsLostButRecovered?: number;
885
+ /**
821
886
  * The total number of packets lost on this stream.
822
887
  */
823
888
  packetsLost?: number;
@@ -846,12 +911,16 @@ type RemoteInboundRtpStats = {
846
911
  */
847
912
  roundTripTimeMeasurements?: number;
848
913
  /**
914
+ * Number of packets with ECT(1) marking that were bleached by a middlebox.
915
+ */
916
+ packetsWithBleachedEct1Marking?: number;
917
+ /**
849
918
  * Additional information attached to this stats
850
919
  */
851
920
  attachments?: Record<string, unknown>;
852
921
  };
853
922
  /**
854
- * Inbound RTPs
923
+ * Inbound RTP Stats
855
924
  */
856
925
  type InboundRtpStats = {
857
926
  /**
@@ -887,6 +956,22 @@ type InboundRtpStats = {
887
956
  */
888
957
  packetsReceived?: number;
889
958
  /**
959
+ * Total number of RTP packets received for this SSRC marked with the ECT(1) marking.
960
+ */
961
+ packetsReceivedWithEct1?: number;
962
+ /**
963
+ * Total number of RTP packets received for this SSRC marked with the CE marking.
964
+ */
965
+ packetsReceivedWithCe?: number;
966
+ /**
967
+ * Total number of RTP packets for which an RFC8888 report has been sent with a zero R bit.
968
+ */
969
+ packetsReportedAsLost?: number;
970
+ /**
971
+ * Total number of RTP packets reported as lost but later recovered in a subsequent RFC8888 report.
972
+ */
973
+ packetsReportedAsLostButRecovered?: number;
974
+ /**
890
975
  * Number of packets lost on the RTP stream.
891
976
  */
892
977
  packetsLost?: number;
@@ -895,7 +980,7 @@ type InboundRtpStats = {
895
980
  */
896
981
  jitter?: number;
897
982
  /**
898
- * The MediaStream ID of the RTP stream.
983
+ * The media stream identification tag from the SDP media section.
899
984
  */
900
985
  mid?: string;
901
986
  /**
@@ -991,15 +1076,15 @@ type InboundRtpStats = {
991
1076
  */
992
1077
  bytesReceived?: number;
993
1078
  /**
994
- * Number of NACKs sent.
1079
+ * Number of NACKs received.
995
1080
  */
996
1081
  nackCount?: number;
997
1082
  /**
998
- * Number of Full Intra Requests sent.
1083
+ * Number of Full Intra Requests received.
999
1084
  */
1000
1085
  firCount?: number;
1001
1086
  /**
1002
- * Number of Picture Loss Indications sent.
1087
+ * Number of Picture Loss Indications received.
1003
1088
  */
1004
1089
  pliCount?: number;
1005
1090
  /**
@@ -1177,10 +1262,14 @@ type OutboundTrackSample = {
1177
1262
  */
1178
1263
  kind: string;
1179
1264
  /**
1180
- * Calculated score for track (details should be added to attachments)
1265
+ * Calculated score for track (details should be added to scoreReasons)
1181
1266
  */
1182
1267
  score?: number;
1183
1268
  /**
1269
+ * Reasons for the score calculation, mapping each reason to how much it contributed to the score
1270
+ */
1271
+ scoreReasons?: Record<string, number>;
1272
+ /**
1184
1273
  * Additional information attached to this stats
1185
1274
  */
1186
1275
  attachments?: Record<string, unknown>;
@@ -1202,16 +1291,20 @@ type InboundTrackSample = {
1202
1291
  */
1203
1292
  kind: string;
1204
1293
  /**
1205
- * Calculated score for track (details should be added to attachments)
1294
+ * Calculated score for track (details should be added to scoreReasons)
1206
1295
  */
1207
1296
  score?: number;
1208
1297
  /**
1298
+ * Reasons for the score calculation, mapping each reason to how much it contributed to the score
1299
+ */
1300
+ scoreReasons?: Record<string, number>;
1301
+ /**
1209
1302
  * Additional information attached to this stats
1210
1303
  */
1211
1304
  attachments?: Record<string, unknown>;
1212
1305
  };
1213
1306
  /**
1214
- * docs
1307
+ * A sample containing statistics and metrics for a WebRTC peer connection
1215
1308
  */
1216
1309
  type PeerConnectionSample = {
1217
1310
  /**
@@ -1223,10 +1316,14 @@ type PeerConnectionSample = {
1223
1316
  */
1224
1317
  attachments?: Record<string, unknown>;
1225
1318
  /**
1226
- * Calculated score for peer connection (details should be added to attachments)
1319
+ * Calculated score for peer connection (details should be added to scoreReasons)
1227
1320
  */
1228
1321
  score?: number;
1229
1322
  /**
1323
+ * Reasons for the score calculation, mapping each reason to how much it contributed to the score
1324
+ */
1325
+ scoreReasons?: Record<string, number>;
1326
+ /**
1230
1327
  * Inbound Track Stats items
1231
1328
  */
1232
1329
  inboundTracks?: InboundTrackSample[];
@@ -1239,19 +1336,19 @@ type PeerConnectionSample = {
1239
1336
  */
1240
1337
  codecs?: CodecStats[];
1241
1338
  /**
1242
- * Inbound RTPs
1339
+ * Inbound RTP Stats
1243
1340
  */
1244
1341
  inboundRtps?: InboundRtpStats[];
1245
1342
  /**
1246
- * Remote Inbound RTPs
1343
+ * Remote Inbound RTP Stats
1247
1344
  */
1248
1345
  remoteInboundRtps?: RemoteInboundRtpStats[];
1249
1346
  /**
1250
- * Outbound RTPs
1347
+ * Outbound RTP Stats
1251
1348
  */
1252
1349
  outboundRtps?: OutboundRtpStats[];
1253
1350
  /**
1254
- * Remote Outbound RTPs
1351
+ * Remote Outbound RTP Stats
1255
1352
  */
1256
1353
  remoteOutboundRtps?: RemoteOutboundRtpStats[];
1257
1354
  /**
@@ -1283,7 +1380,7 @@ type PeerConnectionSample = {
1283
1380
  */
1284
1381
  iceCandidatePairs?: IceCandidatePairStats[];
1285
1382
  /**
1286
- * Certificates
1383
+ * Certificate Stats
1287
1384
  */
1288
1385
  certificates?: CertificateStats[];
1289
1386
  };
@@ -1308,10 +1405,14 @@ type ClientSample = {
1308
1405
  */
1309
1406
  attachments?: Record<string, unknown>;
1310
1407
  /**
1311
- * Calculated score for client (details should be added to attachments)
1408
+ * Calculated score for client (details should be added to scoreReasons)
1312
1409
  */
1313
1410
  score?: number;
1314
1411
  /**
1412
+ * Reasons for the score calculation, mapping each reason to how much it contributed to the score
1413
+ */
1414
+ scoreReasons?: Record<string, number>;
1415
+ /**
1315
1416
  * Samples taken PeerConnections
1316
1417
  */
1317
1418
  peerConnections?: PeerConnectionSample[];
@@ -1585,6 +1686,24 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
1585
1686
  bitPerPixel: number;
1586
1687
  deltaPacketsSent: number;
1587
1688
  deltaBytesSent: number;
1689
+ deltaFramesSent: number;
1690
+ deltaFramesEncoded: number;
1691
+ deltaKeyFramesEncoded: number;
1692
+ deltaNackCount: number;
1693
+ deltaPliCount: number;
1694
+ deltaFirCount: number;
1695
+ deltaRetransmittedPacketsSent: number;
1696
+ deltaRetransmittedBytesSent: number;
1697
+ deltaEncodeTime: number;
1698
+ deltaQualityLimitationResolutionChanges: number;
1699
+ /**
1700
+ * `true` when the codec, encoder implementation or scalability mode changed in this tick.
1701
+ *
1702
+ * Chrome resets `packetsSent`/`bytesSent` on the SSRC when the codec switches
1703
+ * (crbug.com/webrtc/5361), producing sawtooth spikes and negative bitrates. All deltas for the
1704
+ * tick are suppressed so a codec change is never mistaken for a traffic event.
1705
+ */
1706
+ counterResetBoundary: boolean;
1588
1707
  remoteRttInMs?: number;
1589
1708
  remoteFractionLost?: number;
1590
1709
  remoteJitter?: number;
@@ -1599,6 +1718,241 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
1599
1718
  update(stats: OutboundRtpStats): void;
1600
1719
  }
1601
1720
 
1721
+ /**
1722
+ * Small statistics helpers used by the aggregators and detectors.
1723
+ *
1724
+ * Rationale: a mean is a poor summary for call telemetry — one participant with a 1500 ms RTT
1725
+ * skews the average for nine healthy ones. Detectors should reason with medians, high percentiles
1726
+ * and "affected ratios" instead.
1727
+ */
1728
+ /** A distribution summary of a numeric sample set. */
1729
+ type StatsSummary = {
1730
+ count: number;
1731
+ min: number;
1732
+ max: number;
1733
+ mean: number;
1734
+ median: number;
1735
+ p25: number;
1736
+ p75: number;
1737
+ p95: number;
1738
+ };
1739
+ /**
1740
+ * The p-th percentile (0..1) using linear interpolation between closest ranks.
1741
+ * Returns `undefined` for an empty input.
1742
+ */
1743
+ declare function percentile(values: number[], p: number): number | undefined;
1744
+ /** The median (50th percentile). `undefined` for an empty input. */
1745
+ declare function median(values: number[]): number | undefined;
1746
+ /**
1747
+ * Median absolute deviation: `median(|value - median(values)|)`. A robust alternative to standard
1748
+ * deviation for describing how spread out `values` are — one wild outlier shifts a mean-based
1749
+ * deviation a lot, but barely moves a median. `undefined` for an empty input.
1750
+ */
1751
+ declare function medianAbsoluteDeviation(values: number[]): number | undefined;
1752
+ /**
1753
+ * A robust z-score: how many (MAD-based) standard deviations `value` sits above/below the median of
1754
+ * `baseline`.
1755
+ *
1756
+ * Uses the median and MAD instead of the mean and standard deviation so a handful of baseline
1757
+ * outliers can't inflate the "normal" spread and mask a genuine spike — the same reasoning behind
1758
+ * {@link summarize}'s percentiles applies here to a single scalar spread. `1.4826` is the constant
1759
+ * that makes MAD estimate the standard deviation of a normal distribution, so the result reads on
1760
+ * the same scale as a classic z-score.
1761
+ *
1762
+ * A classic z-score divides by zero once every baseline value is identical (`MAD === 0`). Here:
1763
+ * `value` strictly above that constant baseline reads as `Infinity` — a spike with no precedent
1764
+ * whatsoever, however small; `value` at or below it reads as `0`, indistinguishable from the (flat)
1765
+ * baseline rather than a division error.
1766
+ *
1767
+ * ### Worked example
1768
+ *
1769
+ * A baseline of "share of clients reporting congestion", one entry per 10 s bucket, on a healthy
1770
+ * fleet: `[0.02, 0.01, 0.03, 0.04, 0.02]`. Median `0.02`, MAD `0.01`, so the scale is
1771
+ * `1.4826 * 0.01 ≈ 0.0148`.
1772
+ *
1773
+ * ```ts
1774
+ * robustZScore(0.03, baseline); // ≈ 0.67 — an ordinary bucket
1775
+ * robustZScore(0.25, baseline); // ≈ 15.5 — a quarter of the fleet at once; nothing like it before
1776
+ * ```
1777
+ *
1778
+ * Now add one bad bucket to the *baseline* — `[0.02, 0.01, 0.03, 0.04, 0.02, 0.40]`. A mean/stddev
1779
+ * z-score would absorb it: the mean climbs to `0.087` and the stddev to `~0.14`, so a genuine `0.25`
1780
+ * spike scores about `1.2` and looks unremarkable. **One past incident would hide the next one.**
1781
+ * Median and MAD barely move (median `0.025`, MAD `0.01`), so `0.25` still scores `≈15.2`. That
1782
+ * resistance is the entire reason for this function.
1783
+ *
1784
+ * The `Infinity` case is not an edge case in practice — it is a fleet that has been perfectly quiet:
1785
+ *
1786
+ * ```ts
1787
+ * robustZScore(0.10, [ 0, 0, 0, 0, 0 ]); // Infinity — first congestion ever seen
1788
+ * robustZScore(0, [ 0, 0, 0, 0, 0 ]); // 0 — still nothing happening
1789
+ * ```
1790
+ *
1791
+ * `Infinity` clears any finite threshold, which is intended: "we have never seen this" *is* the
1792
+ * strongest possible statistical statement. It is also why a caller must gate on practical
1793
+ * significance too — see `SfuCongestionDetector`, which additionally requires a minimum number of
1794
+ * affected clients, so a single client on a quiet fleet cannot page anyone.
1795
+ *
1796
+ * `undefined` when `baseline` is empty — there is nothing to compare against.
1797
+ */
1798
+ declare function robustZScore(value: number, baseline: number[]): number | undefined;
1799
+ /** Summarize a numeric sample set. Returns `undefined` for an empty input. */
1800
+ declare function summarize(values: number[]): StatsSummary | undefined;
1801
+ /**
1802
+ * Linear-interpolated percentile of an **already ascending** array. The building block behind
1803
+ * {@link percentile} and {@link summarize}; exported so callers computing several percentiles over
1804
+ * the same data can sort once themselves.
1805
+ *
1806
+ * Passing an unsorted array yields a meaningless number rather than an error — sort first.
1807
+ */
1808
+ declare function percentileOfSorted(sorted: number[], p: number): number;
1809
+ /**
1810
+ * A counter-reset-safe delta: the increase of a cumulative counter between two observations.
1811
+ *
1812
+ * Returns `0` when the counter went backwards (reset / SSRC reuse) or when either side is missing.
1813
+ * NOTE the guard is `>=` on **defined** values — a previous value of `0` is a perfectly valid
1814
+ * baseline, so `0 -> 5` correctly yields `5` (using a truthiness check here silently drops the
1815
+ * first interval of every counter, which is exactly when the first loss/freeze event happens).
1816
+ */
1817
+ declare function counterDelta(previous: number | undefined, current: number | undefined): number;
1818
+ /**
1819
+ * Pearson correlation of two equal-length series, clamped to `0..1`.
1820
+ *
1821
+ * Negative or undefined relationships read as `0`, because every caller here asks "does A follow B?"
1822
+ * — an inverse relationship is not a weaker yes, it is a no.
1823
+ */
1824
+ declare function correlation(xs: number[], ys: number[]): number;
1825
+ /** Per-step result of {@link pageHinkley}. */
1826
+ type PageHinkleyResult = {
1827
+ /**
1828
+ * The Page-Hinkley statistic after each observation, same length as the input — one value per
1829
+ * `values[i]`, so `statistic[i]` is "how far the cumulative deviation has grown above its own
1830
+ * historical low, using only `values[0..i]`". Never negative (it's a gap to a *minimum*), and
1831
+ * `0` for as long as the process looks stable.
1832
+ */
1833
+ statistic: number[];
1834
+ /**
1835
+ * Index of the **first** observation whose statistic exceeded `lambda`, else `undefined`.
1836
+ *
1837
+ * This is a one-shot latch over the call's whole input: once found, later observations are not
1838
+ * re-checked, even if the statistic subsequently falls back down (e.g. after a single transient
1839
+ * spike — see the class doc example). "Is it *still* elevated right now" is a different
1840
+ * question, answered by looking at the *tail* of {@link statistic}, or — for a live stream — by
1841
+ * re-running this over a recent window each time (which is what `TrendTester` does) rather than
1842
+ * over the whole history once.
1843
+ */
1844
+ changePointIndex?: number;
1845
+ /** `true` when {@link changePointIndex} is defined. */
1846
+ changeDetected: boolean;
1847
+ };
1848
+ /**
1849
+ * Page-Hinkley test: sequential (online) detection of a sustained **increase** in the mean of
1850
+ * `values` — "has this metric settled onto a durably higher level", as opposed to "did one sample
1851
+ * come in high". A single noisy point should not read as a regression; ten points that are all a
1852
+ * bit higher than before should.
1853
+ *
1854
+ * ### How it works
1855
+ *
1856
+ * At step `i`, three numbers are tracked:
1857
+ *
1858
+ * 1. `mean` — the running average of `values[0..i]` (**not** a fixed baseline — it is recomputed
1859
+ * from everything seen so far, including `values[i]` itself, which is what makes this
1860
+ * *adaptive*: after a real shift, `mean` keeps drifting up to meet the new level, and the
1861
+ * signal below fades back out on its own rather than staying triggered forever).
1862
+ * 2. `cumulative` — running sum of `(values[i] - mean - delta)`. Subtracting `mean` centres each
1863
+ * term on "surprise relative to what we've seen so far"; subtracting `delta` on top means a
1864
+ * small positive surprise still nets to a *negative* contribution, so it doesn't accumulate.
1865
+ * 3. `runningMinimum` — the lowest `cumulative` has ever been.
1866
+ *
1867
+ * The **Page-Hinkley statistic** is `cumulative - runningMinimum`: how far the running sum has
1868
+ * climbed above its own historical floor. Pure noise pulls `cumulative` up and down around a flat
1869
+ * trend, so the gap to `runningMinimum` stays small. A sustained increase pushes `cumulative`
1870
+ * mostly one direction — up — so `runningMinimum` stops updating and the gap grows every step,
1871
+ * crossing `lambda` once the shift is large/long enough to be sure it isn't noise. That first
1872
+ * crossing is {@link PageHinkleyResult.changePointIndex}.
1873
+ *
1874
+ * ### Parameters
1875
+ *
1876
+ * - `delta` — the **drift tolerance** (in the same units as `values`, e.g. ms of RTT): how much of
1877
+ * a step-to-step increase is written off as noise rather than counted towards the cumulative sum.
1878
+ * `0` means even a razor-thin, consistent upward creep eventually accumulates enough to trigger;
1879
+ * raising it requires each observation to clear that bar above the running mean before it
1880
+ * contributes anything (see the class doc's hand-worked example — the same jump that triggers
1881
+ * with `delta: 0` is completely absorbed at `delta: 10`).
1882
+ * - `lambda` — the **detection threshold** the statistic must exceed. It is in "surprise units" (a
1883
+ * sum of deviations, not a single observation's units), so there's no shortcut for picking it
1884
+ * other than trying it against representative data — see the RTT examples in `stats.spec.ts` for
1885
+ * a worked comparison of a low vs. a high `lambda` on the same series. Raising it delays
1886
+ * detection but makes a false positive from a lucky run of noise less likely.
1887
+ *
1888
+ * O(n) — one pass, unlike {@link mannKendall}'s O(n²). To detect a **decrease** instead of an
1889
+ * increase, negate `values` before calling (or negate the result's meaning if you'd rather).
1890
+ */
1891
+ declare function pageHinkley(values: number[], delta?: number, lambda?: number): PageHinkleyResult;
1892
+ /** Result of {@link mannKendall}. */
1893
+ type MannKendallResult = {
1894
+ /** Sum of pairwise signs (`sign(values[j] - values[i])` for every `i < j`). */
1895
+ s: number;
1896
+ /** Variance of {@link s}, corrected for tied values. */
1897
+ variance: number;
1898
+ /** Standard-normal score derived from `s`. `0` when `s` is `0` — no evidence either way. */
1899
+ z: number;
1900
+ /** Two-tailed p-value for the null hypothesis "no monotonic trend". */
1901
+ pValue: number;
1902
+ /** `'increasing'` / `'decreasing'` when significant at `alpha`, else `'no-trend'`. */
1903
+ trend: 'increasing' | 'decreasing' | 'no-trend';
1904
+ };
1905
+ /**
1906
+ * Mann-Kendall trend test: a non-parametric test for a **monotonic** trend in `values`, without
1907
+ * assuming a distribution or a constant rate of change — it only asks "are later values
1908
+ * consistently larger (or smaller) than earlier ones more often than chance would allow".
1909
+ *
1910
+ * Every pair `i < j` votes `+1` (`values[j] > values[i]`), `-1` (`values[j] < values[i]`) or `0`
1911
+ * (tie); `s` is the sum of those votes. Under the null hypothesis of no trend, `s` is
1912
+ * approximately normal with mean `0` and a known variance (corrected here for tied values), which
1913
+ * turns `s` into a Z score and a two-tailed p-value. `trend` is only `'increasing'` / `'decreasing'`
1914
+ * when that p-value clears `alpha` — a handful of mostly-ascending points is exactly what noise
1915
+ * looks like half the time, and this is what keeps that from reading as a trend.
1916
+ *
1917
+ * O(n²) (every pair is compared); fine for the small, per-tick sample counts detectors work with,
1918
+ * not for large historical series.
1919
+ */
1920
+ declare function mannKendall(values: number[], alpha?: number): MannKendallResult;
1921
+ /**
1922
+ * Turns a Mann-Kendall `s` statistic and its `variance` into a Z score, p-value and verdict — the
1923
+ * back half of {@link mannKendall}, split out so an incremental caller that maintains `s` /
1924
+ * `variance` itself (e.g. over a sliding window, correcting for evicted points rather than
1925
+ * recomputing every pair from scratch) doesn't have to reimplement the normal approximation.
1926
+ */
1927
+ declare function mannKendallVerdict(s: number, variance: number, alpha?: number): Pick<MannKendallResult, 'z' | 'pValue' | 'trend'>;
1928
+
1929
+ /** The per-receiver view the aggregator builds for one subscribed track. */
1930
+ type PublishedTrackReceivingDistributionEntry = {
1931
+ numberOfReceivers: number;
1932
+ numberOfHealthyReceivers: number;
1933
+ numberOfDegradedReceivers: number;
1934
+ /** degradedReceivers / receivers (0..1); `0` when there are no receivers. */
1935
+ degradedRatio: number;
1936
+ /** Distribution summaries across receivers (undefined when no receiver reported the metric). */
1937
+ bitrate?: StatsSummary;
1938
+ fractionLost?: StatsSummary;
1939
+ jitter?: StatsSummary;
1940
+ rttInMs?: StatsSummary;
1941
+ jitterBufferDelayInMs?: StatsSummary;
1942
+ concealmentRatio?: StatsSummary;
1943
+ /** Fan-out counters: how many receivers saw the symptom, and the total across them. */
1944
+ freezes: {
1945
+ affectedReceivers: number;
1946
+ total: number;
1947
+ };
1948
+ plis: {
1949
+ affectedReceivers: number;
1950
+ total: number;
1951
+ };
1952
+ concealment: {
1953
+ affectedReceivers: number;
1954
+ };
1955
+ };
1602
1956
  declare class ObservedOutboundTrack implements OutboundTrackSample {
1603
1957
  timestamp: number;
1604
1958
  readonly id: string;
@@ -1609,18 +1963,28 @@ declare class ObservedOutboundTrack implements OutboundTrackSample {
1609
1963
  private _visited;
1610
1964
  appData?: Record<string, unknown>;
1611
1965
  readonly remoteInboundTracks: Set<ObservedInboundTrack>;
1966
+ readonly detectors: Detectors;
1967
+ receivingDistribution?: PublishedTrackReceivingDistributionEntry;
1612
1968
  readonly calculatedScore: CalculatedScore;
1613
1969
  addedAt?: number | undefined;
1614
1970
  removedAt?: number | undefined;
1615
1971
  muted?: boolean;
1616
1972
  attachments?: Record<string, unknown> | undefined;
1973
+ degradedReasons?: string[] | undefined;
1974
+ bitrate?: number | undefined;
1975
+ deltaPacketsSent?: number | undefined;
1976
+ remoteFractionLost?: number | undefined;
1977
+ remoteRttInMs?: number | undefined;
1978
+ qualityLimitationReason?: string | undefined;
1617
1979
  constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _outboundRtps?: ObservedOutboundRtp[] | undefined, _mediaSource?: ObservedMediaSource | undefined);
1618
1980
  get score(): number | undefined;
1619
1981
  get visited(): boolean;
1982
+ get degraded(): boolean;
1620
1983
  getPeerConnection(): ObservedPeerConnection;
1621
1984
  getOutboundRtps(): ObservedOutboundRtp[] | undefined;
1622
1985
  getMediaSource(): ObservedMediaSource | undefined;
1623
1986
  update(stats: OutboundTrackSample): void;
1987
+ private createReceivingDistribution;
1624
1988
  }
1625
1989
 
1626
1990
  declare class ObservedInboundTrack implements InboundTrackSample {
@@ -1638,6 +2002,8 @@ declare class ObservedInboundTrack implements InboundTrackSample {
1638
2002
  removedAt?: number | undefined;
1639
2003
  muted?: boolean;
1640
2004
  attachments?: Record<string, unknown> | undefined;
2005
+ degradationReasons: string[];
2006
+ get degraded(): boolean;
1641
2007
  constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _inboundRtp?: ObservedInboundRtp | undefined, _mediaPlayout?: ObservedMediaPlayout | undefined);
1642
2008
  get score(): number | undefined;
1643
2009
  get visited(): boolean;
@@ -1645,6 +2011,7 @@ declare class ObservedInboundTrack implements InboundTrackSample {
1645
2011
  getInboundRtp(): ObservedInboundRtp | undefined;
1646
2012
  getMediaPlayout(): ObservedMediaPlayout | undefined;
1647
2013
  update(stats: InboundTrackSample): void;
2014
+ private checkDegradation;
1648
2015
  }
1649
2016
 
1650
2017
  declare class ObservedRemoteOutboundRtp implements RemoteOutboundRtpStats {
@@ -1752,6 +2119,46 @@ declare class ObservedInboundRtp implements InboundRtpStats {
1752
2119
  deltaBytesReceived: number;
1753
2120
  deltaReceivedSamples: number;
1754
2121
  deltaSilentConcealedSamples: number;
2122
+ deltaConcealedSamples: number;
2123
+ deltaConcealmentEvents: number;
2124
+ deltaFreezeCount: number;
2125
+ deltaFreezesDuration: number;
2126
+ deltaPliCount: number;
2127
+ deltaNackCount: number;
2128
+ deltaFirCount: number;
2129
+ deltaPacketsDiscarded: number;
2130
+ deltaFramesDecoded: number;
2131
+ deltaFramesReceived: number;
2132
+ deltaFramesRendered: number;
2133
+ deltaFramesDropped: number;
2134
+ deltaKeyFramesDecoded: number;
2135
+ deltaDecodeTime: number;
2136
+ deltaJitterBufferDelay: number;
2137
+ deltaJitterBufferEmittedCount: number;
2138
+ deltaRetransmittedPacketsReceived: number;
2139
+ deltaFecPacketsReceived: number;
2140
+ deltaFecPacketsDiscarded: number;
2141
+ deltaPausesDuration: number;
2142
+ /**
2143
+ * Mean jitter-buffer delay for the frames/samples emitted in this tick (seconds), derived from
2144
+ * the cumulative `jitterBufferDelay` / `jitterBufferEmittedCount` pair — the only correct way to
2145
+ * read those two counters.
2146
+ */
2147
+ jitterBufferDelayInMs?: number;
2148
+ /** Fraction of the samples received in this tick that were concealed (0..1). */
2149
+ concealmentRatio?: number;
2150
+ /** Fraction of the frames received in this tick that were dropped before rendering (0..1). */
2151
+ framesDroppedRatio?: number;
2152
+ /**
2153
+ * `true` when the codec or decoder implementation changed in this tick.
2154
+ *
2155
+ * Chrome resets `packetsReceived`/`bytesReceived` on an SSRC when the codec switches
2156
+ * (crbug.com/webrtc/5361, open since 2015), which shows up as a sawtooth spike or a negative
2157
+ * rate. Every delta in this tick is therefore suppressed to `0` rather than reported as traffic
2158
+ * — otherwise a room-wide codec rollout produces a synchronized fake-degradation alert across
2159
+ * every participant at once.
2160
+ */
2161
+ counterResetBoundary: boolean;
1755
2162
  remoteRttInMs?: number;
1756
2163
  remoteBytesSent?: number;
1757
2164
  remotePacketsSent?: number;
@@ -1941,7 +2348,34 @@ declare class ObservedPeerConnection extends EventEmitter {
1941
2348
  sendingVideoBitrate: number;
1942
2349
  receivingAudioBitrate: number;
1943
2350
  receivingVideoBitrate: number;
2351
+ /**
2352
+ * Median round-trip time of the tick, in ms.
2353
+ *
2354
+ * Prefers {@link rtcpRttInMs} and falls back to {@link iceRttInMs}, so within one tick it always
2355
+ * reports **one** kind of round trip. It used to be the median of both mixed together, which was
2356
+ * a bug: the mixing ratio changed as streams came and went, so the value moved for reasons that
2357
+ * had nothing to do with the network.
2358
+ */
1944
2359
  currentRttInMs?: number;
2360
+ /**
2361
+ * RTT measured by ICE/STUN consent checks, in ms — the trip to **whatever terminates ICE**. In an
2362
+ * SFU topology that is the SFU, so this is the client↔SFU leg, not client↔client.
2363
+ */
2364
+ iceRttInMs?: number;
2365
+ /**
2366
+ * RTT reported by RTCP receiver reports, in ms — an **end-to-end** media-path round trip.
2367
+ *
2368
+ * Not the same trip as {@link iceRttInMs}; the difference between the two is roughly the far side
2369
+ * of the SFU (see {@link sfuHopRttInMs}).
2370
+ */
2371
+ rtcpRttInMs?: number;
2372
+ /**
2373
+ * `rtcpRttInMs - iceRttInMs`, when both are known — an estimate of everything *past* the SFU.
2374
+ *
2375
+ * Useful for splitting "this client's own last mile is slow" (high `iceRttInMs`) from "the path
2376
+ * beyond the SFU is slow" (low ICE, high hop).
2377
+ */
2378
+ sfuHopRttInMs?: number;
1945
2379
  currentJitter?: number;
1946
2380
  usingTCP: boolean;
1947
2381
  usingTURN: boolean;
@@ -2036,6 +2470,487 @@ type OperationSystem = {
2036
2470
  version: string;
2037
2471
  };
2038
2472
 
2473
+ /**
2474
+ * Turning a correlation into a **conclusion**.
2475
+ *
2476
+ * Every detector in this library ultimately reports the same shape of observation: *N clients have
2477
+ * issue X open at once, and here is what they have in common*. That is useful but not yet
2478
+ * actionable — someone still has to know that congestion spread across unrelated calls means the
2479
+ * server, while CPU limitation spread across unrelated calls means a bad client release. This module
2480
+ * holds that interpretation step so it is stated once, consistently, instead of being re-derived by
2481
+ * whoever reads the alert at 3am.
2482
+ *
2483
+ * ### Two functions, because there are two questions
2484
+ *
2485
+ * A detector already knows its scope — it was constructed with an `ObservedCall` or with the
2486
+ * `Observer`. Handing that scope back to a single generic function meant every caller supplied
2487
+ * fields the other scope needed and its own scope ignored: a call-scoped detector passing
2488
+ * `affectedCalls: 1, totalCalls: 1` forever, an observer-scoped one passing a participant ratio that
2489
+ * was deliberately never read. Placeholders like that are a standing invitation to read them as if
2490
+ * they meant something.
2491
+ *
2492
+ * So there are two entry points, each taking only the facts its scope actually has:
2493
+ *
2494
+ * - {@link concludeCallIssue} — within one call. The axis is *how much of the meeting*, and whether
2495
+ * the affected clients all subscribe to one published track.
2496
+ * - {@link concludeObserverIssue} — across calls. The axis is *how many independent calls*, which is
2497
+ * the only thing that separates "one bad room" from "our infrastructure".
2498
+ *
2499
+ * Neither the issue family nor the spread concludes anything alone: congestion in one call is a
2500
+ * meeting problem, congestion in six calls is an infrastructure problem, and the issue type is
2501
+ * identical in both.
2502
+ */
2503
+ /** Where the fault most likely sits, given who is affected. */
2504
+ type IssueFaultDomain =
2505
+ /** Independent calls affected at once — they share only the servers and the network. */
2506
+ 'infrastructure'
2507
+ /** One call, broadly affected — something that call shares (its SFU worker, room, or host). */
2508
+ | 'call'
2509
+ /** The subscribers of one published track — the publisher's path or the forwarding of it. */
2510
+ | 'published-track'
2511
+ /** A single endpoint — its own device or last mile. */
2512
+ | 'endpoint'
2513
+ /** Independent calls affected, but by something endpoints own — a client build, not a server. */
2514
+ | 'client-population'
2515
+ /** Not enough signal to attribute. */
2516
+ | 'unknown';
2517
+ /** A stated verdict, attached to the raised issue payload. */
2518
+ type IssueConclusion = {
2519
+ /** Where to look. */
2520
+ faultDomain: IssueFaultDomain;
2521
+ /** One line, written to be readable in an alert without opening a dashboard. */
2522
+ summary: string;
2523
+ /** What to check first. Omitted when the issue family is unknown to this module. */
2524
+ recommendation?: string;
2525
+ /**
2526
+ * How much the spread alone justifies the verdict, `0..1`. Not a probability — a coarse ranking
2527
+ * so alerting can threshold on it. More independent calls, or a tighter onset, means higher.
2528
+ */
2529
+ confidence: number;
2530
+ };
2531
+ /** The facts a **call-scoped** conclusion is drawn from. */
2532
+ type CallIssueSpread = {
2533
+ issueType: string;
2534
+ /** Distinct clients of this call with the issue open. */
2535
+ affectedClients: number;
2536
+ /** Participants in the call — the denominator. */
2537
+ totalClients: number;
2538
+ /** True when the onsets clustered — a shared trigger rather than drift. */
2539
+ onsetBurst: boolean;
2540
+ /**
2541
+ * Set when the affected clients are the subscriber set of **one published track**.
2542
+ *
2543
+ * The strongest call-scoped statement available: those clients share a publisher and nothing
2544
+ * else, so the receivers are exonerated and the source's path is implicated.
2545
+ */
2546
+ publishedTrackId?: string;
2547
+ };
2548
+ /** The facts an **observer-scoped** conclusion is drawn from. */
2549
+ type ObserverIssueSpread = {
2550
+ issueType: string;
2551
+ /** Distinct clients across the fleet with the issue open. */
2552
+ affectedClients: number;
2553
+ /** Clients in the fleet. Reported for context; it does not gate anything at this scope. */
2554
+ totalClients: number;
2555
+ /** Distinct calls containing at least one affected client. The dimension that matters here. */
2556
+ affectedCalls: number;
2557
+ /** Calls in flight. */
2558
+ totalCalls: number;
2559
+ /** True when the onsets clustered. */
2560
+ onsetBurst: boolean;
2561
+ };
2562
+ /**
2563
+ * Draw the conclusion for a group of clients **within one call**.
2564
+ *
2565
+ * Ordered most-to-least specific: a track-scoped group is a stronger statement than a call-wide one,
2566
+ * and a single affected endpoint is not a statement about the call at all.
2567
+ */
2568
+ declare function concludeCallIssue(spread: CallIssueSpread): IssueConclusion;
2569
+ /**
2570
+ * Draw the conclusion for a group of clients spanning **several calls**.
2571
+ *
2572
+ * One affected call is not an observer-scoped finding — it has an obvious local explanation and the
2573
+ * call-scoped detector has already reported it — so that case returns `call` and says so rather than
2574
+ * dressing it up as a fleet event.
2575
+ *
2576
+ * Which domain breadth implicates depends on the family, and this is the whole reason the module
2577
+ * exists: `congestion` across unrelated calls points at the servers, `cpulimitation` across unrelated
2578
+ * calls points at what those *endpoints* share — a client release, a browser version, shared
2579
+ * virtualised hardware — and pointing an SFU team at the second one wastes a night.
2580
+ */
2581
+ declare function concludeObserverIssue(spread: ObserverIssueSpread): IssueConclusion;
2582
+
2583
+ /**
2584
+ * What every server-raised finding carries, whatever raised it.
2585
+ *
2586
+ * ### Why this is not `ClientIssue`
2587
+ *
2588
+ * `ClientIssue` is a **wire** type: it arrives inside a `ClientSample`, so its `payload` is bound to
2589
+ * what the schema can carry — a record of primitives since schema 3.5.0, a JSON string on samples
2590
+ * from earlier clients. Server-raised findings were reusing it, which forced every detector to `JSON.stringify` a
2591
+ * perfectly good object on the way out and every handler to `JSON.parse` it back on the way in —
2592
+ * paying serialisation on a path where nothing is ever serialised, and losing type information in
2593
+ * both directions. These go straight to an in-process handler, so they carry the object.
2594
+ */
2595
+ type IssueBase = {
2596
+ /** What was found, e.g. `'CROSS_CALL_ISSUE_ONSET_BURST'`. */
2597
+ type: string;
2598
+ /** Observer clock, when the finding was raised. */
2599
+ timestamp: number;
2600
+ /**
2601
+ * What the finding *means* — where to look, and how much the evidence justifies it.
2602
+ *
2603
+ * A first-class field rather than a key inside {@link payload}, because it is the one part every
2604
+ * finding has in common and the one part an alerting rule reads. Burying it in the evidence made
2605
+ * `payload.conclusion.faultDomain` the path to the most important thing in the object.
2606
+ */
2607
+ conclusion?: IssueConclusion;
2608
+ /**
2609
+ * The evidence, and **only** the evidence.
2610
+ *
2611
+ * Deliberately does not repeat `type`, `scope`, or the ids already present on the event that
2612
+ * delivers it. A payload that restates its envelope invites the two to disagree — and they did,
2613
+ * because nothing kept them in step.
2614
+ */
2615
+ payload?: Record<string, unknown>;
2616
+ };
2617
+ /**
2618
+ * A finding about **one call**, raised by `observedCall.addIssue()` and delivered as `call-issue`.
2619
+ *
2620
+ * The call is the event's scope (`{ observedCall, observer }`), so the payload does not carry a
2621
+ * `callId` — read it from the event.
2622
+ */
2623
+ type CallIssue = IssueBase & {
2624
+ scope: 'call';
2625
+ };
2626
+ /**
2627
+ * A finding about the **fleet**, raised by `observer.addIssue()` and delivered as `observer-issue`.
2628
+ *
2629
+ * Raised by detectors and validators that reason across calls, so no single call owns it. Where the
2630
+ * finding does concern specific calls — a cross-call correlation, say — they are named in the
2631
+ * evidence, because that *is* the evidence.
2632
+ */
2633
+ type ObserverIssue = IssueBase & {
2634
+ scope: 'observer';
2635
+ };
2636
+ /**
2637
+ * Either kind, discriminated by {@link IssueBase} + `scope`.
2638
+ *
2639
+ * `scope` is on the issue and not merely implied by which event fired, so a finding stays
2640
+ * self-describing once it leaves the bus — funnelled into one handler, a log line, or a queue.
2641
+ */
2642
+ type Issue = CallIssue | ObserverIssue;
2643
+ /**
2644
+ * The payload as a JSON string, for the boundaries that genuinely need text — a log line, an HTTP
2645
+ * body, a message queue.
2646
+ *
2647
+ * Returns `undefined` for a missing payload, or for one that cannot be serialised (a circular
2648
+ * reference from something an application attached): the caller wanted text, not an exception.
2649
+ */
2650
+ declare function issuePayloadAsString(issue: Pick<IssueBase, 'payload'>): string | undefined;
2651
+
2652
+ /**
2653
+ * The bus events that carry an `observedCall`, i.e. the ones an enricher can attribute to a summary.
2654
+ *
2655
+ * Derived from the event map rather than listed by hand, so it cannot drift: adding a call-scoped
2656
+ * event makes it enrichable automatically, and an enricher on an observer-scoped event
2657
+ * (`observer-issue`, `validation-ready`) will not compile — there is no single call it belongs to,
2658
+ * and quietly writing a fleet-wide fact into every open summary would be worse than a type error.
2659
+ */
2660
+ type CallScopedEventName = {
2661
+ [K in keyof ObserverEvents]: ObserverEvents[K][0] extends ObservedCallScope ? K : never;
2662
+ }[keyof ObserverEvents];
2663
+ /** A function that folds one event into the summary. Runs on every occurrence, for its own call. */
2664
+ type CallSummaryEnricher<K extends CallScopedEventName> = (summary: CallSummary, ...args: ObserverEvents[K]) => void;
2665
+ /** The enricher map: any subset of the call-scoped events, each fully typed against its payload. */
2666
+ type CallSummaryEnrichers = {
2667
+ [K in CallScopedEventName]?: CallSummaryEnricher<K>;
2668
+ };
2669
+ /** The built-in sections. Ask for what you want; anything absent is simply not collected. */
2670
+ type CallSummarySection = 'clients' | 'issues' | 'turnServers' | 'scores';
2671
+ type CallSummaryConfig = {
2672
+ /**
2673
+ * Which built-in sections to accumulate. Empty (the default) collects none of them — a summary
2674
+ * with only `enrich` is a perfectly good summary.
2675
+ */
2676
+ include: CallSummarySection[];
2677
+ /**
2678
+ * Fold arbitrary state in from any call-scoped event, into `summary.attachments`. See
2679
+ * {@link CallSummaryEnrichers}.
2680
+ *
2681
+ * Each enricher runs on **every** occurrence of its event, for its own call, so prefer the
2682
+ * low-frequency lifecycle events (`client-joined`, `call-issue`) over `client-updated`, which fires
2683
+ * once per sample per client. Keep them cheap and side-effect free: an enricher that throws is logged
2684
+ * and skipped rather than allowed to disturb the call, but one that is slow is on the ingestion path.
2685
+ */
2686
+ enrich?: CallSummaryEnrichers;
2687
+ /**
2688
+ * Cap on retained issues. Default `500`.
2689
+ *
2690
+ * A bound on memory per call, so scale it by how long your calls run and how noisy they are, not by
2691
+ * taste — a two-hour call with a struggling participant can raise hundreds. Past the cap issues are
2692
+ * dropped and `summary.truncated.issues` counts them, so the real total stays recoverable as
2693
+ * `issues.length + (truncated?.issues ?? 0)`. `0` collects the count only, keeping no issue objects.
2694
+ */
2695
+ maxIssues: number;
2696
+ /**
2697
+ * Cap on retained client ids. Default `10_000`.
2698
+ *
2699
+ * High because the elements are short strings and the usual reason to read a summary is *who was in
2700
+ * this call*. It exists so a webinar-scale room cannot grow the summary without limit. Overflow is
2701
+ * counted in `summary.truncated.clientIds`; `joined`, `left` and `peak` are unaffected by the cap,
2702
+ * since they are counters rather than a list.
2703
+ */
2704
+ maxClientIds: number;
2705
+ };
2706
+ /**
2707
+ * Who was in the call over its whole life — not just who is in it now.
2708
+ *
2709
+ * Deliberately identifiers and counts only. Anything *about* a client — browser, platform, region —
2710
+ * is already on `observedClient` while the call is live, and belongs in `attachments` via an enricher
2711
+ * if you want it kept. Duplicating it here would mean the library deciding which client attributes
2712
+ * matter, and it would mean reading the highest-frequency event on the bus to do it.
2713
+ */
2714
+ type CallSummaryClients = {
2715
+ /** Every client id seen, in join order, capped by `maxClientIds`. */
2716
+ clientIds: string[];
2717
+ /** The most participants present at any one moment. */
2718
+ peak: number;
2719
+ joined: number;
2720
+ left: number;
2721
+ };
2722
+ /** Which TURN relays carried this call's media. */
2723
+ type CallSummaryTurnServers = {
2724
+ serverUrls: string[];
2725
+ /** Distinct clients seen relaying through any of them. */
2726
+ clientsRelayed: number;
2727
+ };
2728
+ /** The call score over time. Percentiles, not a mean — see `utils/stats`. */
2729
+ type CallSummaryScores = {
2730
+ min?: number;
2731
+ max?: number;
2732
+ median?: number;
2733
+ /** How many score readings went into the above. `0` means nothing was measured. */
2734
+ samples: number;
2735
+ };
2736
+ /** What had to be dropped to stay within the caps. Absent when nothing was. */
2737
+ type CallSummaryTruncation = {
2738
+ issues?: number;
2739
+ clientIds?: number;
2740
+ };
2741
+ /**
2742
+ * An accumulating record of one call's life, finalised when the call closes.
2743
+ *
2744
+ * ### Why this exists
2745
+ *
2746
+ * Everything else in this library is about *now*. Detectors answer "is something wrong right now",
2747
+ * validators answer a structural question once, and both read state that the call throws away when
2748
+ * it ends. Nothing kept the answer to *"what happened in that meeting?"* — who was in it, what was
2749
+ * raised, how it scored — and that is the question asked after the call, by support, by billing, by
2750
+ * whoever is writing the incident note.
2751
+ *
2752
+ * ### It is opt-in, and its sections are opt-in
2753
+ *
2754
+ * `observedCall.summary` is `undefined` unless a summary was configured, and each section is present
2755
+ * only if it was requested. **An absent section means "not collected", never "nothing happened"** —
2756
+ * the same rule as `inconclusive` on a validator. Reading `summary.issues` as "this call had no
2757
+ * issues" when `'issues'` was never in `include` is the one misreading this type invites, so it does
2758
+ * not offer a default-empty section to make it easy.
2759
+ *
2760
+ * ### Read it live, receive it once
2761
+ *
2762
+ * The object is live: read `observedCall.summary` at any point during the call. It is also delivered
2763
+ * on the `call-summary` event, emitted inside `close()` while the call is still reachable — after
2764
+ * that the call is gone from `observer.observedCalls` and there is nothing left to ask.
2765
+ */
2766
+ type CallSummary = {
2767
+ callId: string;
2768
+ /** First client join (client clock), as `ObservedCall` computed it. */
2769
+ startedAt?: number;
2770
+ /** Last client leave. */
2771
+ endedAt?: number;
2772
+ /** `endedAt - startedAt`, when both are known. */
2773
+ durationInMs?: number;
2774
+ /** When the summary itself was finalised (observer clock). Set by `close()`. */
2775
+ closedAt?: number;
2776
+ clients?: CallSummaryClients;
2777
+ /**
2778
+ * Every issue raised against this call, in the order they were raised, capped by `maxIssues`.
2779
+ *
2780
+ * Just the issues — no derived tallies. A count is `issues.length`, a per-type count is one
2781
+ * `filter`, and either is cheaper to write at the call site than to keep correct here. The one
2782
+ * thing you cannot derive is what the cap discarded, which is why `truncated.issues` exists:
2783
+ * the issues actually raised is `issues.length + (truncated?.issues ?? 0)`.
2784
+ */
2785
+ issues?: CallIssue[];
2786
+ turnServers?: CallSummaryTurnServers;
2787
+ scores?: CallSummaryScores;
2788
+ /**
2789
+ * Whatever your enrichers put here. The library never writes to it, so it cannot collide with a
2790
+ * section added in a future version.
2791
+ *
2792
+ * **`attachments`, not `appData`, and the distinction is load-bearing.** `appData` is the live
2793
+ * working state an application hangs off an entity for the entity's lifetime, and it may hold
2794
+ * references that cannot be serialised — a mediasoup router, an `RTCPeerConnection`, a socket. A
2795
+ * summary is the opposite: it outlives the call precisely so it can be *shipped* — archived,
2796
+ * queued, written to a column — and it is handed to you on `call-summary` at the moment the call
2797
+ * it came from is being torn down. Anything unserialisable in it is a reference to something
2798
+ * already gone.
2799
+ *
2800
+ * So put serialisable facts here, the same contract as `attachments` on a `ClientSample`. If you
2801
+ * need the live object, read it off `observedCall` / `observedClient` inside the enricher and
2802
+ * attach what you can serialise: the router's `id`, not the router.
2803
+ */
2804
+ attachments: Record<string, unknown>;
2805
+ /**
2806
+ * What the caps discarded, and how much.
2807
+ *
2808
+ * Present **only** when something was actually dropped. A silently truncated summary is worse
2809
+ * than no summary — someone will count `log.length` and report it as the issue count — so the
2810
+ * shortfall is stated rather than left to be inferred from a suspiciously round number.
2811
+ */
2812
+ truncated?: CallSummaryTruncation;
2813
+ };
2814
+ /** The defaults a summary is created with. Caps are generous but finite; see {@link CallSummary}. */
2815
+ declare const defaultCallSummaryConfig: CallSummaryConfig;
2816
+ /** A fresh summary for `callId`, with only the requested sections present. */
2817
+ declare function createCallSummary(callId: string, config: CallSummaryConfig): CallSummary;
2818
+
2819
+ /**
2820
+ * What a validator concluded, once it is done.
2821
+ *
2822
+ * `S` is the validator's own payload — a discriminated union on `verdict` plus whatever evidence
2823
+ * belongs to each outcome — so a report reads in the language of the thing being checked rather than
2824
+ * in generic pass/fail.
2825
+ *
2826
+ * Note this is a plain intersection with `S`, not `{ [K in keyof S]: S[K] }`. A mapped type over a
2827
+ * union collapses to the keys its members *share* (`keyof (A | B)` is the intersection), which would
2828
+ * silently drop every per-verdict evidence field and leave only `verdict` behind.
2829
+ */
2830
+ type ValidationReport<S extends Record<string, unknown> = Record<string, unknown>> =
2831
+ /** Still running — it has not seen the conditions it needs to judge anything yet. */
2832
+ {
2833
+ ready: false;
2834
+ }
2835
+ /** Done. `verdict` is narrowed by `S` to that validator's own vocabulary. */
2836
+ | ({
2837
+ ready: true;
2838
+ verdict: string;
2839
+ decidedAt: number;
2840
+ } & S);
2841
+ /**
2842
+ * A **one-shot structural check**: something true of the deployment rather than of this moment.
2843
+ *
2844
+ * The distinction from a `Detector` is what changes over time. A detector answers "is something wrong
2845
+ * *right now*" — congestion, a dead relay, a track nobody receives — and the answer legitimately
2846
+ * differs every tick, so it runs every tick, forever. A validator answers "is this deployment built
2847
+ * correctly" — does the SFU pick layers per receiver — and that only changes when you deploy. So a
2848
+ * validator **runs until it knows, then finishes**: it calls {@link onDone} once, the observer drops
2849
+ * it, and nothing more is computed.
2850
+ *
2851
+ * To check again — after a deploy, say — start a new one with `observer.addValidator(...)`. There is
2852
+ * no revalidation timer, because a deploy, not the passage of time, is what makes a structural
2853
+ * verdict stale.
2854
+ *
2855
+ * Validators are held in `observer.validators` and driven by `observer.update()`.
2856
+ */
2857
+ interface Validator<S extends Record<string, unknown> = Record<string, unknown>> {
2858
+ readonly name: string;
2859
+ /**
2860
+ * The conclusion so far. `{ ready: false }` until {@link onDone} fires — and **that is not a
2861
+ * pass**. A validator usually needs specific conditions to occur before it can judge anything,
2862
+ * and plenty of deployments never present them.
2863
+ */
2864
+ readonly report: ValidationReport<S>;
2865
+ /** Called exactly once, when the validator finishes. The observer uses this to unregister it. */
2866
+ onDone: (report: ValidationReport<S>) => void;
2867
+ /** Gather evidence; decide if there is now enough. Called on every `observer.update()`. */
2868
+ update(): void;
2869
+ /**
2870
+ * Give up without a verdict. Finishes with `inconclusive`, so a caller waiting on it is freed.
2871
+ *
2872
+ * `reason` is carried into the report. Worth passing something specific — "cancelled" tells the
2873
+ * reader nothing, whereas "sfu redeployed" explains why a check that was running has no verdict.
2874
+ */
2875
+ cancel: (reason?: string) => void;
2876
+ }
2877
+ /**
2878
+ * The part of a validator the observer needs in order to drive it.
2879
+ *
2880
+ * `Validator<S>` is invariant in `S` — `onDone` takes a `ValidationReport<S>` and `report` returns one
2881
+ * — so a `Set<Validator>` cannot hold validators with different payloads. Driving one only needs the
2882
+ * three members that don't mention `S`.
2883
+ */
2884
+ type RunningValidator = Pick<Validator, 'name' | 'update' | 'cancel'>;
2885
+
2886
+ /** The suffix client-monitor-js appends to the type of a resolution entry. */
2887
+ declare const RESOLVED_ISSUE_SUFFIX = "-resolved";
2888
+ /**
2889
+ * A **stateful** client issue the server currently believes to be open.
2890
+ *
2891
+ * client-monitor-js (>= 4.6.0, `sendResolvedIssuesToServer`) puts an issue's whole lifecycle on the
2892
+ * wire as two entries sharing a `key`:
2893
+ *
2894
+ * ```
2895
+ * raise: { type: 'stuck-decoder', key, payload, timestamp: raisedAt }
2896
+ * resolution: { type: 'stuck-decoder-resolved', key, payload: { raisedAt, comment, …final }, timestamp: resolvedAt }
2897
+ * ```
2898
+ *
2899
+ * The observer opens an `ActiveClientIssue` on the raise and closes it on the matching key, which
2900
+ * turns a stream of point-in-time symptom reports into **intervals**. That is what makes the
2901
+ * difference between "several clients reported congestion in the last 10 seconds" (a guess built on
2902
+ * an arbitrary window) and "several clients are congested *right now, simultaneously*" — the latter
2903
+ * being real evidence of a shared cause.
2904
+ *
2905
+ * Issues still active when the client monitor closes are auto-resolved by the client, so a clean
2906
+ * departure does not leak.
2907
+ */
2908
+ type ActiveClientIssue = {
2909
+ /** Identity shared by the raise and its resolution. Unique per client. */
2910
+ key: string;
2911
+ /** The issue type **without** the `-resolved` suffix (e.g. `'congestion'`). */
2912
+ type: string;
2913
+ /** The client that reported it. */
2914
+ clientId: string;
2915
+ /**
2916
+ * The call that client belongs to. The dimension that separates "one bad meeting" from "our
2917
+ * infrastructure": at observer scope, clients in *different* calls share nothing but the server.
2918
+ */
2919
+ callId: string;
2920
+ /** When the client raised it (client clock). */
2921
+ raisedAt: number;
2922
+ /** When the observer first saw it (server clock) — skew-free, use this for cross-client timing. */
2923
+ observedAt: number;
2924
+ /** Parsed raise payload, when it was JSON. */
2925
+ payload?: Record<string, unknown>;
2926
+ /** `payload.peerConnectionId`, when present — most client detectors report it. */
2927
+ peerConnectionId?: string;
2928
+ /** `payload.trackId`, when present — the join key to a track (and thus to a publisher). */
2929
+ trackId?: string;
2930
+ };
2931
+ /** A closed interval: an {@link ActiveClientIssue} plus how it ended. */
2932
+ type ResolvedActiveClientIssue = ActiveClientIssue & {
2933
+ /** When the client resolved it (client clock). */
2934
+ resolvedAt: number;
2935
+ /** Observer-clock resolution time. */
2936
+ observedResolvedAt: number;
2937
+ /** `resolvedAt - raisedAt` as reported by the client, else derived from observer clocks. */
2938
+ durationInMs: number;
2939
+ /** Free-form note passed to `resolveIssue`. */
2940
+ comment?: string;
2941
+ /** Payload explicitly passed at resolution (the built-in detectors pass their final payload). */
2942
+ resolutionPayload?: Record<string, unknown>;
2943
+ /**
2944
+ * How the interval ended: the client said so, the observer expired it, or the client left
2945
+ * without resolving.
2946
+ */
2947
+ resolvedBy: 'client' | 'timeout' | 'client-closed';
2948
+ };
2949
+ /** `true` when the entry is a resolution companion rather than a raise. */
2950
+ declare function isClientIssueResolutionEntry(issue: ClientIssue): boolean;
2951
+ /** Strip the `-resolved` suffix, so both entries of a lifecycle share one logical type. */
2952
+ declare function baseIssueType(type: string): string;
2953
+
2039
2954
  /** The lifecycle events a sink may emit (a subset of Node's writable-stream events). */
2040
2955
  type ClientSampleSinkEvents = {
2041
2956
  /** The destination is fully written and closed (e.g. a file flushed and its fd closed). */
@@ -2103,11 +3018,11 @@ type RtpCodecParameters = {
2103
3018
  parameter?: string;
2104
3019
  }[];
2105
3020
  };
2106
- type SampleHistoryItem<T extends string> = Record<string, unknown> & {
3021
+ type SampleHistoryItem<T extends string> = {
2107
3022
  type: T;
2108
3023
  timestamp: number;
2109
3024
  };
2110
- type MediasoupRouterSample = Record<string, unknown> & {
3025
+ type MediasoupRouterSample = {
2111
3026
  routerId: string;
2112
3027
  attachments: Record<string, unknown>;
2113
3028
  createdAt: number;
@@ -2151,7 +3066,9 @@ type MediasoupDirectTransportSample = {
2151
3066
  type: 'direct';
2152
3067
  history: MediasoupDirectTransportSampleEventMap[];
2153
3068
  };
2154
- type MediasoupTransportSample = Record<string, unknown> & {
3069
+ type MediasoupTransportSample = {
3070
+ /** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
3071
+ attachments?: Record<string, unknown>;
2155
3072
  id: string;
2156
3073
  createdAt: number;
2157
3074
  connectedAt?: number;
@@ -2167,7 +3084,9 @@ type MediasoupProducerSampleEventMap = {
2167
3084
  type MediasoupProducerSampleEvent = {
2168
3085
  [K in keyof MediasoupProducerSampleEventMap]: SampleHistoryItem<K>;
2169
3086
  }[keyof MediasoupProducerSampleEventMap];
2170
- type MediasoupProducerSample = Record<string, unknown> & {
3087
+ type MediasoupProducerSample = {
3088
+ /** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
3089
+ attachments?: Record<string, unknown>;
2171
3090
  id: string;
2172
3091
  transportId: string;
2173
3092
  createdAt: number;
@@ -2191,7 +3110,9 @@ type MediasoupConsumerSampleEventMap = {
2191
3110
  type MediasoupConsumerSampleEvent = {
2192
3111
  [K in keyof MediasoupConsumerSampleEventMap]: SampleHistoryItem<K>;
2193
3112
  }[keyof MediasoupConsumerSampleEventMap];
2194
- type MediasoupConsumerSample = Record<string, unknown> & {
3113
+ type MediasoupConsumerSample = {
3114
+ /** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
3115
+ attachments?: Record<string, unknown>;
2195
3116
  id: string;
2196
3117
  producerId: string;
2197
3118
  transportId: string;
@@ -2200,7 +3121,9 @@ type MediasoupConsumerSample = Record<string, unknown> & {
2200
3121
  kind: 'audio' | 'video';
2201
3122
  history: MediasoupConsumerSampleEvent[];
2202
3123
  };
2203
- type MediasoupDataProducerSample = Record<string, unknown> & {
3124
+ type MediasoupDataProducerSample = {
3125
+ /** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
3126
+ attachments?: Record<string, unknown>;
2204
3127
  id: string;
2205
3128
  transportId: string;
2206
3129
  createdAt: number;
@@ -2208,7 +3131,9 @@ type MediasoupDataProducerSample = Record<string, unknown> & {
2208
3131
  label: string;
2209
3132
  protocol: string;
2210
3133
  };
2211
- type MediasoupDataConsumerSample = Record<string, unknown> & {
3134
+ type MediasoupDataConsumerSample = {
3135
+ /** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
3136
+ attachments?: Record<string, unknown>;
2212
3137
  id: string;
2213
3138
  dataProducerId: string;
2214
3139
  transportId: string;
@@ -2218,30 +3143,142 @@ type MediasoupDataConsumerSample = Record<string, unknown> & {
2218
3143
  protocol: string;
2219
3144
  };
2220
3145
 
3146
+ /**
3147
+ * Declarative enrichment: return the `attachments` to stamp onto an entity's sample the moment it is
3148
+ * created. Called once per entity, before the corresponding `*-sample-added` event.
3149
+ *
3150
+ * The mediasoup object is handed in, so the common case — mirroring mediasoup's own `appData`, where
3151
+ * applications already keep `participantId`, `purpose` and friends — is a one-liner. Returning
3152
+ * `undefined` attaches nothing.
3153
+ */
3154
+ type MediasoupSampleEnricher = {
3155
+ transport?: (transport: types.Transport) => Record<string, unknown> | undefined;
3156
+ producer?: (producer: types.Producer, transport: types.Transport) => Record<string, unknown> | undefined;
3157
+ consumer?: (consumer: types.Consumer, transport: types.Transport) => Record<string, unknown> | undefined;
3158
+ dataProducer?: (dataProducer: types.DataProducer, transport: types.Transport) => Record<string, unknown> | undefined;
3159
+ dataConsumer?: (dataConsumer: types.DataConsumer, transport: types.Transport) => Record<string, unknown> | undefined;
3160
+ };
2221
3161
  type ObservedMediasoupRouterSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
2222
- routerId: string;
2223
3162
  router: types.Router;
2224
3163
  appData?: AppData;
2225
3164
  attachments?: Record<string, unknown>;
3165
+ /** Stamp `attachments` onto each entity sample as it is created. See {@link MediasoupSampleEnricher}. */
3166
+ enrich?: MediasoupSampleEnricher;
2226
3167
  };
3168
+ /**
3169
+ * Lifecycle hooks for building your own report.
3170
+ *
3171
+ * Each entity announces itself as `<entity>-sample-added` when it appears and
3172
+ * `<entity>-sample-closed` when it goes away, carrying **the live sample object** plus the mediasoup
3173
+ * object it came from. Mutating `sample.attachments` inside a handler is the intended way to extend
3174
+ * a sample on the fly — the object you receive is the one held in `observedRouter.sample`, not a copy.
3175
+ */
2227
3176
  type ObservedMediasoupRouterEvents = {
2228
3177
  close: [];
2229
- };
3178
+ 'transport-sample-added': [{
3179
+ sample: MediasoupTransportSample;
3180
+ transport: types.Transport;
3181
+ }];
3182
+ 'transport-sample-closed': [{
3183
+ sample: MediasoupTransportSample;
3184
+ transport: types.Transport;
3185
+ }];
3186
+ 'producer-sample-added': [{
3187
+ sample: MediasoupProducerSample;
3188
+ producer: types.Producer;
3189
+ transport: types.Transport;
3190
+ }];
3191
+ 'producer-sample-closed': [{
3192
+ sample: MediasoupProducerSample;
3193
+ producer: types.Producer;
3194
+ transport: types.Transport;
3195
+ }];
3196
+ 'consumer-sample-added': [{
3197
+ sample: MediasoupConsumerSample;
3198
+ consumer: types.Consumer;
3199
+ transport: types.Transport;
3200
+ }];
3201
+ 'consumer-sample-closed': [{
3202
+ sample: MediasoupConsumerSample;
3203
+ consumer: types.Consumer;
3204
+ transport: types.Transport;
3205
+ }];
3206
+ 'data-producer-sample-added': [{
3207
+ sample: MediasoupDataProducerSample;
3208
+ dataProducer: types.DataProducer;
3209
+ transport: types.Transport;
3210
+ }];
3211
+ 'data-producer-sample-closed': [{
3212
+ sample: MediasoupDataProducerSample;
3213
+ dataProducer: types.DataProducer;
3214
+ transport: types.Transport;
3215
+ }];
3216
+ 'data-consumer-sample-added': [{
3217
+ sample: MediasoupDataConsumerSample;
3218
+ dataConsumer: types.DataConsumer;
3219
+ transport: types.Transport;
3220
+ }];
3221
+ 'data-consumer-sample-closed': [{
3222
+ sample: MediasoupDataConsumerSample;
3223
+ dataConsumer: types.DataConsumer;
3224
+ transport: types.Transport;
3225
+ }];
3226
+ };
2230
3227
  declare interface ObservedMediasoupRouter {
2231
3228
  on<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
2232
3229
  off<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
2233
3230
  once<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
2234
3231
  emit<U extends keyof ObservedMediasoupRouterEvents>(event: U, ...args: ObservedMediasoupRouterEvents[U]): boolean;
2235
3232
  }
3233
+ /**
3234
+ * Observes a live mediasoup `Router` by subscribing to its `observer` API and **accumulates** its
3235
+ * topology and lifecycle into an in-memory `MediasoupRouterSample` (`observedRouter.sample`):
3236
+ * transports, producers, consumers, data producers/consumers, their state-change history and
3237
+ * `createdAt` / `closedAt`. The sample grows for the life of the router (closed entities are kept,
3238
+ * with their `closedAt` set) and is yours to read, snapshot, or persist.
3239
+ *
3240
+ * NOTE: this is intentionally the simplest approach — everything is held in memory. For very large
3241
+ * routers (e.g. ~100 participants producing and consuming on one router, where consumers grow as
3242
+ * O(N²)) this can become substantial; in that case do your own periodic sampling/persistence and
3243
+ * discard what you don't need (see the README's "Memory & large meetings" note).
3244
+ */
2236
3245
  declare class ObservedMediasoupRouter<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
2237
3246
  readonly router: types.Router;
2238
- readonly sample: MediasoupRouterSample;
2239
3247
  appData: AppData;
3248
+ readonly sample: MediasoupRouterSample;
2240
3249
  readonly webrtcTransportIds: Set<string>;
2241
- get attachments(): Record<string, unknown>;
2242
3250
  closed: boolean;
3251
+ private readonly _transportSamples;
3252
+ private readonly _producerSamples;
3253
+ private readonly _consumerSamples;
3254
+ private readonly _dataProducerSamples;
3255
+ private readonly _dataConsumerSamples;
3256
+ private readonly _enrich?;
2243
3257
  constructor(settings: ObservedMediasoupRouterSettings<AppData>);
2244
3258
  get id(): string;
3259
+ get attachments(): Record<string, unknown>;
3260
+ getTransportSample(id: string): MediasoupTransportSample | undefined;
3261
+ getProducerSample(id: string): MediasoupProducerSample | undefined;
3262
+ getConsumerSample(id: string): MediasoupConsumerSample | undefined;
3263
+ getDataProducerSample(id: string): MediasoupDataProducerSample | undefined;
3264
+ getDataConsumerSample(id: string): MediasoupDataConsumerSample | undefined;
3265
+ /**
3266
+ * Merge `attachments` into an entity's sample, whichever kind it is.
3267
+ *
3268
+ * Ids are unique across mediasoup entity kinds, so one method covers all of them. Returns `false`
3269
+ * when the id is unknown — a real answer instead of failing quietly, which matters when the
3270
+ * annotation is driven by application events that may race the mediasoup ones.
3271
+ */
3272
+ attachTo(id: string, attachments: Record<string, unknown>): boolean;
3273
+ /**
3274
+ * A **detached deep copy** of the current sample — the basis for building your own report.
3275
+ *
3276
+ * `this.sample` is live: its arrays grow and its `history` entries are appended as the router
3277
+ * runs, so a report built directly on it keeps changing after you think you're done. This returns
3278
+ * a snapshot that never moves.
3279
+ */
3280
+ snapshot(): MediasoupRouterSample;
3281
+ close(): void;
2245
3282
  addTransport: (transport: types.Transport) => void;
2246
3283
  addWebRtcTransport(transport: types.WebRtcTransport): void;
2247
3284
  addPlainTransport(transport: types.PlainTransport): void;
@@ -2251,8 +3288,19 @@ declare class ObservedMediasoupRouter<AppData extends Record<string, unknown> =
2251
3288
  addConsumer(transport: types.Transport, consumer: types.Consumer): void;
2252
3289
  addDataProducer(transport: types.Transport, dataProducer: types.DataProducer): void;
2253
3290
  addDataConsumer(transport: types.Transport, dataConsumer: types.DataConsumer): void;
2254
- close(): void;
2255
3291
  private attachRouterListeners;
3292
+ private _addTransportSample;
3293
+ private _addProducerSample;
3294
+ private _addConsumerSample;
3295
+ private _addDataProducerSample;
3296
+ private _addDataConsumerSample;
3297
+ /**
3298
+ * Run an enricher and merge what it returns.
3299
+ *
3300
+ * Takes a thunk rather than a value so the **invocation** is inside the guard — application code
3301
+ * runs here, and a throwing enricher must not take the router's bookkeeping down with it.
3302
+ */
3303
+ private _applyEnrichment;
2256
3304
  private _attachTransportObserverListeners;
2257
3305
  }
2258
3306
 
@@ -2293,16 +3341,36 @@ type ObserverEvents = {
2293
3341
  reason: SampleRejectedReason;
2294
3342
  sample: ClientSample;
2295
3343
  }];
3344
+ /** An observer-scoped (cross-call / SFU-wide) finding raised by `observer.addIssue(...)`. */
3345
+ 'observer-issue': [ObserverEventBase & {
3346
+ issue: ObserverIssue;
3347
+ }];
3348
+ /**
3349
+ * A validator decided. Fires once per settle, not per tick — the point of a validator is that it
3350
+ * stops talking once it knows.
3351
+ */
3352
+ 'validation-ready': [ObserverEventBase & {
3353
+ validator: string;
3354
+ report: ValidationReport;
3355
+ }];
2296
3356
  'mediasoup-router-added': [ObservedMediasoupRouterScope];
2297
3357
  'mediasoup-router-removed': [ObservedMediasoupRouterScope];
2298
- 'mediasoup-router-matched-with-call': [ObservedMediasoupRouterScope & ObservedCallScope];
3358
+ 'mediasoup-router-matched-with-peer-connection': [ObservedMediasoupRouterScope & ObservedPeerConnectionScope];
2299
3359
  'call-added': [ObservedCallScope];
2300
3360
  'call-updated': [ObservedCallScope];
2301
3361
  'call-closed': [ObservedCallScope];
2302
3362
  'call-empty': [ObservedCallScope];
2303
3363
  'call-not-empty': [ObservedCallScope];
2304
3364
  'call-issue': [ObservedCallScope & {
2305
- issue: ClientIssue;
3365
+ issue: CallIssue;
3366
+ }];
3367
+ /**
3368
+ * A call's summary was finalised. Emitted from inside `close()`, while the call is still in
3369
+ * `observer.observedCalls` — after that there is nothing left to ask. Only fires for calls that
3370
+ * had a summary configured.
3371
+ */
3372
+ 'call-summary': [ObservedCallScope & {
3373
+ summary: CallSummary;
2306
3374
  }];
2307
3375
  'client-added': [ObservedClientScope];
2308
3376
  'client-sink-created': [ObservedClientScope & {
@@ -2321,6 +3389,14 @@ type ObserverEvents = {
2321
3389
  'client-issue': [ObservedClientScope & {
2322
3390
  issue: ClientIssue;
2323
3391
  }];
3392
+ /**
3393
+ * A stateful client issue ended — the client sent its `<type>-resolved` companion, or the
3394
+ * observer force-closed it because the client went away. Carries the finished **interval**
3395
+ * (`raisedAt` → `resolvedAt`, `durationInMs`).
3396
+ */
3397
+ 'client-issue-resolved': [ObservedClientScope & {
3398
+ resolvedIssue: ResolvedActiveClientIssue;
3399
+ }];
2324
3400
  'client-metadata': [ObservedClientScope & {
2325
3401
  metaData: ClientMetaData;
2326
3402
  }];
@@ -2492,6 +3568,61 @@ type ObserverEvents = {
2492
3568
  }];
2493
3569
  };
2494
3570
 
3571
+ /**
3572
+ * Something that wants to be **handed** open client issues rather than to go looking for them.
3573
+ *
3574
+ * Register one with `activeIssuesRegistry.addIssueTracker(type, tracker)` and it receives every
3575
+ * issue of that type when it opens ({@link add}) and when it closes ({@link delete}). A detector
3576
+ * implementing this pays only for the issues it actually consumes.
3577
+ *
3578
+ * `ActiveIssuesRegistry` implements it too, which is how a call's registry feeds the observer's.
3579
+ */
3580
+ interface ActiveIssueTracker {
3581
+ /** An issue of a subscribed type opened. */
3582
+ add(issue: ActiveClientIssue): void;
3583
+ /**
3584
+ * An issue this tracker was given has closed.
3585
+ *
3586
+ * Return `true` if it was actually held. Returning `false` is legitimate and not an error — a
3587
+ * tracker that counts *occurrences* (see `SfuCongestionDetector`) deliberately ignores
3588
+ * resolutions, because when a symptom ended says nothing about how many endpoints reported it.
3589
+ */
3590
+ delete(issue: ActiveClientIssue): boolean;
3591
+ /** How many issues this tracker currently holds. */
3592
+ size: number;
3593
+ /** Drop everything. Called when the owning scope closes. */
3594
+ clear(): void;
3595
+ has(issue: ActiveClientIssue): boolean;
3596
+ }
3597
+
3598
+ /**
3599
+ * One client's currently **open** stateful issues, keyed by `ClientIssue.key`.
3600
+ *
3601
+ * The server-side mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0):
3602
+ * a raise opens an entry, the matching `<type>-resolved` closes it, and the client's own close
3603
+ * force-resolves whatever is left. That turns point-in-time symptom reports into **intervals**,
3604
+ * which is what lets detectors ask "are these clients broken *at the same time*" rather than "did
3605
+ * they both report something recently".
3606
+ *
3607
+ * Keyed by `key` rather than by type on purpose: one client can have several issues of the same type
3608
+ * open at once (one per track), and they resolve independently.
3609
+ *
3610
+ * Every change is forwarded to the call's `ActiveIssuesRegistry`, which forwards to the observer's —
3611
+ * so the client owns the storage and the wider scopes get their views maintained as it happens.
3612
+ */
3613
+ declare class ObservedClientIssueRegistry {
3614
+ private readonly registry?;
3615
+ private readonly issues;
3616
+ constructor(registry?: ActiveIssueTracker | undefined);
3617
+ get size(): number;
3618
+ keys(): IterableIterator<string>;
3619
+ values(): IterableIterator<ActiveClientIssue>;
3620
+ get(key: string): ActiveClientIssue | undefined;
3621
+ add(issue: ActiveClientIssue): this;
3622
+ remove(key: string): ActiveClientIssue | undefined;
3623
+ clear(): void;
3624
+ }
3625
+
2495
3626
  type ObservedClientSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
2496
3627
  clientId: string;
2497
3628
  appData?: AppData;
@@ -2570,10 +3701,20 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
2570
3701
  totalScoreSum: number;
2571
3702
  numberOfScoreMeasurements: number;
2572
3703
  readonly mediaDevices: MediaDeviceInfo[];
2573
- issues: ClientIssue[];
2574
- private _injections;
3704
+ /**
3705
+ * The client's currently **open** stateful issues, keyed by `ClientIssue.key` — the server-side
3706
+ * mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0).
3707
+ *
3708
+ * This turns point-in-time symptom reports into intervals, which is what lets detectors ask
3709
+ * "are these clients broken *at the same time*" instead of "did they both report something
3710
+ * recently". Entries are opened by a raise, closed by the matching `<type>-resolved` entry, and
3711
+ * force-closed when the client closes.
3712
+ */
3713
+ readonly activeIssues: ObservedClientIssueRegistry;
3714
+ private _pendingInjections;
3715
+ private _activeSample?;
2575
3716
  private closeTimer?;
2576
- constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall);
3717
+ constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall, activeIssues: ObservedClientIssueRegistry);
2577
3718
  get numberOfPeerConnections(): number;
2578
3719
  get score(): number | undefined;
2579
3720
  close(): void;
@@ -2582,13 +3723,28 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
2582
3723
  injectEvent(event: ClientEvent): void;
2583
3724
  injectIssue(issue: ClientIssue): void;
2584
3725
  injectExtensionStat(stat: ExtensionStat): void;
2585
- injectAttachment(key: string, value: unknown): void;
3726
+ injectAttachment(attachments: Record<string, unknown>): void;
2586
3727
  addMetadata(metadata: ClientMetaData): void;
3728
+ /**
3729
+ * Process one `clientIssues[]` entry.
3730
+ *
3731
+ * Entries come in two flavours (client-monitor-js >= 4.6.0):
3732
+ *
3733
+ * - a **raise** — opens an {@link ActiveClientIssue} under `issue.key` and emits `client-issue`;
3734
+ * - a **resolution** — `type` ends in `-resolved` and carries the same `key`; it closes the
3735
+ * matching active issue and emits `client-issue-resolved`.
3736
+ *
3737
+ * Keyless entries are one-shot: reported, never tracked. A re-raise of a key already active
3738
+ * refreshes the payload rather than opening a second interval.
3739
+ */
2587
3740
  addIssue(issue: ClientIssue): void;
2588
3741
  addExtensionStats(stats: ExtensionStat): void;
3742
+ /** Close the active issue a `<type>-resolved` entry refers to, and announce the finished interval. */
3743
+ private _resolveIssue;
2589
3744
  private _processClientEvent;
2590
3745
  private _updatePeerConnection;
2591
- private _mergeInjections;
3746
+ private _mergePendingInjections;
3747
+ private _flushPendingInjections;
2592
3748
  /** Emit an Observer-bus event scoped to this client (or a peer connection under it). */
2593
3749
  private _notify;
2594
3750
  }
@@ -2629,7 +3785,29 @@ declare class RemoteTrackResolver {
2629
3785
  private readonly resolvers;
2630
3786
  private readonly _publisherIdToOutboundTrack;
2631
3787
  private readonly _subscriberIdToInboundTrack;
3788
+ /**
3789
+ * Tracks whose publisher id the strategy could not resolve **yet**.
3790
+ *
3791
+ * A track announces itself once, but its `attachments` are replaced on every sample, so a key that
3792
+ * is missing from the first sample can appear on the second — and a strategy backed by an
3793
+ * application's own mapping (a server-side `ssrc -> producerId` table, say) is inherently racy
3794
+ * against sample arrival. Resolving only at `*-track-added` meant losing those tracks for their
3795
+ * entire life, silently: an unresolvable outbound track never even reaches
3796
+ * `unconsumedOutboundTracks`, so it is invisible to `UnconsumedTrackDetector` too.
3797
+ *
3798
+ * So they wait here and are retried on their own `*-track-updated`, i.e. exactly when new stats
3799
+ * arrived for them. A track leaves on its first successful resolution or on removal, which makes
3800
+ * a linked track cost one `Set.has` per update and bounds these sets by the unresolved tracks
3801
+ * alive right now.
3802
+ */
3803
+ private readonly _pendingInboundTracks;
3804
+ private readonly _pendingOutboundTracks;
2632
3805
  constructor(observedCall: ObservedCall, resolvers: RemoteTrackResolvers);
3806
+ /** Tracks still waiting for a resolvable publisher id. Diagnostics; normally both are empty. */
3807
+ get pendingTrackCounts(): {
3808
+ inbound: number;
3809
+ outbound: number;
3810
+ };
2633
3811
  /** The published (outbound) track for a publisher id, if any. */
2634
3812
  getOutboundTrackByPublisherId(publisherId: string): ObservedOutboundTrack | undefined;
2635
3813
  /** The subscribed (inbound) track for a subscriber id, if the strategy resolves subscriber ids. */
@@ -2642,184 +3820,2398 @@ declare class RemoteTrackResolver {
2642
3820
  private _removeOutboundTrack;
2643
3821
  }
2644
3822
 
2645
- interface Updater {
2646
- readonly name: string;
2647
- readonly description?: string;
2648
- close(): void;
2649
- }
2650
-
2651
3823
  interface Detector {
2652
3824
  readonly name: string;
2653
3825
  /** Called on every entity update; may raise issues via the entity it observes. */
2654
3826
  update(): void;
3827
+ /**
3828
+ * Optional teardown, called when the detector is removed from its registry (or the registry is
3829
+ * cleared, which happens when the owning call/observer closes). Implement it when the detector
3830
+ * subscribes to events or holds timers, so it doesn't leak.
3831
+ */
3832
+ close?(): void;
2655
3833
  }
2656
3834
 
2657
- declare class Detectors {
2658
- private _detectors;
2659
- constructor(...detectors: Detector[]);
2660
- get listOfNames(): string[];
2661
- add(detector: Detector): void;
2662
- remove(detector: Detector): void;
2663
- update(): void;
2664
- clear(): void;
2665
- }
2666
-
2667
- type ObservedCallUpdateConfig = {
2668
- updatePolicy?: 'update-on-any-client-updated' | 'update-when-all-client-updated' | 'none';
3835
+ declare const CallConcurrentIssueTypes: {
3836
+ /** Several participants of this call have the same issue open **at the same time**. */
3837
+ readonly concurrentClientIssues: "CONCURRENT_CLIENT_ISSUES";
3838
+ /** Those issues also *began* together — the signature of one shared event. */
3839
+ readonly issueOnsetBurst: "ISSUE_ONSET_BURST";
2669
3840
  };
2670
- type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = ObservedCallUpdateConfig & {
2671
- callId: string;
2672
- appData?: AppData;
2673
- closeCallIfEmptyForMs?: number;
3841
+ type CallConcurrentIssueDetectorConfig = {
3842
+ /**
3843
+ * The issue types to watch. **Required, and must not be empty** — the detector subscribes to
3844
+ * exactly these and sees nothing else.
3845
+ */
3846
+ issueTypes: string[];
3847
+ /**
3848
+ * Participants the call needs before a ratio means anything. Default `3`.
3849
+ *
3850
+ * In a 1:1 call "half the participants" is one person, which is a client problem and not a call
3851
+ * problem — `2` effectively disables the ratio gate. Raise it for large-meeting products where you
3852
+ * only care once a handful are affected.
3853
+ */
3854
+ minClients: number;
3855
+ /**
3856
+ * Distinct clients that must share the issue. Default `3`.
3857
+ *
3858
+ * The absolute floor under `affectedRatioThreshold`, so a small call cannot clear a ratio with two
3859
+ * unlucky people. Sensible range `2`–`5`; `2` is the lowest that can still mean "more than one
3860
+ * participant", which is the whole premise.
3861
+ */
3862
+ minAffectedClients: number;
3863
+ /**
3864
+ * Fraction of the call's participants that must share it, `0`–`1`. Default `0.5`.
3865
+ *
3866
+ * Typical `0.3`–`0.7`. Lower catches partial events — a subset on one SFU worker — at the cost of
3867
+ * firing on a few coincidentally unhappy participants; `1` demands literally everyone, which real
3868
+ * incidents rarely produce because someone always reconnects first.
3869
+ */
3870
+ affectedRatioThreshold: number;
3871
+ /**
3872
+ * Onsets falling within this span (ms) escalate the finding to `ISSUE_ONSET_BURST` — they did not
3873
+ * just overlap, they started together. Default `2_000`.
3874
+ *
3875
+ * Bound this by your sampling period, not below it: onsets are only known as accurately as clients
3876
+ * report them, so a window shorter than one sampling period can only fire by luck. Typical
3877
+ * `1_000`–`5_000`. Wider makes the escalation meaningless, since unrelated issues drift into the
3878
+ * same window.
3879
+ */
3880
+ onsetBurstWindowInMs: number;
3881
+ /**
3882
+ * Re-arm time per issue type (ms). Default `60_000`.
3883
+ *
3884
+ * A shared event is one incident, not one per tick. Too low and a persistent problem raises an
3885
+ * issue every tick for as long as it lasts; too high and a genuinely new occurrence is swallowed
3886
+ * by the previous one's cooldown. Typical `30_000`–`300_000`.
3887
+ */
3888
+ cooldownMs: number;
2674
3889
  };
2675
- type ObservedCallEvents = {
2676
- update: [];
2677
- newclient: [ObservedClient];
2678
- empty: [];
2679
- 'not-empty': [];
2680
- close: [];
3890
+ /** What the detector currently knows about one issue type in this call. */
3891
+ type CallConcurrentIssueGroup = {
3892
+ type: string;
3893
+ issues: ActiveClientIssue[];
3894
+ clientIds: string[];
3895
+ affectedRatio: number;
3896
+ totalClients: number;
3897
+ /**
3898
+ * Spread of the onsets, in **observer** time (ms) — `max(observedAt) - min(observedAt)`.
3899
+ *
3900
+ * Measured on the observer clock on purpose: `raisedAt` comes from each client's own clock, and
3901
+ * comparing those across machines makes clock skew look like a shared event.
3902
+ */
3903
+ onsetSpreadInMs: number;
3904
+ firstObservedAt: number;
2681
3905
  };
2682
- declare interface ObservedCall {
2683
- on<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
2684
- off<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
2685
- once<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
2686
- emit<U extends keyof ObservedCallEvents>(event: U, ...args: ObservedCallEvents[U]): boolean;
2687
- }
2688
- declare class ObservedCall<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
2689
- readonly observer: Observer;
2690
- updater?: Updater;
2691
- scoreCalculator: ScoreCalculator;
2692
- readonly detectors: Detectors;
2693
- readonly callId: string;
2694
- readonly observedClients: Map<string, ObservedClient<Record<string, unknown>>>;
2695
- readonly clientsUsedTurn: Set<string>;
2696
- readonly calculatedScore: CalculatedScore;
2697
- remoteTrackResolver?: RemoteTrackResolver;
2698
- totalAddedClients: number;
2699
- totalRemovedClients: number;
2700
- numberOfIssues: number;
2701
- numberOfPeerConnections: number;
2702
- numberOfInboundRtpStreams: number;
2703
- numberOfOutboundRtpStreams: number;
2704
- numberOfDataChannels: number;
2705
- maxNumberOfClients: number;
2706
- deltaNumberOfIssues: number;
2707
- appData: AppData;
2708
- closed: boolean;
2709
- startedAt?: number;
2710
- endedAt?: number;
2711
- closedAt?: number;
2712
- readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs'>;
2713
- /** Ancestry base shared by all Observer-bus events originating at this call. */
2714
- readonly eventScope: ObservedCallScope;
2715
- private closeTimer?;
2716
- constructor(settings: ObservedCallSettings<AppData>, observer: Observer);
2717
- get numberOfClients(): number;
2718
- get score(): number | undefined;
2719
- /** Raise a call-level (server-side) issue; surfaced on the Observer bus as `call-issue`. */
2720
- addIssue(issue: ClientIssue): void;
3906
+ /**
3907
+ * Answers **"is this meeting in trouble?"** — several participants of one call with the same issue
3908
+ * open simultaneously.
3909
+ *
3910
+ * The client already decides *what* is wrong for itself — `congestion`, `ice-disconnected`,
3911
+ * `audio-concealment`, `video-decoder-overloaded` — with hysteresis and multi-signal confirmation
3912
+ * behind each verdict. Re-deriving those server-side from raw counters would be strictly worse. What
3913
+ * the server uniquely knows is *how many other participants of the same call are in that state right
3914
+ * now*, which is the difference between "one person's Wi-Fi" and "this room is broken".
3915
+ *
3916
+ * Concurrency is judged from the **open interval set**, not a window of recent reports. A window has
3917
+ * to guess whether a symptom is still happening; an interval is closed by the client when the episode
3918
+ * actually ends (client-monitor-js >= 4.6.0 ships the `<type>-resolved` companion for exactly this).
3919
+ *
3920
+ * ```ts
3921
+ * observedCall.addDetector('call-concurrent-issue-detector', {
3922
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
3923
+ * });
3924
+ * ```
3925
+ *
3926
+ * For the cross-call version of this question — which is a different question, not this one with a
3927
+ * bigger denominator — see `ObserverConcurrentIssueDetector`.
3928
+ */
3929
+ declare class CallConcurrentIssueDetector implements Detector, ActiveIssueTracker {
3930
+ private readonly _call;
3931
+ static readonly NAME: "call-concurrent-issue-detector";
3932
+ readonly name: "call-concurrent-issue-detector";
3933
+ private readonly _config;
3934
+ private readonly _lastRaisedAt;
3935
+ /** issue type -> the issues of that type currently open in this call. */
3936
+ private readonly _byType;
3937
+ private _size;
3938
+ /** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
3939
+ lastGroups: CallConcurrentIssueGroup[];
3940
+ constructor(_call: ObservedCall, config?: Partial<CallConcurrentIssueDetectorConfig>);
3941
+ get size(): number;
3942
+ has(issue: ActiveClientIssue): boolean;
3943
+ add(issue: ActiveClientIssue): void;
3944
+ delete(issue: ActiveClientIssue): boolean;
3945
+ clear(): void;
3946
+ update(): void;
2721
3947
  close(): void;
2722
- getObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(clientId: string): ObservedClient<ClientAppData> | undefined;
2723
- createObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>): ObservedClient<ClientAppData> | undefined;
2724
- getOrCreateObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>): ObservedClient<ClientAppData> | undefined;
2725
- update(context?: AcceptContext): void;
2726
- private _onClientUpdate;
2727
- private _clientJoined;
2728
- private _clientLeft;
2729
- /** Emit an Observer-bus event scoped to this call. */
2730
- private _notify;
2731
- }
2732
-
2733
- type Middleware<T> = (input: T, next: (nextInput: T) => void) => void;
2734
- interface Processor<T> {
2735
- finalCallback?: Callback<T>;
2736
- process(value: T): void;
2737
- addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
2738
- removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
2739
- }
2740
- type Callback<T> = (input: T) => void;
2741
- declare class MiddlewareProcessor<T> implements Processor<T> {
2742
- private stack;
2743
- finalCallback?: Callback<T>;
2744
- addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
2745
- removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
2746
- process(value: T): void;
3948
+ private _groupOf;
2747
3949
  }
2748
3950
 
2749
- type SampleRejectedReason = 'observer-closed' | 'missing-callId' | 'missing-clientId';
2750
-
3951
+ /** A point on the earth, as an application reports it for one client. */
3952
+ type ClientLocation = {
3953
+ latitude: number;
3954
+ longitude: number;
3955
+ };
2751
3956
  /**
2752
- * Optional, free-form context supplied to `accept()`. A single context object is
2753
- * threaded down the accept chain (Observer -> Client -> PeerConnection) and is
2754
- * merged into the `appData` of entities created during the accept pass.
3957
+ * Approximate cell width at each geohash length, for choosing a precision.
3958
+ *
3959
+ * Index is the character count; the value is the rough cell size at the equator. Cells are taller
3960
+ * than they are wide at high latitudes, so treat these as an order of magnitude, not a radius.
2755
3961
  */
2756
- type AcceptContext = Record<string, unknown>;
2757
- /** The payload threaded through `accept()` middlewares: the sample and its optional context. */
2758
- type AcceptMiddlewarePayload = {
2759
- sample: ClientSample;
2760
- context?: AcceptContext;
2761
- };
3962
+ declare const GEOHASH_CELL_SIZES: readonly ["", "~5000 km", "~1250 km", "~156 km", "~39 km", "~5 km", "~1.2 km", "~150 m"];
2762
3963
  /**
2763
- * A global middleware run on every sample passed to `observer.accept()`, in registration order,
2764
- * **before** the sample is dispatched to any call/client. It may inspect or mutate the sample
2765
- * (e.g. set/normalize `callId`/`clientId`, enrich, redact) or the context, then call
2766
- * `next(payload)` to continue the chain. Not calling `next` **drops** the sample.
3964
+ * Encode a point as a geohash of `precision` characters — a **grid cell key**, not a cluster.
3965
+ *
3966
+ * ### Why cells rather than "within N kilometres"
3967
+ *
3968
+ * Grouping clients "within a radius" sounds like the natural thing and is a much worse fit. It is a
3969
+ * clustering problem, not a keying one: the groups depend on which client you start from, two
3970
+ * clients can each be within the radius of a third but not of each other, group identity is not
3971
+ * stable as participants join and leave, and maintaining it costs pairwise distance work. None of
3972
+ * that survives contact with a detector that has to produce the *same* group name on every tick so
3973
+ * a cooldown and a control group mean anything.
3974
+ *
3975
+ * A geohash prefix is a plain function of the coordinates: O(1), stable for the life of the client,
3976
+ * and usable directly as a population label. The honest cost is that a cell boundary can separate
3977
+ * two clients who are physically adjacent, which splits a real group into two smaller ones. That
3978
+ * biases towards **missing** a finding rather than inventing one, which is the right direction for
3979
+ * something that raises issues.
3980
+ *
3981
+ * Returns `undefined` for coordinates that are not finite or not on the earth, rather than encoding
3982
+ * nonsense into a plausible-looking cell key.
2767
3983
  */
2768
- type AcceptMiddleware = Middleware<AcceptMiddlewarePayload>;
2769
- type ObserverUpdateConfig = {
2770
- updatePolicy?: 'update-on-any-call-updated' | 'update-when-all-call-updated' | 'none';
3984
+ declare function geohash(location: ClientLocation, precision?: number): string | undefined;
3985
+
3986
+ declare const ClientPopulationIssueTypes: {
3987
+ /** One issue type is concentrated on one client population while the rest of the fleet is fine. */
3988
+ readonly clientPopulationIssue: "CLIENT_POPULATION_ISSUE";
2771
3989
  };
2772
- /** Produces the initial `appData` for a call created without an explicit `appData`. */
2773
- type CallAppDataFactory = (params: {
2774
- callId: string;
2775
- observer: Observer;
2776
- }) => Record<string, unknown>;
2777
- /** Produces the initial `appData` for a client created without an explicit `appData`. */
2778
- type ClientAppDataFactory = (params: {
2779
- clientId: string;
2780
- observedCall: ObservedCall;
2781
- }) => Record<string, unknown>;
2782
- type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = ObserverUpdateConfig & {
2783
- defaultCallUpdatePolicy?: ObservedCallSettings['updatePolicy'];
2784
- appData?: AppData;
2785
- closeClientIfIdleForMs?: number;
2786
- closeCallIfEmptyForMs?: number;
3990
+ /** The client attribute to group by. One axis per detector — see the class description. */
3991
+ type ClientPopulationAxis = 'browser' | 'engine' | 'platform' | 'operationSystem' | 'location';
3992
+ /**
3993
+ * Reads a client's coordinates, for `groupBy: 'location'`. **Required for that axis.**
3994
+ *
3995
+ * There is no coordinate field in `ClientSample`, so the shape is yours: read it off
3996
+ * `client.attachments`, off a custom meta item, or from an `appData` field your accept middleware
3997
+ * filled in. Return `undefined` for clients whose location you do not know — they are then excluded
3998
+ * from both the population and the control group, exactly like a client that never reported its
3999
+ * browser.
4000
+ */
4001
+ type ClientLocationResolver = (client: ObservedClient) => ClientLocation | undefined;
4002
+ type ClientPopulationIssueDetectorConfig = {
4003
+ /**
4004
+ * The issue types to watch. **Required, and must not be empty.**
4005
+ *
4006
+ * **Match the issue family to the axis.** On the endpoint axes (`browser`, `engine`, `platform`,
4007
+ * `operationSystem`) the types worth grouping are the ones an endpoint owns: `cpulimitation`,
4008
+ * `encoder-bottleneck`, `capture-bottleneck`, `stuck-decoder`, `video-decoder-overloaded`.
4009
+ * Grouping a *network* symptom by browser is a category error — `congestion` clusters by ISP and
4010
+ * geography, not by build, and the detector would happily report a browser correlation that is
4011
+ * really a "most of our users are on Chrome" artefact.
4012
+ *
4013
+ * On the `location` axis it is the other way round: group the network symptoms — `congestion`,
4014
+ * `ice-disconnected`, `unstable-ice-path` — and not the endpoint ones, since there is no reason a
4015
+ * decoder should stall by geography.
4016
+ */
4017
+ issueTypes: string[];
4018
+ /** Which client attribute to group by. Default `'browser'`. */
4019
+ groupBy: ClientPopulationAxis;
2787
4020
  /**
2788
- * Optional factory invoked when a call is created without an explicit `appData`
2789
- * (e.g. lazily by `accept()`), so apps can enrich appData without pre-creating the
2790
- * entity. `appData` is application-owned; it is never modified by the `accept()` context.
4021
+ * Where to read a client's coordinates. **Required when `groupBy` is `'location'`**, ignored
4022
+ * otherwise. See {@link ClientLocationResolver}.
2791
4023
  */
2792
- createCallAppData?: CallAppDataFactory;
2793
- /** Same as `createCallAppData`, for clients. Receives the (already-created) parent call. */
2794
- createClientAppData?: ClientAppDataFactory;
4024
+ resolveClientLocation?: ClientLocationResolver;
2795
4025
  /**
2796
- * Optional factory invoked when a client is created, producing a per-client sink that
2797
- * receives every sample the client accepts (or `undefined` for no sink). The destination
2798
- * can be derived from `callId` / `clientId`.
4026
+ * Geohash characters to group locations by, i.e. how coarse a "place" is. Default `3` (~156 km).
4027
+ *
4028
+ * `2` ~1250 km, `3` ~156 km, `4` ~39 km, `5` ~5 km. Coarser cells hold more clients, which is what
4029
+ * makes a rate mean anything, so start coarse: a city-sized cell rarely has `minPopulationSize`
4030
+ * participants in it. See `utils/geohash` for why this is a grid cell and not a radius.
2799
4031
  */
2800
- createClientSink?: ClientSampleSinkFactory;
4032
+ locationPrecision: number;
2801
4033
  /**
2802
- * Optional factory invoked when a call is created, producing the call's `RemoteTrackResolver`
2803
- * (or `undefined` for none). Use the built-ins
2804
- * (`createDefaultMediasoupRemoteTrackResolverFactory()` / `createP2pRemoteTrackResolverFactory()`)
2805
- * or build a `RemoteTrackResolver` with custom publisher/subscriber id resolvers.
4034
+ * Group by `name` only, or by `name + version`. Default `true` (include version).
4035
+ *
4036
+ * Version is usually the point: "Chrome" is not actionable, "Chrome 141" is, because it names a
4037
+ * thing that changed on a date. Set to `false` when comparing whole engines.
4038
+ */
4039
+ includeVersion: boolean;
4040
+ /**
4041
+ * Clients in a population before its rate means anything. Default `20`.
4042
+ *
4043
+ * Higher than the other detectors' minimums on purpose: this one compares *rates*, and a rate over
4044
+ * five clients is not a rate. Sensible range `20`–`100`. On the `location` axis this is the field
4045
+ * most likely to silence the detector — a city-sized cell rarely holds twenty concurrent
4046
+ * participants, so reach for a coarser `locationPrecision` before lowering this.
4047
+ */
4048
+ minPopulationSize: number;
4049
+ /**
4050
+ * Affected clients required within the population. Default `5`.
4051
+ *
4052
+ * Checked independently of `affectedRatioThreshold`, so one unlucky user on a rare browser cannot
4053
+ * page anyone however striking the ratio looks. Sensible range `5`–`20`.
4054
+ */
4055
+ minAffectedClients: number;
4056
+ /**
4057
+ * Share of the population that must be affected, `0`–`1`. Default `0.3`.
4058
+ *
4059
+ * Lower than the per-call thresholds deliberately: an issue hitting 30% of one browser version while
4060
+ * the rest of the fleet is clean is already a strong signal, and endpoint faults rarely affect
4061
+ * *everyone* on a build. Typical `0.2`–`0.5`. This is the weakest of the gates —
4062
+ * `minRelativeRisk` is what makes the finding mean anything.
4063
+ */
4064
+ affectedRatioThreshold: number;
4065
+ /**
4066
+ * How many times worse the suspect population must be than the rest of the fleet. Default `3`.
4067
+ *
4068
+ * **This is the gate that makes the finding mean anything** — see the class description. Typical
4069
+ * `2`–`10`. At `2` you will see populations that are merely somewhat worse, which is often just a
4070
+ * different usage pattern; at `10` only stark, unambiguous concentrations survive. A spotless
4071
+ * control group yields `Infinity`, which clears any threshold, so the minimum-count gates above are
4072
+ * what stop that from being trivial.
4073
+ */
4074
+ minRelativeRisk: number;
4075
+ /**
4076
+ * Clients **outside** the suspect population before a comparison is possible. Default `20`.
4077
+ *
4078
+ * "Worse than everyone else" needs an everyone else. Sensible range `20`–`100`. Note the practical
4079
+ * consequence: a fleet that is overwhelmingly one browser can never have that browser reported,
4080
+ * because there is no control group left — which is honest, since at 95% Chrome you cannot separate a
4081
+ * Chrome fault from a fleet-wide one.
2806
4082
  */
2807
- createTrackResolver?: RemoteTrackResolverFactory;
4083
+ minControlSize: number;
4084
+ /**
4085
+ * Re-arm time per (population, issue type) in ms. Default `300_000`.
4086
+ *
4087
+ * Long by design: a bad client build is a condition lasting days, not an event, and the action it
4088
+ * prompts — ship a fix, roll back a version — is not one you take twice an hour. Typical
4089
+ * `300_000`–`3_600_000`.
4090
+ */
4091
+ cooldownMs: number;
2808
4092
  };
2809
- declare interface Observer {
2810
- on<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
2811
- off<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
2812
- once<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
2813
- emit<U extends keyof ObserverEvents>(event: U, ...args: ObserverEvents[U]): boolean;
4093
+ /** The rollup for one population on one issue type. */
4094
+ type ClientPopulation = {
4095
+ /**
4096
+ * e.g. `'Chrome 141'`, or `'Chrome'` when `includeVersion` is off. For the `'location'` axis this
4097
+ * is the geohash cell — never the coordinates themselves, so an archived payload carries a place
4098
+ * at the configured resolution and not a person's position.
4099
+ */
4100
+ population: string;
4101
+ axis: ClientPopulationAxis;
4102
+ issueType: string;
4103
+ clients: number;
4104
+ affectedClients: number;
4105
+ affectedRatio: number;
4106
+ affectedClientIds: string[];
4107
+ /** Everyone not in this population. */
4108
+ controlClients: number;
4109
+ controlAffectedClients: number;
4110
+ controlAffectedRatio: number;
4111
+ /** `affectedRatio / controlAffectedRatio`. `Infinity` when the control group is completely clean. */
4112
+ relativeRisk: number;
4113
+ };
4114
+ /**
4115
+ * Finds an issue that is concentrated on **one kind of client** — one browser, one browser version,
4116
+ * one OS — rather than on anything the servers own.
4117
+ *
4118
+ * ### Why this exists
4119
+ *
4120
+ * The other observer-scoped detectors all answer "who else has this open, and what do they share?"
4121
+ * with the answer *the infrastructure*, because clients in unrelated calls share nothing else. That
4122
+ * inference is right for network symptoms and **wrong for endpoint symptoms**, and the difference
4123
+ * matters at 3am. `cpulimitation` opening across six unrelated calls is not an SFU event: CPU is
4124
+ * owned by the endpoint, so what those endpoints have in common is a client release, a browser
4125
+ * update, or a fleet of identical VDI hosts. `IssueConclusion` already says exactly this — it maps
4126
+ * the endpoint-capacity family to a `client-population` fault domain instead of `infrastructure` —
4127
+ * but until now nothing in the library actually computed the grouping that claim refers to. This
4128
+ * detector is that computation.
4129
+ *
4130
+ * It is the one correlation in this library that is neither per-call nor per-server. A client knows
4131
+ * its own browser and nothing about anyone else's; only something sitting above the whole fleet can
4132
+ * notice that every complaint is coming from the same build.
4133
+ *
4134
+ * ### The control group is the whole point
4135
+ *
4136
+ * "30% of Chrome 141 users report encoder-bottleneck" is not a finding on its own. If 30% of
4137
+ * *everyone* reports it, Chrome 141 is not the story — you have a fleet-wide problem and this
4138
+ * detector would be pointing at the largest population rather than at a cause. Naive share-based
4139
+ * grouping always indicts whichever browser is most popular, which is why the gate here is
4140
+ * **relative risk**: the suspect population's rate divided by the rate among everyone else. A
4141
+ * population only qualifies when it is `minRelativeRisk` times worse than the rest of the fleet, and
4142
+ * only when the rest of the fleet is large enough (`minControlSize`) for "the rest of the fleet" to
4143
+ * be a real measurement.
4144
+ *
4145
+ * A completely clean control group gives `Infinity`, which is honest — nobody outside this
4146
+ * population has the problem at all — and is exactly why `minAffectedClients` and
4147
+ * `minPopulationSize` are checked independently, so a single unlucky user on a rare browser cannot
4148
+ * page anyone.
4149
+ *
4150
+ * ### One axis per detector
4151
+ *
4152
+ * `groupBy` takes a single attribute. Add a second instance if you want a second axis:
4153
+ *
4154
+ * ```ts
4155
+ * observer.addObserverDetector('client-population-issue-detector', {
4156
+ * issueTypes: [ 'cpulimitation', 'encoder-bottleneck', 'stuck-decoder' ],
4157
+ * groupBy: 'browser',
4158
+ * });
4159
+ *
4160
+ * observer.on('observer-issue', ({ issue }) => {
4161
+ * if (issue.type !== ClientPopulationIssueTypes.clientPopulationIssue) return;
4162
+ * // → { population: 'Chrome 141', issueType: 'encoder-bottleneck',
4163
+ * // affectedRatio: 0.34, controlAffectedRatio: 0.02, relativeRisk: 17 }
4164
+ * });
4165
+ * ```
4166
+ *
4167
+ * Deliberately not a cross-product of every axis at once: an issue that clusters on macOS *and* on
4168
+ * Safari is usually one fact reported twice, and a detector that emits both leaves the reader to
4169
+ * work out which one is causal. Pick the axis you want to reason about.
4170
+ *
4171
+ * ### The `location` axis
4172
+ *
4173
+ * With `groupBy: 'location'` the population is a **geohash cell** rather than a client attribute, so
4174
+ * the same machinery answers a different question: *is this symptom concentrated in one place?* That
4175
+ * is the grouping the note above says browsers cannot give you, and it is the one that matters for
4176
+ * network symptoms.
4177
+ *
4178
+ * The observer does not derive "this client's RTT jumped" — `client-monitor`'s `CongestionDetector`
4179
+ * already owns that verdict, comparing each peer connection's RTT against its own EWMA baseline and
4180
+ * requiring a bandwidth-limitation corroboration before it raises `congestion`. Absolute RTT is not
4181
+ * comparable between clients anyway: someone 200 ms away is *always* 200 ms away, so the only signal
4182
+ * is deviation from that client's own baseline, which is exactly what the client already measures.
4183
+ * This detector's contribution is the part no endpoint can see — that many of the affected clients
4184
+ * are in the same place at the same time.
4185
+ *
4186
+ * ```ts
4187
+ * observer.addObserverDetector('client-population-issue-detector', {
4188
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
4189
+ * groupBy: 'location',
4190
+ * locationPrecision: 3, // ~156 km cells
4191
+ * resolveClientLocation: (client) => client.attachments?.geo as { latitude: number, longitude: number },
4192
+ * });
4193
+ * ```
4194
+ *
4195
+ * Coordinates are not in `ClientSample`, so `resolveClientLocation` is required — see
4196
+ * {@link ClientLocationResolver}. Only the cell key reaches the issue payload, never the
4197
+ * coordinates, which matters because these payloads are archived into call summaries.
4198
+ *
4199
+ * **The limitation to state plainly: geography is confounded with your topology.** The control group
4200
+ * is "everyone outside this cell", which cannot separate *"the path into this region degraded"* from
4201
+ * *"the SFU that happens to serve this region degraded"*. If a region maps largely onto one
4202
+ * deployment, both hypotheses fit the same evidence. The discriminator is whether clients elsewhere
4203
+ * on the same server also degraded, which is what `SfuCongestionDetector` and
4204
+ * `TurnServerHealthDetector` answer — so the conclusion here points at them rather than claiming an
4205
+ * attribution it cannot support.
4206
+ *
4207
+ * ### Clients that never reported their metadata
4208
+ *
4209
+ * `browser` / `engine` / `platform` / `operationSystem` arrive as client metadata and may be absent —
4210
+ * a client that closed before sending them, or an application that does not collect them. Those
4211
+ * clients are excluded from **both** the population and the control group rather than bucketed as
4212
+ * `'unknown'`. The same applies to a client whose location `resolveClientLocation` cannot supply. A synthetic `'unknown'` population would be a mixture of every real one, so any rate
4213
+ * computed for it means nothing, and leaving those clients in the control group would dilute the
4214
+ * comparison with clients whose kind we cannot verify.
4215
+ */
4216
+ declare class ClientPopulationIssueDetector implements Detector, ActiveIssueTracker {
4217
+ private readonly _observer;
4218
+ static readonly NAME: "client-population-issue-detector";
4219
+ readonly name: "client-population-issue-detector";
4220
+ private readonly _config;
4221
+ private readonly _lastRaisedAt;
4222
+ private readonly _issues;
4223
+ /** The populations that qualified on the most recent `update()`. Exposed for tests/dashboards. */
4224
+ lastPopulations: ClientPopulation[];
4225
+ constructor(_observer: Observer, config?: Partial<ClientPopulationIssueDetectorConfig>);
4226
+ get size(): number;
4227
+ has(issue: ActiveClientIssue): boolean;
4228
+ add(issue: ActiveClientIssue): void;
4229
+ delete(issue: ActiveClientIssue): boolean;
4230
+ clear(): void;
4231
+ update(): void;
4232
+ close(): void;
4233
+ /** `undefined` when the client never reported this attribute — see the class description. */
4234
+ private _populationOf;
4235
+ private _rollupOf;
4236
+ private _riskText;
4237
+ /** Bigger populations and starker contrasts are harder to produce by chance. */
4238
+ private _confidenceOf;
2814
4239
  }
2815
- declare class Observer<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
2816
- readonly config: ObserverConfig<AppData>;
4240
+
4241
+ declare const PublisherFaultTypes: {
4242
+ /**
4243
+ * A publisher is reporting trouble on its own send path **while** its subscribers report trouble
4244
+ * receiving it. Both ends agree, so the source is implicated rather than inferred.
4245
+ */
4246
+ readonly corroboratedPublisherFault: "CORROBORATED_PUBLISHER_FAULT";
4247
+ };
4248
+ type PublisherFaultCorroborationDetectorConfig = {
4249
+ /**
4250
+ * Issue types raised by the **publishing** client about its own outbound path. **Required.**
4251
+ *
4252
+ * The natural set from client-monitor-js: `encoder-bottleneck`, `capture-bottleneck`,
4253
+ * `dry-outbound-track`. All three mean "I am failing to produce or send this properly", which is
4254
+ * the half of the story the receivers cannot see.
4255
+ */
4256
+ publisherIssueTypes: string[];
4257
+ /**
4258
+ * Issue types raised by the **subscribing** clients about the track they receive. **Required.**
4259
+ *
4260
+ * The natural set: `freezed-video-track`, `dry-inbound-track`, `video-recovery-failed`. These say
4261
+ * "I am not getting this properly", which is the half the publisher cannot see.
4262
+ */
4263
+ receiverIssueTypes: string[];
4264
+ /**
4265
+ * Subscribers of the track that must be complaining at the same time. Default `2`.
4266
+ *
4267
+ * `1` still yields a genuine two-sided corroboration — publisher and one receiver agreeing is
4268
+ * already more than either says alone — but `2` rules out the case where a single receiver's own
4269
+ * downlink is at fault and merely coincides with the publisher's complaint. Sensible range `1`–`3`;
4270
+ * higher mostly costs you findings in small calls, where a track may only have two subscribers.
4271
+ */
4272
+ minAffectedReceivers: number;
4273
+ /**
4274
+ * Re-arm time per published track (ms). Default `60_000`.
4275
+ *
4276
+ * Typical `30_000`–`300_000`. This detector raises the highest-confidence finding in the library,
4277
+ * so it is the one you least want repeating every tick.
4278
+ */
4279
+ cooldownMs: number;
4280
+ };
4281
+ /** The two-sided evidence behind one finding. */
4282
+ type CorroboratedPublisherFault = {
4283
+ trackId: string;
4284
+ kind: string;
4285
+ publisherClientId: string;
4286
+ /** The publisher's own open issue types on this track. */
4287
+ publisherIssueTypes: string[];
4288
+ /** The receiver-side open issue types across this track's subscribers. */
4289
+ receiverIssueTypes: string[];
4290
+ receivers: number;
4291
+ affectedReceivers: number;
4292
+ affectedClientIds: string[];
4293
+ publisherBitrate?: number;
4294
+ };
4295
+ /**
4296
+ * Fires only when **both ends of one published track are complaining at the same time**: the
4297
+ * publisher about its own send path, and its subscribers about receiving it.
4298
+ *
4299
+ * ### How this differs from `IssueFanOutDetector`
4300
+ *
4301
+ * Fan-out sees one end. It observes that most of Alice's subscribers are unhappy and *infers* that
4302
+ * the fault is on Alice's side, because the affected clients share a publisher and nothing else. That
4303
+ * inference is sound, and it is still a inference: the same observation is produced by the SFU
4304
+ * mangling Alice's stream on the way out, with Alice herself perfectly healthy.
4305
+ *
4306
+ * This detector removes the inference. When Alice reports `encoder-bottleneck` *and* four of her six
4307
+ * subscribers report `freezed-video-track` in the same window, there is nothing left to deduce — the
4308
+ * source said it was struggling and the receivers confirmed the consequence. That is the strongest
4309
+ * statement this library can make about where a fault sits, and it is only available to something
4310
+ * holding both ends at once. Neither the publisher nor any receiver can reach this conclusion alone.
4311
+ *
4312
+ * Run both: fan-out is broader and catches the SFU-forwarding case where the publisher is fine;
4313
+ * this one is narrower and, when it fires, needs no interpretation.
4314
+ *
4315
+ * ### Silence here is not health
4316
+ *
4317
+ * A quiet detector means only that the two halves have not coincided — most commonly because the
4318
+ * publisher is genuinely fine and the fault is in forwarding, which is exactly the case `fan-out`
4319
+ * exists to report. Do not read "no corroborated fault" as "no publisher-side problem".
4320
+ *
4321
+ * ```ts
4322
+ * observedCall.addDetector('publisher-fault-corroboration-detector', {
4323
+ * publisherIssueTypes: [ 'encoder-bottleneck', 'capture-bottleneck', 'dry-outbound-track' ],
4324
+ * receiverIssueTypes: [ 'freezed-video-track', 'dry-inbound-track' ],
4325
+ * });
4326
+ * ```
4327
+ *
4328
+ * ### Requires a `RemoteTrackResolver`
4329
+ *
4330
+ * Matching a publisher's issue to its subscribers' issues needs the publisher↔subscriber links. With
4331
+ * no resolver the detector does nothing rather than guessing.
4332
+ */
4333
+ declare class PublisherFaultCorroborationDetector implements Detector, ActiveIssueTracker {
4334
+ private readonly _call;
4335
+ static readonly NAME: "publisher-fault-corroboration-detector";
4336
+ readonly name: "publisher-fault-corroboration-detector";
4337
+ private readonly _config;
4338
+ private readonly _lastRaisedAt;
4339
+ /** Open publisher-side issues that name a track. */
4340
+ private readonly _publisherIssues;
4341
+ /** Open receiver-side issues that name a track. */
4342
+ private readonly _receiverIssues;
4343
+ /** The faults corroborated on the most recent `update()`. Exposed for tests/dashboards. */
4344
+ lastFaults: CorroboratedPublisherFault[];
4345
+ constructor(_call: ObservedCall, config?: Partial<PublisherFaultCorroborationDetectorConfig>);
4346
+ get size(): number;
4347
+ has(issue: ActiveClientIssue): boolean;
4348
+ add(issue: ActiveClientIssue): void;
4349
+ delete(issue: ActiveClientIssue): boolean;
4350
+ clear(): void;
4351
+ update(): void;
4352
+ close(): void;
4353
+ /**
4354
+ * Resolve a publisher-side issue's `trackId` to the outbound track it is about.
4355
+ *
4356
+ * Looked up through the reporting client's own peer connections: the issue names its client, so
4357
+ * the search is bounded by that client's transports rather than by the size of the call.
4358
+ */
4359
+ private _outboundTrackOf;
4360
+ }
4361
+
4362
+ declare const ObserverConcurrentIssueTypes: {
4363
+ /**
4364
+ * The same issue is open in **several unrelated calls at once**. Those clients share no meeting,
4365
+ * no publisher and no room — only the infrastructure serving them.
4366
+ */
4367
+ readonly crossCallConcurrentIssues: "CROSS_CALL_CONCURRENT_ISSUES";
4368
+ /** The cross-call version that also started together. The strongest "it's us" signal available. */
4369
+ readonly crossCallIssueOnsetBurst: "CROSS_CALL_ISSUE_ONSET_BURST";
4370
+ };
4371
+ type ObserverConcurrentIssueDetectorConfig = {
4372
+ /**
4373
+ * The issue types to watch. **Required, and must not be empty** — the detector subscribes to
4374
+ * exactly these and sees nothing else.
4375
+ */
4376
+ issueTypes: string[];
4377
+ /**
4378
+ * Distinct clients that must share the open issue. Default `3`.
4379
+ *
4380
+ * Absolute, deliberately — see `affectedCallRatioThreshold` for why no client *ratio* exists at
4381
+ * this scope. Sensible range `3`–`10`; scale it with fleet size, since three clients is a real
4382
+ * signal across five calls and background noise across five hundred.
4383
+ */
4384
+ minAffectedClients: number;
4385
+ /**
4386
+ * Minimum number of *distinct calls* the affected clients must span. Default `2`.
4387
+ *
4388
+ * This is what makes an observer-scoped finding mean something a call-scoped one doesn't. Without
4389
+ * it, one thirty-person meeting with congestion satisfies every client-count threshold and raises
4390
+ * a fleet-wide alert for what is really one bad room — which `CallConcurrentIssueDetector` has
4391
+ * already reported. Requiring two or more independent calls is the difference between a
4392
+ * coincidence and a shared cause.
4393
+ */
4394
+ minAffectedCalls: number;
4395
+ /**
4396
+ * Fraction of the calls in flight that must be affected. Default `0` — off, because absolute
4397
+ * counts matter more than ratios here: three broken calls out of a thousand is still worth
4398
+ * knowing about, and a ratio threshold would hide it. Raise it if you only care about fleet-wide
4399
+ * events.
4400
+ *
4401
+ * Note there is deliberately **no participant ratio** at this scope. Six broken calls out of forty
4402
+ * is a handful of clients against the whole fleet, so any meaningful client ratio would suppress
4403
+ * exactly the finding this detector exists to produce.
4404
+ */
4405
+ affectedCallRatioThreshold: number;
4406
+ /**
4407
+ * Onsets falling within this span (ms) escalate the finding to `CROSS_CALL_ISSUE_ONSET_BURST`.
4408
+ * Default `2_000`.
4409
+ *
4410
+ * This is the strongest evidence the detector produces: independent calls starting to fail *at the
4411
+ * same instant* has no explanation other than something they share. Keep it at or above your
4412
+ * sampling period — onsets are only as precise as clients report them — and no wider than a few
4413
+ * seconds, or unrelated failures drift into the same window. Typical `1_000`–`5_000`.
4414
+ */
4415
+ onsetBurstWindowInMs: number;
4416
+ /**
4417
+ * Re-arm time per issue type (ms). Default `60_000`.
4418
+ *
4419
+ * Typical `60_000`–`600_000`. A fleet-wide event is one incident: without a cooldown a sustained
4420
+ * outage would raise an issue on every tick for its whole duration.
4421
+ */
4422
+ cooldownMs: number;
4423
+ };
4424
+ /** What the detector currently knows about one issue type across the fleet. */
4425
+ type ObserverConcurrentIssueGroup = {
4426
+ type: string;
4427
+ issues: ActiveClientIssue[];
4428
+ clientIds: string[];
4429
+ totalClients: number;
4430
+ affectedRatio: number;
4431
+ callIds: string[];
4432
+ totalCalls: number;
4433
+ affectedCallRatio: number;
4434
+ /** Per-call breakdown, largest first — the first question anyone asks is "which calls, how badly?". */
4435
+ perCall: {
4436
+ callId: string;
4437
+ affectedClients: number;
4438
+ totalClients: number;
4439
+ }[];
4440
+ /**
4441
+ * Spread of the onsets, in **observer** time (ms). Client clocks are never compared across
4442
+ * machines here — skew between them would masquerade as a synchronized event.
4443
+ */
4444
+ onsetSpreadInMs: number;
4445
+ firstObservedAt: number;
4446
+ };
4447
+ /**
4448
+ * Answers **"is our infrastructure in trouble?"** — the same issue open across several *unrelated*
4449
+ * calls at the same moment.
4450
+ *
4451
+ * This is not the call-scoped question with a bigger denominator, which is why it is a separate
4452
+ * detector with separate gates and its own finding types. Participant count alone is a bad fleet
4453
+ * signal: one thirty-person meeting where everybody is congested clears every client threshold, yet
4454
+ * it has an obvious local explanation. Clients in *different* calls share no room, no publisher and
4455
+ * no host — only the servers and the network. When the same issue opens across several of them at
4456
+ * once, the infrastructure is the only remaining common factor, and that is the finding worth paging
4457
+ * someone about.
4458
+ *
4459
+ * ```ts
4460
+ * observer.addObserverDetector('observer-concurrent-issue-detector', {
4461
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
4462
+ * minAffectedCalls: 3,
4463
+ * });
4464
+ *
4465
+ * observer.on('observer-issue', ({ issue }) => {
4466
+ * if (issue.type === ObserverConcurrentIssueTypes.crossCallIssueOnsetBurst) page(issue);
4467
+ * });
4468
+ * // → CROSS_CALL_ISSUE_ONSET_BURST { issueType: 'congestion', calls: 40, affectedCalls: 6, … }
4469
+ * ```
4470
+ *
4471
+ * Onsets are compared on the **observer clock** (`observedAt`), never on client clocks: participants
4472
+ * degrading together within a couple of seconds is far more likely to be a deploy, a TURN failover or
4473
+ * a link flap than a coincidence — but only if the timestamps being compared came from one clock.
4474
+ */
4475
+ declare class ObserverConcurrentIssueDetector implements Detector, ActiveIssueTracker {
4476
+ private readonly _observer;
4477
+ static readonly NAME: "observer-concurrent-issue-detector";
4478
+ readonly name: "observer-concurrent-issue-detector";
4479
+ private readonly _config;
4480
+ private readonly _lastRaisedAt;
4481
+ /** issue type -> the issues of that type currently open anywhere in the fleet. */
4482
+ private readonly _byType;
4483
+ private _size;
4484
+ /** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
4485
+ lastGroups: ObserverConcurrentIssueGroup[];
4486
+ constructor(_observer: Observer, config?: Partial<ObserverConcurrentIssueDetectorConfig>);
4487
+ get size(): number;
4488
+ has(issue: ActiveClientIssue): boolean;
4489
+ add(issue: ActiveClientIssue): void;
4490
+ delete(issue: ActiveClientIssue): boolean;
4491
+ clear(): void;
4492
+ update(): void;
4493
+ close(): void;
4494
+ private _groupOf;
4495
+ }
4496
+
4497
+ declare const IssueFanOutTypes: {
4498
+ /** Most receivers of one published track have the same issue open → the fault follows the source. */
4499
+ readonly publishedTrackIssueFanOut: "PUBLISHED_TRACK_ISSUE_FAN_OUT";
4500
+ /** Exactly one receiver of a track has it → that receiver's own problem. */
4501
+ readonly singleReceiverIssue: "SINGLE_RECEIVER_ISSUE";
4502
+ };
4503
+ type IssueFanOutDetectorConfig = {
4504
+ /**
4505
+ * The receiver-side issue types to attribute to publishers. **Required, and must not be empty** —
4506
+ * the detector subscribes to exactly these.
4507
+ *
4508
+ * There is no "all types" option. Which of a receiver's complaints are worth blaming a publisher
4509
+ * for is application knowledge: `freezed-video-track` fanning out across a track's subscribers
4510
+ * implicates the source, `cpulimitation` fanning out the same way implicates the receivers'
4511
+ * hardware and would be a false accusation.
4512
+ */
4513
+ issueTypes: string[];
4514
+ /**
4515
+ * Receivers a track needs before a ratio means anything. Default `3`.
4516
+ *
4517
+ * With two receivers, "60% affected" is one of them — which is the single-receiver case below, not
4518
+ * a fan-out. Sensible range `2`–`5`; in small calls a published track rarely has more than a couple
4519
+ * of subscribers, so raising this can silence the detector entirely.
4520
+ */
4521
+ minReceivers: number;
4522
+ /**
4523
+ * Fraction of a track's receivers that must share the issue, `0`–`1`. Default `0.6`.
4524
+ *
4525
+ * The higher this is, the more the finding points at the publisher rather than at the network
4526
+ * between: *everyone* receiving this track badly is hard to explain any other way. Typical
4527
+ * `0.5`–`0.8`. Below `0.5` you are reporting "some receivers", which usually means their own
4528
+ * last miles.
4529
+ */
4530
+ affectedRatioThreshold: number;
4531
+ /**
4532
+ * Also report when exactly one receiver is affected. Default `true`.
4533
+ *
4534
+ * Kept on because the finding is *useful and correctly weaker*: it is raised with a lower
4535
+ * confidence and the opposite conclusion — one unhappy receiver out of eight points at that
4536
+ * receiver, not at the publisher. Turn it off if you only want publisher-blaming findings and
4537
+ * treat single-receiver trouble as the client's own business.
4538
+ */
4539
+ reportSingleReceiver: boolean;
4540
+ /**
4541
+ * Re-arm time per (track, issue type) in ms. Default `60_000`.
4542
+ *
4543
+ * Per track, so a call with many bad publishers still reports each of them. Typical
4544
+ * `30_000`–`300_000`.
4545
+ */
4546
+ cooldownMs: number;
4547
+ };
4548
+ /**
4549
+ * Attributes **client-reported issues to the published track they are about**, then asks how far
4550
+ * the problem fans out across that track's receivers.
4551
+ *
4552
+ * The join is what makes this possible: a receiver-side issue payload carries `trackId` (the client
4553
+ * detectors report it for every track-scoped issue), the observer resolves that to an inbound track,
4554
+ * and `RemoteTrackResolver` links the inbound track to the `remoteOutboundTrack` that published it.
4555
+ * With the whole subscriber set of one source in hand, the verdict is straightforward and is the
4556
+ * single most useful thing a server can say:
4557
+ *
4558
+ * - **most receivers of Alice's track are affected** → the fault is on Alice's path — her uplink, the
4559
+ * SFU's ingress, or its forwarding of that stream.
4560
+ * - **one receiver of Alice's track is affected** → that receiver's downlink. Nothing to do with
4561
+ * Alice, even though the symptom is reported against her stream.
4562
+ *
4563
+ * Deliberately generic over the issue vocabulary: `freezed-video-track`, `keyframe-storm`,
4564
+ * `audio-concealment`, `video-decoder-overloaded`, `stuck-decoder` and anything a custom client
4565
+ * detector invents all fan out the same way, so one mechanism replaces a family of symptom-specific
4566
+ * detectors.
4567
+ *
4568
+ * ### It walks the affected tracks, never all of them
4569
+ *
4570
+ * The detector is fed open issues by the call's registry and keeps only those carrying a `trackId`.
4571
+ * Each tick it resolves *those* tracks to their publishers — never the published tracks of the call,
4572
+ * of which there are many more and almost all of them fine. A call with nothing wrong costs one
4573
+ * `size === 0` check.
4574
+ *
4575
+ * ### Requires a `RemoteTrackResolver`
4576
+ *
4577
+ * Without publisher↔subscriber links there is no way to know which receivers belong to one source,
4578
+ * so the detector does nothing when the call has no resolver. It does not fall back to guessing:
4579
+ * "one receiver of an unknown set" is not a statement worth raising.
4580
+ */
4581
+ declare class IssueFanOutDetector implements Detector, ActiveIssueTracker {
4582
+ private readonly _call;
4583
+ static readonly NAME = "issue-fan-out-detector";
4584
+ readonly name = "issue-fan-out-detector";
4585
+ private readonly _config;
4586
+ private readonly _lastRaisedAt;
4587
+ /** Open issues that name a track. Issues without a `trackId` cannot be attributed and are dropped. */
4588
+ private readonly _trackIssues;
4589
+ constructor(_call: ObservedCall, config?: Partial<IssueFanOutDetectorConfig>);
4590
+ get size(): number;
4591
+ has(issue: ActiveClientIssue): boolean;
4592
+ add(issue: ActiveClientIssue): void;
4593
+ delete(issue: ActiveClientIssue): boolean;
4594
+ clear(): void;
4595
+ update(): void;
4596
+ close(): void;
4597
+ /**
4598
+ * Resolve the issue's `trackId` to the outbound track that published it.
4599
+ *
4600
+ * Looked up through the reporting client's own peer connections rather than by scanning the call:
4601
+ * the issue names its client, so the search is bounded by that client's transports (typically one
4602
+ * or two) instead of by the size of the meeting.
4603
+ */
4604
+ private _publisherOf;
4605
+ }
4606
+
4607
+ /**
4608
+ * One completed sampling bucket: how many/which clients reported congestion during it.
4609
+ *
4610
+ * `totalClients` (and therefore `congestedClientRatio`) is a snapshot of `observer.numberOfClients`
4611
+ * taken when the bucket closes — an approximation of "how many clients could have been congested",
4612
+ * not a claim that every client sent exactly one sample within the bucket. Good enough for a ratio
4613
+ * that only needs to be comparable bucket-to-bucket.
4614
+ */
4615
+ type SfuCongestionDetectorBucket = {
4616
+ observedAt: number;
4617
+ totalClients: number;
4618
+ congestedClients: number;
4619
+ congestedClientRatio: number;
4620
+ affectedClientIds: string[];
4621
+ affectedCallIds: string[];
4622
+ };
4623
+ type SfuCongestionDetectorReport = {
4624
+ affectedCallIds: string[];
4625
+ affectedClientIds: string[];
4626
+ congestedClientRatio: number;
4627
+ totalNumberOfClients: number;
4628
+ numberOfCongestedClients: number;
4629
+ historySize: number;
4630
+ baselineCongestedClientRatio: number;
4631
+ robustZ: number;
4632
+ absoluteIncrease: number;
4633
+ relativeIncrease: number;
4634
+ };
4635
+ type SfuCongestionDetectorConfig = {
4636
+ /**
4637
+ * Client issue types counted as "congestion" for this indicator. Default `[ 'congestion' ]`.
4638
+ *
4639
+ * Keep this narrow. Every type you add widens what counts as a congested client, and the whole
4640
+ * method rests on comparing *like with like* across time buckets — mixing in a type that appears
4641
+ * for unrelated reasons raises the baseline and buries the spike you are looking for.
4642
+ */
4643
+ consumedClientIssueTypes: string[];
4644
+ /** The `observer-issue` type raised when a bucket is judged congested. Default `'sfu-congestion'`. */
4645
+ emittedObserverIssueType: string;
4646
+ /**
4647
+ * How long your clients take to send a sample (ms). Default `10_000`.
4648
+ *
4649
+ * **Set this to your collector's actual sampling period** — it is a description of your clients,
4650
+ * not a tuning knob. The bucket is `samplesSendingTimeInMs * 2`, so every client gets a fair
4651
+ * chance to report at least once inside each bucket. Set it too short and clients that simply had
4652
+ * not reported yet look absent, so the ratio jumps around on sampling noise; too long and the
4653
+ * detector reacts slowly and averages a spike away.
4654
+ */
4655
+ samplesSendingTimeInMs: number;
4656
+ /**
4657
+ * Completed buckets kept as history — the candidate plus its baseline. Default `30`.
4658
+ *
4659
+ * At the default bucket size this is ~10 minutes of baseline. Sensible range `10`–`60`. Longer is
4660
+ * more robust to a single odd bucket but slower to accept a genuinely changed normal (a growth
4661
+ * spurt, a new region coming online); shorter adapts quickly but lets a sustained problem become
4662
+ * the new baseline and stop being reported.
4663
+ */
4664
+ historySize: number;
4665
+ /**
4666
+ * Buckets required before any judgement is made. Default `5`.
4667
+ *
4668
+ * Below this the detector is silent, which is the point: a median and MAD over two buckets is not
4669
+ * a baseline. Costs `minHistorySize * samplesSendingTimeInMs * 2` of warm-up after start — about
4670
+ * 100 s at the defaults. Do not lower it to make a test fire faster; shorten the bucket instead.
4671
+ */
4672
+ minHistorySize: number;
4673
+ /**
4674
+ * Distinct congested clients required in the candidate bucket. Default `3`.
4675
+ *
4676
+ * The absolute floor beneath every ratio below, so that a tiny fleet cannot produce a finding: two
4677
+ * unhappy clients out of four is 50% and means nothing. Raise it on a large fleet where three
4678
+ * clients is always noise.
4679
+ */
4680
+ minAffectedClients: number;
4681
+ /**
4682
+ * How far the candidate's congested-client **ratio** must exceed the baseline median, in absolute
4683
+ * terms (`0`–`1`). Default `0.05`, i.e. five percentage points.
4684
+ *
4685
+ * This is the practical-significance gate: it stops a statistically striking move from 0.5% to 2%
4686
+ * being reported as an event. Typical `0.03`–`0.15`.
4687
+ */
4688
+ minAbsoluteRatioIncrease: number;
4689
+ /**
4690
+ * How many times the baseline median the candidate ratio must reach. Default `2`.
4691
+ *
4692
+ * Multiplicative counterpart to the absolute gate — both must pass. Typical `1.5`–`3`. Below
4693
+ * `1.5` ordinary fluctuation qualifies; above ~`4` only near-total events do.
4694
+ */
4695
+ minRelativeRatioIncrease: number;
4696
+ /**
4697
+ * Robust z-score the candidate must reach against a median+MAD baseline. Default `3`.
4698
+ *
4699
+ * The statistical-significance gate. `3` is the conventional "clearly outside normal variation";
4700
+ * `2` is noticeably chattier, `4`–`5` only for very stable fleets. Median and MAD rather than mean
4701
+ * and standard deviation on purpose — a couple of past incidents in the history would inflate a
4702
+ * standard deviation enough to hide the next one. Note that a perfectly flat baseline gives
4703
+ * `MAD = 0`, where any increase scores `Infinity`; the two ratio gates above are what keep that
4704
+ * honest.
4705
+ */
4706
+ robustZThreshold: number;
4707
+ };
4708
+ /** The statistical/practical-significance verdict for one candidate bucket against its baseline. */
4709
+ type SfuCongestionDetectorEvaluation = {
4710
+ isCongested: boolean;
4711
+ baselineCongestedClientRatio: number;
4712
+ robustZ: number;
4713
+ absoluteIncrease: number;
4714
+ relativeIncrease: number;
4715
+ };
4716
+ /**
4717
+ * Detects a **shared** congestion event: many clients, across different calls, reporting congestion
4718
+ * inside the same slice of time.
4719
+ *
4720
+ * Only add this when the observer's calls all come from the **same SFU** — the finding's whole
4721
+ * meaning is "these clients have nothing in common except that server", and that is only true if the
4722
+ * server really is the common factor.
4723
+ *
4724
+ * ### Why fixed-interval buckets, and not the update tick
4725
+ *
4726
+ * The obvious implementation counts congested clients on each `update()`. It is wrong here, for two
4727
+ * separate reasons:
4728
+ *
4729
+ * - **The tick is not evenly spaced.** `update()` fires when a client is updated, so its rate is a
4730
+ * function of how many clients are connected and how their sampling happens to interleave. Two
4731
+ * counts taken from windows of different length are not comparable, and this detector's entire
4732
+ * job is to compare a count against earlier counts.
4733
+ * - **Clients report on their own schedule.** A client sends a sample roughly every
4734
+ * `samplesSendingTimeInMs`, unsynchronised with every other client. A window shorter than that
4735
+ * systematically undercounts — half the congested clients simply hadn't spoken yet — and the
4736
+ * undercount varies with arrival phase, which is noise indistinguishable from signal.
4737
+ *
4738
+ * So the detector runs on a wall-clock interval and closes a bucket every
4739
+ * `samplesSendingTimeInMs`, giving every client a fair chance to be heard in each one. Buckets are
4740
+ * equal-length and equally lagged, which is what makes bucket-to-bucket comparison mean something.
4741
+ * {@link update} is deliberately empty: nothing here is driven by the update tick.
4742
+ *
4743
+ * ### Occurrences, not intervals
4744
+ *
4745
+ * Unlike `ConcurrentIssueDetector`, this one ignores resolutions — see {@link delete}. It counts how
4746
+ * many *distinct clients reported* congestion in a bucket, not how many are still congested. A
4747
+ * client that hits congestion and immediately drops its bitrate resolves the issue within seconds
4748
+ * and would vanish from an open-interval view, yet it is exactly the evidence wanted here.
4749
+ */
4750
+ declare class SfuCongestionDetector implements Detector, ActiveIssueTracker {
4751
+ private readonly _observer;
4752
+ static readonly NAME = "sfu-congestion-detector";
4753
+ readonly name = "sfu-congestion-detector";
4754
+ private readonly _config;
4755
+ private readonly _history;
4756
+ private readonly _trackedIssues;
4757
+ private timer;
4758
+ private _lastEvaluatedBucket;
4759
+ constructor(_observer: Observer, config?: Partial<SfuCongestionDetectorConfig>);
4760
+ get size(): number;
4761
+ /** Record the issue against the bucket currently open. The timer, not this, closes the bucket. */
4762
+ add(issue: ActiveClientIssue): void;
4763
+ /**
4764
+ * Deliberately a no-op returning `false`.
4765
+ *
4766
+ * Resolutions are not interesting here. A congested client typically fixes its own symptom by
4767
+ * dropping bitrate hard, so the issue closes within seconds — but it still *happened*, and it is
4768
+ * evidence that the server was under pressure during this bucket. What matters is how many
4769
+ * distinct clients reported congestion within the bucket and whether that count suddenly jumps,
4770
+ * not how long any one client's issue stayed open.
4771
+ *
4772
+ * Nothing leaks: the tracked set is emptied wholesale every time a bucket closes.
4773
+ */
4774
+ delete(_issue: ActiveClientIssue): boolean;
4775
+ clear(): void;
4776
+ has(issue: ActiveClientIssue): boolean;
4777
+ close(): void;
4778
+ /** The completed buckets kept so far, oldest first. Read-only — for introspection/tests. */
4779
+ get history(): readonly SfuCongestionDetectorBucket[];
4780
+ /**
4781
+ * Intentionally empty — see the class description.
4782
+ *
4783
+ * Everything here is driven by the bucket timer, because the update tick is neither evenly spaced
4784
+ * nor long enough for every client to have reported. Counting on it would compare windows of
4785
+ * different lengths and call the difference a signal.
4786
+ */
4787
+ update(): void;
4788
+ private _closeBucket;
4789
+ /**
4790
+ * Evaluate only the latest completed bucket (the candidate) against the buckets before it (the
4791
+ * baseline) — never against itself. Reached once per newly-closed bucket via {@link update}; the
4792
+ * identity check below additionally guards against evaluating the same bucket twice, in case
4793
+ * `update()` is ever called again before the next rotation.
4794
+ */
4795
+ private _evaluateLatestBucket;
4796
+ /**
4797
+ * Is `candidate` — the latest completed bucket — abnormally high compared with the `baseline`
4798
+ * buckets before it?
4799
+ *
4800
+ * Requires both **statistical** significance (a robust z-score against a median+MAD baseline —
4801
+ * deliberately not Mann-Kendall, which asks "is this a monotonic trend", not "is the latest point
4802
+ * an outlier"; a single sudden spike on an otherwise flat series is exactly what should trigger
4803
+ * here and exactly what a trend test would miss) and **practical** significance (enough affected
4804
+ * clients, and a big enough absolute/relative jump — a statistically significant move in a tiny
4805
+ * or trivial ratio is not worth an alert).
4806
+ */
4807
+ private _evaluateBucket;
4808
+ }
4809
+
4810
+ declare const TrackDeliveryMismatchTypes: {
4811
+ /**
4812
+ * The source is sending, but **none** of its subscribers are receiving → the media is being lost
4813
+ * between the publisher and the receivers. In an SFU that means the forwarding path.
4814
+ */
4815
+ readonly publishedTrackNotDelivered: "PUBLISHED_TRACK_NOT_DELIVERED";
4816
+ /**
4817
+ * The source is sending and most subscribers are fine, but **some** are dry → those consumers are
4818
+ * broken individually (in mediasoup, the usual mitigation is recreating the consumer).
4819
+ */
4820
+ readonly receiverTrackNotDelivered: "RECEIVER_TRACK_NOT_DELIVERED";
4821
+ /**
4822
+ * The source itself stopped producing, so its subscribers being dry is expected and **not** an
4823
+ * SFU fault. Reported so the other two verdicts can be trusted as *not* being this.
4824
+ */
4825
+ readonly publisherTrackDry: "PUBLISHER_TRACK_DRY";
4826
+ };
4827
+ type TrackDeliveryMismatchDetectorConfig = {
4828
+ /**
4829
+ * The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`, which is
4830
+ * what client-monitor-js raises. Only change it if you raise your own equivalent.
4831
+ */
4832
+ dryInboundIssueType: string;
4833
+ /**
4834
+ * The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`, as raised
4835
+ * by client-monitor-js. The pairing of these two types is the whole detector: their **disagreement**
4836
+ * is the finding.
4837
+ */
4838
+ dryOutboundIssueType: string;
4839
+ /**
4840
+ * Subscribers required before "all of them" means anything. Default `2`.
4841
+ *
4842
+ * With one subscriber, "every receiver is dry" is a single client's report and carries no more
4843
+ * weight than the client issue already does. Sensible range `2`–`4`.
4844
+ */
4845
+ minReceivers: number;
4846
+ /**
4847
+ * Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`.
4848
+ *
4849
+ * `1` — literally all of them — on purpose. The inference here is sharp: the publisher says it is
4850
+ * sending and *every* receiver says nothing arrives, so the fault is between them, in the SFU's
4851
+ * forwarding. Lowering it to `0.8` admits mixed evidence, where some receivers do get the media, and
4852
+ * the conclusion no longer follows: that is a per-receiver problem and `IssueFanOutDetector`'s
4853
+ * question. Do not lower it without deciding what the finding then means.
4854
+ */
4855
+ allReceiversRatio: number;
4856
+ /** Re-arm time per (track, verdict) in ms. Default `60_000`. Typical `30_000`–`300_000`. */
4857
+ cooldownMs: number;
4858
+ };
4859
+ /**
4860
+ * Answers **"is the media actually getting through?"** by joining the two ends of a published track.
4861
+ *
4862
+ * A dry track is the clearest possible symptom — no bytes are arriving — but on its own it is
4863
+ * ambiguous, and the ambiguity is precisely what a single endpoint cannot resolve. A receiver seeing
4864
+ * silence cannot tell whether the camera was switched off, the SFU stopped forwarding, or its own
4865
+ * consumer wedged. All three look identical from the browser.
4866
+ *
4867
+ * With the publisher↔subscriber links this becomes a three-way decision:
4868
+ *
4869
+ * | publisher | subscribers | verdict |
4870
+ * |---|---|---|
4871
+ * | sending | **all** dry | `PUBLISHED_TRACK_NOT_DELIVERED` — the SFU/forwarding path |
4872
+ * | sending | **some** dry | `RECEIVER_TRACK_NOT_DELIVERED` — those consumers (recreate them) |
4873
+ * | dry | any dry | `PUBLISHER_TRACK_DRY` — the source stopped; not an SFU fault |
4874
+ *
4875
+ * The publisher side is judged from **both** signals available: its own `dry-outbound-track` issue
4876
+ * when the client reports one, and — as the fallback, and the corroboration when it does not — the
4877
+ * observed outbound RTP (`deltaPacketsSent`). That combination is what makes the first row
4878
+ * trustworthy: the server can state that packets demonstrably left the publisher during the same
4879
+ * interval in which every receiver got nothing.
4880
+ *
4881
+ * This is the "SFU forwarding mismatch" check, and notably it needs **no** mediasoup instrumentation
4882
+ * — the client's own dry-track verdicts plus the resolver links are sufficient.
4883
+ */
4884
+ declare class TrackDeliveryMismatchDetector implements Detector, ActiveIssueTracker {
4885
+ private readonly call;
4886
+ static readonly NAME: "track-delivery-mismatch-detector";
4887
+ readonly name: "track-delivery-mismatch-detector";
4888
+ private readonly _config;
4889
+ private readonly _lastRaisedAt;
4890
+ private readonly dryOutboundTracks;
4891
+ private readonly dryInboundTracks;
4892
+ constructor(call: ObservedCall, config?: Partial<TrackDeliveryMismatchDetectorConfig>);
4893
+ close(): void;
4894
+ add(issue: ActiveClientIssue): void;
4895
+ delete(issue: ActiveClientIssue): boolean;
4896
+ get size(): number;
4897
+ clear(): void;
4898
+ has(issue: ActiveClientIssue): boolean;
4899
+ update(): void;
4900
+ }
4901
+
4902
+ declare const TurnServerHealthTypes: {
4903
+ /** One TURN server's clients are in trouble while other servers' clients are fine. */
4904
+ readonly turnServerDegraded: "TURN_SERVER_DEGRADED";
4905
+ };
4906
+ type TurnServerHealthDetectorConfig = {
4907
+ /**
4908
+ * Clients a server must be carrying before its ratio means anything. Default `5`.
4909
+ *
4910
+ * With two relayed clients, "half are degraded" is one person having a bad time. Sensible range
4911
+ * `5`–`20`; raise it if you run many small TURN deployments, since each needs enough traffic to be
4912
+ * measurable on its own.
4913
+ */
4914
+ minClientsPerServer: number;
4915
+ /**
4916
+ * Fraction of a server's clients that must have an open issue, `0`–`1`. Default `0.5`.
4917
+ *
4918
+ * Typical `0.4`–`0.7`. Remember each finding also carries the *other* servers' ratios, so the
4919
+ * threshold is not doing the comparison on its own — a server at 50% next to peers at 45% reads very
4920
+ * differently from one next to peers at 3%. Below `0.3` you will report servers that are merely
4921
+ * carrying unlucky clients.
4922
+ */
4923
+ degradedRatioThreshold: number;
4924
+ /**
4925
+ * Which client issue types count as "in trouble". Empty (the default) means **any** open issue.
4926
+ *
4927
+ * The permissive default is deliberate and unusual for this library: the question is not *what* is
4928
+ * wrong with each client but whether trouble clusters on one relay, and a relay problem shows up as
4929
+ * whatever symptom each client happens to notice first. Narrow it to network types
4930
+ * (`congestion`, `ice-disconnected`) if endpoint issues like `cpulimitation` are common enough in
4931
+ * your fleet to blur the comparison between servers.
4932
+ */
4933
+ issueTypes: string[];
4934
+ /**
4935
+ * Consecutive `observer.update()` ticks the condition must hold before raising. Default `2`.
4936
+ *
4937
+ * The de-bounce. `1` reacts immediately and will fire on a single tick where several clients
4938
+ * happened to be mid-reconnect; `2`–`3` costs a tick or two of delay and removes most of that.
4939
+ * Note this counts *ticks*, not time, so how long it actually waits depends on your sample rate.
4940
+ */
4941
+ consecutiveTicks: number;
4942
+ /**
4943
+ * Re-arm time per server (ms). Default `60_000`.
4944
+ *
4945
+ * Shorter than the outage detector's, because degradation is a condition you may want re-reported as
4946
+ * it persists or worsens, not a single event. Typical `60_000`–`300_000`.
4947
+ */
4948
+ cooldownMs: number;
4949
+ };
4950
+ /** The per-server view this detector builds. */
4951
+ type TurnServerHealth = {
4952
+ serverUrl: string;
4953
+ /** Distinct clients whose media is relayed through this server. */
4954
+ clients: number;
4955
+ /** Of those, how many currently have at least one open issue. */
4956
+ degradedClients: number;
4957
+ degradedRatio: number;
4958
+ affectedClientIds: string[];
4959
+ /** The open issue types seen on this server's clients, most common first. */
4960
+ issueTypes: string[];
4961
+ };
4962
+ /**
4963
+ * An **observer-level** detector that groups relayed clients by the TURN server carrying them and
4964
+ * compares the servers against each other.
4965
+ *
4966
+ * Counting TURN usage is not useful on its own; knowing that `turn-eu-1` has 22 of 30 clients in
4967
+ * trouble while `turn-eu-2` has 1 of 34 is. Because the comparison spans calls it lives on
4968
+ * `observer.detectors` and raises `observer-issue` — one actionable alert instead of fifty
4969
+ * per-client ones. Each finding carries the other servers' ratios as context, since "half the
4970
+ * clients here are unhappy" only means something relative to the rest of the fleet.
4971
+ *
4972
+ * Whether a client is in trouble comes from **its own reported issues**, not from thresholds applied
4973
+ * here. The client already decides that far better than a server-side rule could; the value this
4974
+ * adds is the grouping — the dimension no endpoint can see.
4975
+ *
4976
+ * For a relay that has stopped serving entirely, see `TurnServerOutageDetector`: this detector needs
4977
+ * clients *on* the server to ask how many are unhappy, and an outage takes them away.
4978
+ */
4979
+ declare class TurnServerHealthDetector implements Detector {
4980
+ private readonly _observer;
4981
+ static readonly NAME = "turn-server-health-detector";
4982
+ readonly name = "turn-server-health-detector";
4983
+ private readonly _config;
4984
+ private readonly _streaks;
4985
+ private readonly _lastRaisedAt;
4986
+ /** The per-server rollup computed on the most recent `update()`. */
4987
+ lastServers: TurnServerHealth[];
4988
+ constructor(_observer: Observer, config?: Partial<TurnServerHealthDetectorConfig>);
4989
+ update(): void;
4990
+ close(): void;
4991
+ private _serverHealth;
4992
+ }
4993
+
4994
+ declare const TurnServerOutageTypes: {
4995
+ /** One TURN server's relayed population collapsed while the rest of the fleet is fine. */
4996
+ readonly turnServerOutage: "TURN_SERVER_OUTAGE";
4997
+ };
4998
+ type TurnServerOutageDetectorConfig = {
4999
+ /**
5000
+ * Clients a server must have been carrying at its peak before its collapse means anything. Default
5001
+ * `5`.
5002
+ *
5003
+ * Below this, one or two people leaving looks like an outage. Sensible range `5`–`50`; the higher it
5004
+ * is the more confident the finding, and the more small deployments go unwatched.
5005
+ */
5006
+ minClientsAtPeak: number;
5007
+ /**
5008
+ * Fraction of the peak population that must be gone or disrupted. Default `0.8` — an outage is
5009
+ * near-total by definition; partial degradation is `TurnServerHealthDetector`'s question.
5010
+ */
5011
+ lossRatioThreshold: number;
5012
+ /**
5013
+ * Window the peak population is measured over (ms). Default `120_000`.
5014
+ *
5015
+ * Long enough to span a real outage's onset, short enough that yesterday's peak is not held against
5016
+ * today. Typical `60_000`–`600_000`. Too long and the natural end of a busy period reads as a
5017
+ * collapse; too short and a gradual failure never shows a peak to fall from.
5018
+ */
5019
+ peakWindowMs: number;
5020
+ /**
5021
+ * Require a healthy **control group** — clients not relayed through this server that are still
5022
+ * connected — before blaming the server. Without this, a call ending, a fleet-wide network
5023
+ * event, or the observer shutting down all look exactly like a TURN outage. Default `true`.
5024
+ */
5025
+ requireControlGroup: boolean;
5026
+ /**
5027
+ * Clients elsewhere before the control group is worth anything. Default `5`.
5028
+ *
5029
+ * If you run a single TURN server there is never a control group, so with `requireControlGroup: true`
5030
+ * this detector can never fire — which is correct rather than unfortunate: with one relay you cannot
5031
+ * distinguish "the relay died" from "everyone went home". Sensible range `5`–`20`.
5032
+ */
5033
+ minControlGroupClients: number;
5034
+ /**
5035
+ * Fraction of the control group that must still be healthy, `0`–`1`. Default `0.7`.
5036
+ *
5037
+ * The evidence that the rest of the world is fine. Typical `0.6`–`0.9`. Set it too high and a
5038
+ * concurrent unrelated problem elsewhere masks a real outage; too low and a fleet-wide network event
5039
+ * gets blamed on whichever server lost clients first.
5040
+ */
5041
+ controlGroupHealthyRatio: number;
5042
+ /**
5043
+ * Consecutive `observer.update()` ticks the condition must hold before raising. Default `2`.
5044
+ *
5045
+ * Counts ticks, not time. `1` will fire on a single tick where a batch of clients happened to be
5046
+ * between samples; `2`–`4` is the useful range for something this consequential to declare.
5047
+ */
5048
+ consecutiveTicks: number;
5049
+ /**
5050
+ * Re-arm time (ms) per server. Long by default (`300_000`) — an outage is one event, not one
5051
+ * per tick, and a server that stays down would otherwise alert forever.
5052
+ */
5053
+ cooldownMs: number;
5054
+ };
5055
+ /**
5056
+ * Detects a **TURN server outage** — a relay that has stopped serving — by watching its client
5057
+ * population collapse while the rest of the fleet carries on.
5058
+ *
5059
+ * This is the case its sibling `TurnServerHealthDetector` structurally *cannot* see, and the
5060
+ * distinction is worth being precise about. That detector groups clients by the server relaying them
5061
+ * and asks how many are reporting issues. It needs clients on the server to ask the question. When a
5062
+ * TURN server goes down completely, allocation fails: existing sessions drop, and new clients never
5063
+ * obtain a relay candidate through it at all, so they are never attributed to it. The server's
5064
+ * population goes to zero and the health detector falls silent for the worst possible reason — it
5065
+ * has nobody left to ask. Degradation makes clients unhappy; an outage makes them *disappear*.
5066
+ *
5067
+ * So the signal here is absence, measured against the server's own recent peak:
5068
+ *
5069
+ * - clients gone entirely (their relayed peer connections closed, or they re-negotiated onto a
5070
+ * different path), plus
5071
+ * - clients still attributed to the server whose ICE or connection state is `disconnected` /
5072
+ * `failed` / `closed` — the ones mid-collapse, which is what you catch if you look during the
5073
+ * outage rather than after it.
5074
+ *
5075
+ * ### The control group is the whole design
5076
+ *
5077
+ * Absence is a dangerous signal: a call ending, everyone going home at 6pm, a fleet-wide network
5078
+ * event, and the observer itself shutting down all produce exactly the same collapse. The detector
5079
+ * therefore refuses to blame a server unless clients **not** relayed through it are demonstrably
5080
+ * still connected — `requireControlGroup`, on by default. "Everyone on `turn-eu-1` vanished" is
5081
+ * ambiguous; "everyone on `turn-eu-1` vanished while 200 clients elsewhere are fine" is an outage.
5082
+ *
5083
+ * That comparison is only available to something watching every call at once, which is why this is
5084
+ * an observer-level detector raising `observer-issue` — one alert for the fleet, not one per
5085
+ * abandoned call.
5086
+ *
5087
+ * ### Caveats worth knowing before you tune it
5088
+ *
5089
+ * Clients that fail over cleanly to a second TURN server still count as lost here, which is
5090
+ * correct — the server did stop serving them — but it means a well-configured fleet with automatic
5091
+ * failover reports outages that users never felt. That is the intended behaviour: the failover
5092
+ * worked *and* the server is down are both true, and you want to know the second one.
5093
+ *
5094
+ * A genuinely quiet server (last call of the day ends) is suppressed by the control group, not by
5095
+ * the collapse test. If you run a small deployment where the control group is routinely below
5096
+ * `minControlGroupClients`, this detector will stay quiet — prefer alerting on your TURN server's
5097
+ * own health checks there, since a handful of clients cannot distinguish these cases.
5098
+ */
5099
+ declare class TurnServerOutageDetector implements Detector {
5100
+ private readonly _observer;
5101
+ static readonly NAME = "turn-server-outage-detector";
5102
+ readonly name = "turn-server-outage-detector";
5103
+ private readonly _config;
5104
+ /** serverUrl -> recent population observations, used to derive the windowed peak. */
5105
+ private readonly _peaks;
5106
+ private readonly _streaks;
5107
+ private readonly _lastRaisedAt;
5108
+ constructor(_observer: Observer, config?: Partial<TurnServerOutageDetectorConfig>);
5109
+ update(): void;
5110
+ close(): void;
5111
+ /** Distinct clients on a server, split by whether their relayed transport is actually up. */
5112
+ private _populationOf;
5113
+ /** Record this tick's population and return the peak across `peakWindowMs`. */
5114
+ private _recordAndPeak;
5115
+ /**
5116
+ * Everyone *not* relayed through `serverUrl`: clients on other TURN servers plus every client
5117
+ * the observer knows about that isn't relayed at all. The healthy share of that group is what
5118
+ * separates "this server broke" from "everything broke".
5119
+ */
5120
+ private _controlGroup;
5121
+ }
5122
+
5123
+ declare const UnconsumedTrackTypes: {
5124
+ /** A track is being published to the SFU that nobody is subscribed to — pure wasted uplink. */
5125
+ readonly unconsumedPublishedTrack: "UNCONSUMED_PUBLISHED_TRACK";
5126
+ };
5127
+ type UnconsumedTrackDetectorConfig = {
5128
+ /**
5129
+ * How long a track must stay unconsumed **while still sending** before it is reported (ms).
5130
+ * Default `30_000`.
5131
+ *
5132
+ * This is the main guard against a false alarm, because a gap between publishing and the first
5133
+ * subscription is completely normal at join time — and again after every renegotiation. Sensible
5134
+ * range `15_000`–`120_000`. Too low and you report every join; too high and you tolerate wasted
5135
+ * uplink for longer than you need to. Waste is not an outage, so err high.
5136
+ */
5137
+ minUnconsumedDurationInMs: number;
5138
+ /**
5139
+ * Ignore tracks sending below this bitrate (**bits per second**). Default `50_000` (50 kbps).
5140
+ *
5141
+ * The point of the detector is wasted bandwidth, and a track trickling keep-alive packets wastes
5142
+ * none worth an alert. Typical `20_000`–`100_000`: muted or paused tracks sit near zero, a real
5143
+ * video track is hundreds of kbps. Set it to `0` to report every unconsumed track regardless of
5144
+ * cost.
5145
+ */
5146
+ minBitrate: number;
5147
+ /**
5148
+ * Re-arm time per track (ms). Default `300_000`.
5149
+ *
5150
+ * Long on purpose: an unconsumed track usually *stays* unconsumed, so a short cooldown means a
5151
+ * steady drip of the same finding for the life of the call. Typical `300_000`–`900_000`.
5152
+ */
5153
+ cooldownMs: number;
5154
+ };
5155
+ /**
5156
+ * Finds tracks that are **published but consumed by nobody** — uplink and SFU ingress spent on media
5157
+ * that is never forwarded anywhere.
5158
+ *
5159
+ * This is the one detector that reads the resolver's *silence* as the signal: an outbound track with
5160
+ * an empty `remoteInboundTracks` set, still pushing packets. It reads `call.unconsumedOutboundTracks`,
5161
+ * which the resolver maintains as tracks gain and lose subscribers, so a healthy call costs one
5162
+ * `size === 0` check rather than a walk over every published track. The usual causes are a participant
5163
+ * publishing while everyone has them hidden or muted-in-UI, a simulcast layer no viewer's bandwidth
5164
+ * ever selects, or an application that forgot to stop a track after the last subscriber left.
5165
+ *
5166
+ * It is deliberately slow to fire: `minUnconsumedDurationInMs` must elapse with the track still
5167
+ * sending, because a brief gap between publishing and the first subscription is completely normal at
5168
+ * join time.
5169
+ *
5170
+ * ### Careful: this detector is only sound with a resolver
5171
+ *
5172
+ * "No subscribers" and "no resolver configured" produce the identical observation — an empty link
5173
+ * set. Without a `RemoteTrackResolver` this would report *every* published track in the call as
5174
+ * unconsumed, so it checks `call.remoteTrackResolver` at runtime and does nothing without one.
5175
+ */
5176
+ declare class UnconsumedTrackDetector implements Detector {
5177
+ private readonly call;
5178
+ static readonly NAME = "unconsumed-track-detector";
5179
+ readonly name = "unconsumed-track-detector";
5180
+ readonly config: UnconsumedTrackDetectorConfig;
5181
+ /** trackId -> when it was first seen sending with no subscribers. */
5182
+ private readonly _unconsumedSince;
5183
+ private readonly _lastRaisedAt;
5184
+ constructor(call: ObservedCall, config?: Partial<UnconsumedTrackDetectorConfig>);
5185
+ update(): void;
5186
+ close(): void;
5187
+ }
5188
+
5189
+ /**
5190
+ * Detectors that reason **across calls**, created once on the observer.
5191
+ *
5192
+ * Adding one means: give the class a `static readonly NAME`, add its entry here, and add a `case` to
5193
+ * `Observer.addObserverDetector`. This map is what types the call site — the config is checked
5194
+ * against the right detector and an unknown name won't compile.
5195
+ */
5196
+ type AvailableObserverScopeDetectorsConfigs = {
5197
+ [SfuCongestionDetector.NAME]: SfuCongestionDetectorConfig;
5198
+ [ObserverConcurrentIssueDetector.NAME]: ObserverConcurrentIssueDetectorConfig;
5199
+ [ClientPopulationIssueDetector.NAME]: ClientPopulationIssueDetectorConfig;
5200
+ [TurnServerHealthDetector.NAME]: TurnServerHealthDetectorConfig;
5201
+ [TurnServerOutageDetector.NAME]: TurnServerOutageDetectorConfig;
5202
+ };
5203
+ /**
5204
+ * Detectors that reason **within one call**, created for every call the observer opens.
5205
+ *
5206
+ * Note there is no detector in both maps. "Is this meeting in trouble?" and "is our infrastructure in
5207
+ * trouble?" are different questions with different gates and different findings, so they are separate
5208
+ * classes — `CallConcurrentIssueDetector` and `ObserverConcurrentIssueDetector` — rather than one
5209
+ * class branching on what it was handed.
5210
+ */
5211
+ type AvailableCallScopeDetectorsConfigs = {
5212
+ [UnconsumedTrackDetector.NAME]: UnconsumedTrackDetectorConfig;
5213
+ [TrackDeliveryMismatchDetector.NAME]: TrackDeliveryMismatchDetectorConfig;
5214
+ [CallConcurrentIssueDetector.NAME]: CallConcurrentIssueDetectorConfig;
5215
+ [IssueFanOutDetector.NAME]: IssueFanOutDetectorConfig;
5216
+ [PublisherFaultCorroborationDetector.NAME]: PublisherFaultCorroborationDetectorConfig;
5217
+ };
5218
+ type AvailableDetectorsConfigs = AvailableObserverScopeDetectorsConfigs | AvailableCallScopeDetectorsConfigs;
5219
+ declare class Detectors {
5220
+ private _detectors;
5221
+ constructor(...detectors: Detector[]);
5222
+ /**
5223
+ * Every registered detector, in registration order.
5224
+ *
5225
+ * This is **the** way to get hold of an instance: `addDetector` / `addObserverDetector` are
5226
+ * chainable and return the owning entity, so the registry is where instances live. Read it to
5227
+ * inspect a detector's state, or to pick one out and {@link remove} it.
5228
+ *
5229
+ * A copy, not the live array — a caller iterating this while removing would otherwise skip
5230
+ * entries, and that is exactly what "remove the ones that look like X" does.
5231
+ */
5232
+ get instances(): Detector[];
5233
+ /** Iterate the registry directly: `for (const detector of call.detectors)`. */
5234
+ [Symbol.iterator](): IterableIterator<Detector>;
5235
+ /** The names in registration order. Duplicates are meaningful — see {@link getAll}. */
5236
+ get listOfNames(): string[];
5237
+ get size(): number;
5238
+ add(detector: Detector): void;
5239
+ /** The first detector registered under `name`. See {@link getAll} when several can share one. */
5240
+ get(name: string): Detector | undefined;
5241
+ /**
5242
+ * Every detector registered under `name`.
5243
+ *
5244
+ * More than one is legitimate: `ClientPopulationIssueDetector` is meant to be added once per
5245
+ * `groupBy` axis, and two instances of it share a name.
5246
+ */
5247
+ getAll(name: string): Detector[];
5248
+ has(name: string): boolean;
5249
+ /** Remove one specific instance. Returns `false` if it was not registered here. */
5250
+ remove(detector: Detector): boolean;
5251
+ /**
5252
+ * Remove **every** detector registered under `name`, returning how many were removed.
5253
+ *
5254
+ * All of them rather than the first, because a name can legitimately be registered more than once
5255
+ * (see {@link getAll}) and "remove the `client-population-issue-detector`" cannot sensibly mean
5256
+ * "remove whichever axis happens to be first in the array". Removing all of them is the only
5257
+ * behaviour that leaves the registry in a state the caller can predict from the name alone.
5258
+ *
5259
+ * Each removed detector gets `close()`, so trackers unsubscribe from the issue registry, bus
5260
+ * listeners drop, and timers clear — a detector removed without closing keeps being fed issues
5261
+ * forever.
5262
+ */
5263
+ removeByName(name: string): number;
5264
+ update(): void;
5265
+ clear(): void;
5266
+ private _close;
5267
+ }
5268
+
5269
+ /**
5270
+ * The set of client issues currently believed to be **open**, plus the fan-out that pushes them to
5271
+ * whoever asked for them.
5272
+ *
5273
+ * ### Push, not poll
5274
+ *
5275
+ * A detector does not scan for the issues it cares about; it registers as an
5276
+ * {@link ActiveIssueTracker} for the types it consumes and is handed them as they open and close.
5277
+ * The cost of a detector is then proportional to the issues it actually receives, not to the number
5278
+ * of participants — a healthy 500-client fleet does no work per tick.
5279
+ *
5280
+ * ```ts
5281
+ * observer.activeIssuesRegistry.addIssueTracker('congestion', detector);
5282
+ * ```
5283
+ *
5284
+ * There is **no wildcard**. A tracker names the types it consumes, and nothing else reaches it. "Feed
5285
+ * me everything and I'll work out what matters" pushes the decision from the application — which
5286
+ * knows its client build and its issue vocabulary — onto a detector that has to guess, and it makes
5287
+ * the cost of a subscription unbounded and invisible. If a detector should watch five issue types,
5288
+ * the caller lists five issue types.
5289
+ *
5290
+ * ### Two levels
5291
+ *
5292
+ * Every call owns a registry constructed with the observer's as its `parent`. An add or delete
5293
+ * touches both, so a call-scoped tracker sees only that call's issues while an observer-scoped one
5294
+ * sees the fleet — without either side iterating the other. The child keeps **its own** storage:
5295
+ * `size` is this scope's count, and {@link clear} (called when the call closes) removes only this
5296
+ * scope's issues from the parent and never touches the parent's tracker registrations.
5297
+ *
5298
+ * ### Only keyed issues arrive here
5299
+ *
5300
+ * An issue without a `key` has no lifecycle — nothing can ever close it — so treating it as "active"
5301
+ * would mean holding a symptom that may have ended long ago. Keyless issues stay one-shot: emitted
5302
+ * as `client-issue`, never registered. `client-monitor-js` >= 4.6.0 sends `key` on everything
5303
+ * stateful.
5304
+ */
5305
+ declare class ActiveIssuesRegistry implements ActiveIssueTracker {
5306
+ private readonly parent?;
5307
+ private readonly issues;
5308
+ private readonly typesToTrackers;
5309
+ constructor(parent?: ActiveIssueTracker | undefined);
5310
+ get size(): number;
5311
+ /**
5312
+ * The open issues in this scope, in insertion order.
5313
+ *
5314
+ * Insertion order is age order (`observedAt` is assigned on insert), which is what lets a consumer
5315
+ * stop at the first entry newer than its cutoff instead of scanning the whole set.
5316
+ */
5317
+ values(): IterableIterator<ActiveClientIssue>;
5318
+ [Symbol.iterator](): IterableIterator<ActiveClientIssue>;
5319
+ has(issue: ActiveClientIssue): boolean;
5320
+ add(issue: ActiveClientIssue): this;
5321
+ delete(issue: ActiveClientIssue): boolean;
5322
+ /** Feed `tracker` every issue of `type` as it opens and closes. One call per type; no wildcard. */
5323
+ addIssueTracker(type: string, tracker: ActiveIssueTracker): this;
5324
+ removeIssueTracker(tracker: ActiveIssueTracker): this;
5325
+ /**
5326
+ * Drop every issue in this scope, e.g. because the call closed.
5327
+ *
5328
+ * Deletes through {@link delete} so the parent sheds exactly this scope's issues. Tracker
5329
+ * *registrations* survive: a detector subscribed to the observer's registry must keep receiving
5330
+ * issues after any one call ends.
5331
+ */
5332
+ clear(): void;
5333
+ private _trackIssue;
5334
+ private _untrackIssue;
5335
+ /**
5336
+ * Apply `apply` to every tracker interested in `type`.
5337
+ *
5338
+ * A tracker throwing must not abort the fan-out: the issue has already been added to (or removed
5339
+ * from) this registry, so a partial dispatch would leave the remaining trackers permanently out of
5340
+ * step with it. One broken detector should not desynchronise the others.
5341
+ */
5342
+ private _trackersOf;
5343
+ private _safely;
5344
+ }
5345
+
5346
+ type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
5347
+ callId: string;
5348
+ appData?: AppData;
5349
+ closeCallIfEmptyForMs?: number;
5350
+ /**
5351
+ * When `true`, the call's `update()` is invoked whenever a client accepts a sample. When `false`, it is not.
5352
+ *
5353
+ * DEFAULT: `true` — the call is updated on every client sample, which is the most common use case.
5354
+ */
5355
+ autoUpdateOnClientUpdate?: boolean;
5356
+ };
5357
+ type ObservedCallEvents = {
5358
+ update: [];
5359
+ newclient: [ObservedClient];
5360
+ empty: [];
5361
+ 'not-empty': [];
5362
+ close: [];
5363
+ };
5364
+ declare interface ObservedCall {
5365
+ on<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
5366
+ off<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
5367
+ once<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
5368
+ emit<U extends keyof ObservedCallEvents>(event: U, ...args: ObservedCallEvents[U]): boolean;
5369
+ }
5370
+ declare class ObservedCall<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
5371
+ readonly observer: Observer;
5372
+ readonly activeIssuesRegistry: ActiveIssuesRegistry;
5373
+ scoreCalculator: ScoreCalculator;
5374
+ readonly detectors: Detectors;
5375
+ readonly callId: string;
5376
+ readonly observedClients: Map<string, ObservedClient<Record<string, unknown>>>;
5377
+ readonly clientsUsedTurn: Set<string>;
5378
+ readonly calculatedScore: CalculatedScore;
5379
+ remoteTrackResolver?: RemoteTrackResolver;
5380
+ /**
5381
+ * The accumulating record of this call's life, or `undefined` when no summary was configured.
5382
+ *
5383
+ * Live — read it at any point during the call. It is also delivered once on `call-summary` when
5384
+ * the call closes. See `CallSummary`: an absent section means "not collected", never "nothing
5385
+ * happened".
5386
+ */
5387
+ summary?: CallSummary;
5388
+ /**
5389
+ * Published tracks that currently have **no** subscriber linked to them.
5390
+ *
5391
+ * Maintained by the `RemoteTrackResolver` at the exact moments a track gains or loses its last
5392
+ * subscriber — the only moments the answer can change. `UnconsumedTrackDetector` reads this
5393
+ * instead of walking every published track in the call, so in a healthy call (where the set is
5394
+ * empty) it does no work at all.
5395
+ *
5396
+ * Empty when no resolver is configured: without links, "no subscribers" is unknowable.
5397
+ */
5398
+ readonly unconsumedOutboundTracks: Set<ObservedOutboundTrack>;
5399
+ totalAddedClients: number;
5400
+ totalRemovedClients: number;
5401
+ numberOfIssues: number;
5402
+ numberOfPeerConnections: number;
5403
+ numberOfInboundRtpStreams: number;
5404
+ numberOfOutboundRtpStreams: number;
5405
+ numberOfDataChannels: number;
5406
+ maxNumberOfClients: number;
5407
+ deltaNumberOfIssues: number;
5408
+ appData: AppData;
5409
+ closed: boolean;
5410
+ startedAt?: number;
5411
+ endedAt?: number;
5412
+ closedAt?: number;
5413
+ readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs' | 'autoUpdateOnClientUpdate'>;
5414
+ /** Ancestry base shared by all Observer-bus events originating at this call. */
5415
+ readonly eventScope: ObservedCallScope;
5416
+ private closeTimer?;
5417
+ constructor(settings: ObservedCallSettings<AppData>, observer: Observer, activeIssuesRegistry: ActiveIssuesRegistry);
5418
+ get numberOfClients(): number;
5419
+ get score(): number | undefined;
5420
+ /**
5421
+ * Build a call-scoped detector onto this call. Chainable.
5422
+ *
5423
+ * To get a handle on what was built — to inspect it, or to remove that exact instance later — read
5424
+ * it back off the registry: `call.detectors.getAll(name)`, or `call.detectors.instances`.
5425
+ */
5426
+ addDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
5427
+ /**
5428
+ * Start accumulating this call's summary, if the observer was configured for summaries.
5429
+ *
5430
+ * Called by `createObservedCall`; you should not need it. It takes no configuration of its own on
5431
+ * purpose: the collector subscribes to exactly the events the observer's `include` requires, so a
5432
+ * per-call section outside that set would be created and then never written to — an empty section
5433
+ * that reads as "nothing happened". One shape per observer is the only shape that can be filled.
5434
+ *
5435
+ * The collector builds it rather than this method, so the resolved configuration never has to
5436
+ * leave the one object that owns it. Returns `undefined` when summaries are off, and is
5437
+ * idempotent: an existing summary is kept, not restarted.
5438
+ */
5439
+ enableSummary(): CallSummary | undefined;
5440
+ /**
5441
+ * Remove a detector from **this call** by name, returning how many were removed.
5442
+ *
5443
+ * **Every** instance under the name goes — a name can legitimately be registered more than once.
5444
+ * When you want one of them specifically, go through the registry, which deals in instances:
5445
+ *
5446
+ * ```ts
5447
+ * const [ first ] = call.detectors.getAll('issue-fan-out-detector');
5448
+ *
5449
+ * call.detectors.remove(first);
5450
+ * ```
5451
+ *
5452
+ * Either route `close()`s the detector, so it unsubscribes from `activeIssuesRegistry` — without
5453
+ * that the registry keeps feeding a detector nobody is running any more, and its tracked set grows
5454
+ * for the life of the call.
5455
+ *
5456
+ * To stop building it on *future* calls too, use `observer.removeCallDetector(name)`.
5457
+ */
5458
+ removeDetector(name: keyof AvailableCallScopeDetectorsConfigs): number;
5459
+ /**
5460
+ * Raise a call-level (server-side) finding; surfaced on the Observer bus as `call-issue`.
5461
+ *
5462
+ * `payload` is an **object** and holds evidence only — it is delivered to an in-process handler,
5463
+ * so there is nothing to serialise for. `scope` is stamped here, and the `callId` is already on
5464
+ * the event, so neither belongs in the payload. Put the interpretation in `conclusion`.
5465
+ */
5466
+ addIssue(issue: Omit<CallIssue, 'scope'>): void;
5467
+ close(): void;
5468
+ getObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(clientId: string): ObservedClient<ClientAppData> | undefined;
5469
+ createObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>, acceptCtx?: AcceptContext): ObservedClient<ClientAppData> | undefined;
5470
+ getOrCreateObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>, acceptCtx?: AcceptContext): ObservedClient<ClientAppData> | undefined;
5471
+ update(context?: AcceptContext): void;
5472
+ private _onClientUpdate;
5473
+ private _clientJoined;
5474
+ private _clientLeft;
5475
+ /** Emit an Observer-bus event scoped to this call. */
5476
+ private _notify;
5477
+ }
5478
+
5479
+ type Middleware<T> = (input: T, next: (nextInput: T) => void) => void;
5480
+ interface Processor<T> {
5481
+ finalCallback?: Callback<T>;
5482
+ process(value: T): void;
5483
+ addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
5484
+ removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
5485
+ }
5486
+ type Callback<T> = (input: T) => void;
5487
+ declare class MiddlewareProcessor<T> implements Processor<T> {
5488
+ private stack;
5489
+ finalCallback?: Callback<T>;
5490
+ addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
5491
+ removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
5492
+ process(value: T): void;
5493
+ }
5494
+
5495
+ /** Raised once if one receiver turns out to be dragging a publisher down for everyone. */
5496
+ declare const LOWEST_COMMON_DENOMINATOR_ISSUE = "WORST_RECEIVER_CONTAGION";
5497
+ /** The measurements behind a decided verdict — everything needed to check the call yourself. */
5498
+ type SimulcastReceiverEvidence = {
5499
+ callId: string;
5500
+ trackId: string;
5501
+ publisherClientId: string;
5502
+ worstReceiverClientId: string;
5503
+ publisherBitrate: number;
5504
+ worstReceiverBitrate: number;
5505
+ medianReceiverBitrate: number;
5506
+ /** How closely the publisher's bitrate followed the **worst** receiver's, `0..1`. */
5507
+ trackingWithWorst: number;
5508
+ /** The same against the **median** receiver — the control. */
5509
+ trackingWithMedian: number;
5510
+ };
5511
+ type SimulcastReceiverReportPayload = ({
5512
+ /** The publisher held up while one receiver lagged: layers are being chosen per consumer. */
5513
+ verdict: 'layer-decided-per-receiver';
5514
+ evidence: SimulcastReceiverEvidence;
5515
+ } | {
5516
+ /** The publisher tracked its worst receiver: everyone is getting the lowest common denominator. */
5517
+ verdict: 'layer-decided-lowest-common-denominator';
5518
+ evidence: SimulcastReceiverEvidence;
5519
+ } | {
5520
+ /** Gave up without the conditions needed to judge. **Not a pass.** */
5521
+ verdict: 'inconclusive';
5522
+ reason: string;
5523
+ }) & {
5524
+ startedAt: number;
5525
+ /** How many times the check actually ran — i.e. how much the verdict is worth. */
5526
+ checks: number;
5527
+ };
5528
+ type SimulcastReceiverValidatorConfig = {
5529
+ /**
5530
+ * Receivers a published track needs before the comparison means anything. Default `3`.
5531
+ *
5532
+ * The question is whether *one* receiver drags *the others* down, which needs at least one other to
5533
+ * be dragged — so `3` gives a worst receiver plus two to compare against. `2` is the technical
5534
+ * minimum but makes "median of the others" a single number. Sensible range `3`–`5`.
5535
+ */
5536
+ minReceivers: number;
5537
+ /**
5538
+ * How long bitrates are correlated over (ms). Default `10_000`.
5539
+ *
5540
+ * Long enough to contain a real adaptation response — the publisher's encoder reacting to a
5541
+ * bandwidth estimate takes seconds, not milliseconds. Typical `10_000`–`30_000`. Too short and you
5542
+ * catch transient jitter rather than a sustained relationship; too long and a genuine change is
5543
+ * averaged out by the healthy period around it.
5544
+ */
5545
+ windowMs: number;
5546
+ /**
5547
+ * Samples needed inside the window before it can be judged. Default `5`.
5548
+ *
5549
+ * A correlation over two or three points is meaningless. Combined with `windowMs` this implies a
5550
+ * sampling period: 5 samples in 10 s needs clients reporting at least every ~2 s. If your collector
5551
+ * is slower, widen `windowMs` rather than lowering this.
5552
+ */
5553
+ minSamples: number;
5554
+ /**
5555
+ * The worst receiver must be at most this share of the median receiver, `0`–`1`. Default `0.5`.
5556
+ *
5557
+ * The precondition, not the finding: unless somebody is genuinely doing much worse than the rest,
5558
+ * there is nothing for the publisher to be dragged *by* and the check has nothing to look at.
5559
+ * Typical `0.4`–`0.7`. Higher makes the check run more often on weaker evidence; lower means it
5560
+ * rarely finds a qualifying situation at all.
5561
+ */
5562
+ outlierRatioThreshold: number;
5563
+ /**
5564
+ * How closely the publisher must track the worst receiver to count as dragged, `0`–`1`. Default
5565
+ * `0.8`.
5566
+ *
5567
+ * This is the finding: the publisher sending at ≥80% of the *worst* receiver's rate means it has
5568
+ * collapsed to the lowest common denominator instead of serving everyone else properly. Typical
5569
+ * `0.7`–`0.9`. Toward `1` you only catch total collapse; below ~`0.6` normal encoder behaviour can
5570
+ * look like dragging.
5571
+ */
5572
+ trackingRatioThreshold: number;
5573
+ /**
5574
+ * Clean checks required before concluding per-receiver adaptation works. Default `3`.
5575
+ *
5576
+ * One clean check could be luck — the qualifying moment might simply not have been bad enough. This
5577
+ * is what stops a lucky sample from being reported as a pass, so raising it strengthens the verdict
5578
+ * at the cost of taking longer to reach one. Typical `3`–`10`.
5579
+ */
5580
+ minChecks: number;
5581
+ };
5582
+ /**
5583
+ * Answers one question: **does this SFU adapt each receiver on its own, or does one bad receiver
5584
+ * drag the publisher down for everyone?**
5585
+ *
5586
+ * That is what simulcast (or SVC) exists to prevent. With several encodings available the server can
5587
+ * hand the struggling participant a lower layer and leave everyone else alone. Without it — or with
5588
+ * a server that relays RTCP end to end instead of terminating it, so the publisher's bandwidth
5589
+ * estimate collapses to the minimum across all receivers — the only way to serve the slowest
5590
+ * participant is to make the source send less, and everybody gets the lowest common denominator.
5591
+ *
5592
+ * The two causes are worth naming because the *observation* cannot separate them: the publisher's
5593
+ * bitrate tracking its worst receiver looks identical either way. What the check establishes is
5594
+ * whether per-receiver adaptation is happening at all. If the verdict is
5595
+ * `layer-decided-lowest-common-denominator`, look at both — is simulcast/SVC actually enabled with
5596
+ * layers selected per consumer, and is the SFU terminating receiver reports rather than forwarding
5597
+ * them?
5598
+ *
5599
+ * ### The control matters more than the correlation
5600
+ *
5601
+ * "Publisher follows worst receiver" alone proves nothing: when the whole call degrades together,
5602
+ * the publisher follows *everyone*, and that is ordinary adaptation working correctly. The verdict
5603
+ * only goes against the deployment when the publisher tracks the worst receiver **more closely than
5604
+ * it tracks the median** — the worst receiver is leading, not merely coinciding.
5605
+ *
5606
+ * Likewise, a window with no outlier in it is not evidence of health, it is an untested SFU: if
5607
+ * nobody is struggling, there is nothing for per-receiver adaptation to do. Those windows are
5608
+ * skipped and never counted in `checks`.
5609
+ *
5610
+ * ### Why a validator, not a detector
5611
+ *
5612
+ * This is a property of the SFU build and configuration, not of this moment: a server doing
5613
+ * per-receiver layer selection at 09:00 still is at 17:00. Re-deriving it every tick cannot produce
5614
+ * new information — it would only keep a sliding window alive per published track for the life of
5615
+ * every call. So it decides once, reports, and releases that state.
5616
+ *
5617
+ * ```ts
5618
+ * observer.on('validation-ready', ({ validator, report }) => {
5619
+ * if (validator !== 'simulcast-receivers' || !report.ready) return;
5620
+ * console.log(report.verdict); // 'layer-decided-per-receiver' | ... | 'inconclusive'
5621
+ * });
5622
+ *
5623
+ * observer.addValidator('simulcast-receivers');
5624
+ * onDeploy(() => observer.addValidator('simulcast-receivers')); // check again
5625
+ * ```
5626
+ *
5627
+ * ### `inconclusive` is not a pass
5628
+ *
5629
+ * The check only runs when a publisher has several receivers and one of them is far behind the
5630
+ * median; plenty of healthy deployments never present that. Concluding from the absence of a failure
5631
+ * would verify nothing, so `checks` counts the times the check genuinely ran, and a validator that
5632
+ * is cancelled (or whose observer closes) finishes `inconclusive` with the reason why.
5633
+ */
5634
+ declare class SimulcastReceiverValidator implements Validator<SimulcastReceiverReportPayload> {
5635
+ private readonly _observer;
5636
+ readonly onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void;
5637
+ static readonly NAME: "simulcast-receivers";
5638
+ readonly name: "simulcast-receivers";
5639
+ readonly startedAt: number;
5640
+ report: ValidationReport<SimulcastReceiverReportPayload>;
5641
+ private readonly _config;
5642
+ private readonly _windows;
5643
+ private _checks;
5644
+ private _done;
5645
+ constructor(_observer: Observer, onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void, config?: Partial<SimulcastReceiverValidatorConfig>);
5646
+ /** How many times the comparison actually ran. `0` means nothing was established. */
5647
+ get checks(): number;
5648
+ /** Give up without a verdict, freeing anything waiting on this validator. */
5649
+ cancel(reason?: string): void;
5650
+ update(): void;
5651
+ /** Returns `true` when a verdict was reached and the caller should stop iterating. */
5652
+ private _inspect;
5653
+ /**
5654
+ * Settle on a verdict, exactly once.
5655
+ *
5656
+ * The guard is not paranoia: `onDone` removes this validator from the observer, and a second call
5657
+ * would emit a second `validation-ready` for a validator that is no longer registered — e.g. when
5658
+ * `observer.close()` cancels a validator that decided earlier in the same tick.
5659
+ */
5660
+ private _finish;
5661
+ private _windowOf;
5662
+ }
5663
+
5664
+ /** Raised once if the resolver turns out never to link anything. */
5665
+ declare const UNRESOLVED_TRACK_LINKS_ISSUE = "REMOTE_TRACK_LINKS_UNRESOLVED";
5666
+ /** What the check actually saw, whichever way it went. */
5667
+ type RemoteTrackLinkEvidence = {
5668
+ /** Calls that presented the conditions for linking: a resolver, ≥2 clients, and inbound tracks. */
5669
+ eligibleCalls: number;
5670
+ /** Inbound tracks seen across those calls. */
5671
+ inboundTracks: number;
5672
+ /** Of those, how many were linked to the outbound track that published them. */
5673
+ linkedInboundTracks: number;
5674
+ /** `linkedInboundTracks / inboundTracks`. */
5675
+ linkedRatio: number;
5676
+ /** A call that presented the conditions, for the reader to go and look at. */
5677
+ exampleCallId?: string;
5678
+ };
5679
+ type RemoteTrackResolverReportPayload = ({
5680
+ /** The resolver is linking subscribers to publishers. The detectors that need links will work. */
5681
+ verdict: 'links-resolved';
5682
+ evidence: RemoteTrackLinkEvidence;
5683
+ } | {
5684
+ /** Every condition for linking was met, repeatedly, and nothing was ever linked. */
5685
+ verdict: 'no-links-resolved';
5686
+ evidence: RemoteTrackLinkEvidence;
5687
+ } | {
5688
+ /** Never saw a call that could have been linked. **Not a pass.** */
5689
+ verdict: 'inconclusive';
5690
+ reason: string;
5691
+ }) & {
5692
+ startedAt: number;
5693
+ /** How many times the check genuinely ran — i.e. how much the verdict is worth. */
5694
+ checks: number;
5695
+ };
5696
+ type RemoteTrackResolverValidatorConfig = {
5697
+ /**
5698
+ * Participants a call needs before it can plausibly have publisher↔subscriber links. Default `2`.
5699
+ *
5700
+ * A one-person call has nobody to subscribe to anyone, so counting it would dilute the ratio with
5701
+ * calls that *could not* have produced a link. `2` is the true minimum here and there is little
5702
+ * reason to raise it.
5703
+ */
5704
+ minClients: number;
5705
+ /**
5706
+ * Inbound tracks a call must have before it counts as a check. Default `2`.
5707
+ *
5708
+ * Same idea: no subscribed tracks means nothing to link. Sensible range `2`–`5`.
5709
+ */
5710
+ minInboundTracks: number;
5711
+ /**
5712
+ * Share of inbound tracks that must be linked to conclude the resolver works, `0`–`1`. Default
5713
+ * `0.5`.
5714
+ *
5715
+ * Deliberately lenient, because a partially-linked call is normal: tracks arrive before their
5716
+ * publisher is known, and simulcast layers or probing streams may have no publisher at all. The
5717
+ * question is "is this resolver wired up", not "is every track linked". Typical `0.3`–`0.7`. Raising
5718
+ * it toward `1` turns the check into a strictness audit and it will report failure on healthy
5719
+ * systems.
5720
+ */
5721
+ linkedRatioThreshold: number;
5722
+ /**
5723
+ * Eligible calls to observe before concluding either way. Default `3`.
5724
+ *
5725
+ * One call could be a race — every track happening to arrive before its publisher. Typical `3`–`10`.
5726
+ * Note that a low value makes a *pass* less trustworthy than a failure: linking nothing repeatedly is
5727
+ * conclusive, linking things once might be luck.
5728
+ */
5729
+ minChecks: number;
5730
+ };
5731
+ /**
5732
+ * Answers one question: **is the `RemoteTrackResolver` actually linking anything?**
5733
+ *
5734
+ * ### Why this is worth a validator
5735
+ *
5736
+ * Four things in this library are built on publisher↔subscriber links —
5737
+ * `IssueFanOutDetector`, `TrackDeliveryMismatchDetector`, `UnconsumedTrackDetector` and
5738
+ * `SimulcastReceiverValidator`. Every one of them checks `call.remoteTrackResolver` and, finding no
5739
+ * links, correctly does nothing rather than guessing.
5740
+ *
5741
+ * That is the right behaviour and it produces a nasty failure mode: a resolver wired to the wrong id
5742
+ * field, or a mediasoup `producerId` the application never attaches, leaves all four permanently
5743
+ * silent — and **silence is what a healthy deployment looks like too**. You would conclude your
5744
+ * calls were clean when in fact nothing was ever examined. This check exists to make that specific
5745
+ * mistake loud.
5746
+ *
5747
+ * ### `inconclusive` is not a pass
5748
+ *
5749
+ * A verdict is only reached from calls that *could* have been linked: a resolver configured, at
5750
+ * least `minClients` participants, and at least `minInboundTracks` inbound tracks present. A
5751
+ * one-to-one deployment, a lobby full of audio-only listeners, or a quiet period never presents
5752
+ * those conditions — and concluding "resolver works" from calls that had nothing to resolve would be
5753
+ * the very mistake this validator is here to catch. `checks` counts the eligible calls actually
5754
+ * seen; a validator cancelled before reaching `minChecks` finishes `inconclusive` and says so.
5755
+ *
5756
+ * ```ts
5757
+ * observer.on('validation-ready', ({ validator, report }) => {
5758
+ * if (validator !== 'remote-track-resolver' || !report.ready) return;
5759
+ * if (report.verdict === 'no-links-resolved') alert('resolver misconfigured — 4 detectors are inert');
5760
+ * });
5761
+ *
5762
+ * observer.addValidator('remote-track-resolver');
5763
+ * ```
5764
+ *
5765
+ * Run it once at start-up, and again after changing the resolver or the SFU's id scheme. Like every
5766
+ * validator it is one-shot: the answer is a property of the wiring, not of this moment.
5767
+ */
5768
+ declare class RemoteTrackResolverValidator implements Validator<RemoteTrackResolverReportPayload> {
5769
+ private readonly _observer;
5770
+ readonly onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void;
5771
+ static readonly NAME: "remote-track-resolver";
5772
+ readonly name: "remote-track-resolver";
5773
+ readonly startedAt: number;
5774
+ report: ValidationReport<RemoteTrackResolverReportPayload>;
5775
+ private readonly _config;
5776
+ /** Accumulated across every eligible call seen, so one small call cannot decide alone. */
5777
+ private _eligibleCalls;
5778
+ private _inboundTracks;
5779
+ private _linkedInboundTracks;
5780
+ private _exampleCallId?;
5781
+ private _checks;
5782
+ private _done;
5783
+ constructor(_observer: Observer, onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void, config?: Partial<RemoteTrackResolverValidatorConfig>);
5784
+ /** How many eligible calls were actually examined. `0` means nothing was established. */
5785
+ get checks(): number;
5786
+ cancel(reason?: string): void;
5787
+ update(): void;
5788
+ private _evidence;
5789
+ private _finish;
5790
+ }
5791
+
5792
+ /** Raised once if the deployment is not actually delivering the codec it thinks it is. */
5793
+ declare const CODEC_MISMATCH_ISSUE = "CODEC_INCONSISTENCY";
5794
+ /** What the check saw across a call's participants. */
5795
+ type CodecEvidence = {
5796
+ callId: string;
5797
+ kind: 'audio' | 'video';
5798
+ /** Every mime type in use in that call, most common first — e.g. `[ 'video/VP8', 'video/H264' ]`. */
5799
+ mimeTypes: string[];
5800
+ /** How many clients used each, in the same order as {@link mimeTypes}. */
5801
+ clientsPerMimeType: number[];
5802
+ /** Clients considered — those that reported at least one codec of this kind. */
5803
+ clients: number;
5804
+ /** The codec the check was told to expect, when it was given one. */
5805
+ expected?: string;
5806
+ };
5807
+ type CodecConsistencyReportPayload = ({
5808
+ /** One codec per media kind, and it is the expected one if an expectation was given. */
5809
+ verdict: 'codec-consistent';
5810
+ evidence: CodecEvidence[];
5811
+ } | {
5812
+ /** Participants of one call are split across different codecs. */
5813
+ verdict: 'codec-split';
5814
+ evidence: CodecEvidence[];
5815
+ } | {
5816
+ /** Consistent, but not what the deployment believes it negotiated. */
5817
+ verdict: 'unexpected-codec';
5818
+ evidence: CodecEvidence[];
5819
+ } | {
5820
+ /** Never saw a call with enough participants reporting codecs. **Not a pass.** */
5821
+ verdict: 'inconclusive';
5822
+ reason: string;
5823
+ }) & {
5824
+ startedAt: number;
5825
+ checks: number;
5826
+ };
5827
+ type CodecConsistencyValidatorConfig = {
5828
+ /**
5829
+ * The mime type you believe you are delivering, per kind — e.g.
5830
+ * `{ video: 'video/VP8', audio: 'audio/opus' }`.
5831
+ *
5832
+ * Optional. Without it the check still reports a *split* (participants disagreeing with each
5833
+ * other), which needs no expectation to be a fact. With it, the check can additionally catch the
5834
+ * case where everyone agrees on the wrong thing — a silent fallback that nothing else notices.
5835
+ */
5836
+ expected?: Partial<Record<'audio' | 'video', string>>;
5837
+ /**
5838
+ * Which kinds to inspect. Default `[ 'audio', 'video' ]`.
5839
+ *
5840
+ * Narrow it when only one matters: audio codec splits are the ones that usually cost transcoding,
5841
+ * while video splits are more often a deliberate per-client decision.
5842
+ */
5843
+ kinds: ('audio' | 'video')[];
5844
+ /**
5845
+ * Participants a call needs before disagreement is meaningful. Default `3`.
5846
+ *
5847
+ * In a 1:1 call "the participants disagree" is two clients differing, which can be a legitimate
5848
+ * negotiation outcome rather than a fault. Sensible range `3`–`5`.
5849
+ */
5850
+ minClients: number;
5851
+ /**
5852
+ * Calls to inspect before concluding. Default `3`.
5853
+ *
5854
+ * A structural property of your negotiation, so a handful of calls is plenty — but one call could be
5855
+ * an unusual mix of participants. Typical `3`–`10`. Higher delays the verdict without adding much,
5856
+ * since the answer does not vary call to call.
5857
+ */
5858
+ minChecks: number;
5859
+ };
5860
+ /**
5861
+ * Answers: **is every participant of a call actually using the same codec — and is it the one you
5862
+ * think you negotiated?**
5863
+ *
5864
+ * ### Why the server has to answer this
5865
+ *
5866
+ * A client knows only its own codec. It cannot tell whether it is the odd one out, and an SFU that
5867
+ * forwards without transcoding cannot serve a call where participants disagree — so a split is a
5868
+ * real fault with a very confusing symptom: some pairs of participants see each other and some do
5869
+ * not, with no error anywhere. Only something holding every participant of a call at once can see
5870
+ * the split at all.
5871
+ *
5872
+ * The second half is the quieter failure. A deployment configured for VP9 or AV1 will fall back to
5873
+ * VP8 whenever one endpoint cannot negotiate the preferred codec, and nothing reports that — the
5874
+ * call works, the bitrate is higher than it should be, and the team believes it shipped AV1 months
5875
+ * ago. Give the check an `expected` mime type and it will say so.
5876
+ *
5877
+ * ### Why a validator and not a detector
5878
+ *
5879
+ * The answer is a property of the deployment — SDP munging, codec preferences, the SFU build — not
5880
+ * of this moment. A deployment that negotiates VP8 at 09:00 negotiates VP8 at 17:00. Re-deriving it
5881
+ * every tick would walk every codec of every peer connection of every call, forever, to re-learn a
5882
+ * constant. So it decides once and stops.
5883
+ *
5884
+ * Start it again after a deploy, or after changing codec preferences:
5885
+ *
5886
+ * ```ts
5887
+ * observer.addValidator('codec-consistency', {
5888
+ * expected: { video: 'video/VP9', audio: 'audio/opus' },
5889
+ * });
5890
+ *
5891
+ * observer.on('validation-ready', ({ validator, report }) => {
5892
+ * if (validator !== 'codec-consistency' || !report.ready) return;
5893
+ * // 'codec-consistent' | 'codec-split' | 'unexpected-codec' | 'inconclusive'
5894
+ * });
5895
+ * ```
5896
+ *
5897
+ * ### `inconclusive` is not a pass
5898
+ *
5899
+ * Only calls with at least `minClients` participants *reporting codecs of that kind* count as a
5900
+ * check. An audio-only deployment will never say anything about video, and concluding "video codecs
5901
+ * are consistent" from calls that carried no video would verify nothing.
5902
+ *
5903
+ * ### Comparison is by mime type only
5904
+ *
5905
+ * `sdpFmtpLine` carries profile and level — `profile-level-id` for H.264, `profile-id` for VP9 — and
5906
+ * two clients on the same mime type with different profiles are not truly interchangeable. That is
5907
+ * deliberately out of scope: fmtp differences are common, usually benign, and would make this check
5908
+ * noisy enough to ignore. It answers the coarse question, which is the one that is actually wrong in
5909
+ * practice.
5910
+ */
5911
+ declare class CodecConsistencyValidator implements Validator<CodecConsistencyReportPayload> {
5912
+ private readonly _observer;
5913
+ readonly onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void;
5914
+ static readonly NAME: "codec-consistency";
5915
+ readonly name: "codec-consistency";
5916
+ readonly startedAt: number;
5917
+ report: ValidationReport<CodecConsistencyReportPayload>;
5918
+ private readonly _config;
5919
+ private readonly _inspectedCallIds;
5920
+ private _evidence;
5921
+ private _checks;
5922
+ private _done;
5923
+ constructor(_observer: Observer, onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void, config?: Partial<CodecConsistencyValidatorConfig>);
5924
+ get checks(): number;
5925
+ cancel(reason?: string): void;
5926
+ update(): void;
5927
+ /** One evidence entry per configured kind that the call actually carried. */
5928
+ private _tallyOf;
5929
+ private _finish;
5930
+ }
5931
+
5932
+ /**
5933
+ * The validators `observer.addValidator(name, config)` knows how to build, and the config each takes.
5934
+ *
5935
+ * Adding one means: write the class with a `static readonly NAME`, add its entry here, and add a
5936
+ * `case` to `addValidator`. The map is what gives the call site its types —
5937
+ * `addValidator('simulcast-receivers', { … })` type-checks the config against the right validator,
5938
+ * and an unknown name won't compile.
5939
+ */
5940
+ type AvailableValidatorConfigs = {
5941
+ [SimulcastReceiverValidator.NAME]: SimulcastReceiverValidatorConfig;
5942
+ [RemoteTrackResolverValidator.NAME]: RemoteTrackResolverValidatorConfig;
5943
+ [CodecConsistencyValidator.NAME]: CodecConsistencyValidatorConfig;
5944
+ };
5945
+ /** A validator name that can be started. */
5946
+ type ValidatorName = keyof AvailableValidatorConfigs;
5947
+
5948
+ /**
5949
+ * Keeps every configured `CallSummary` up to date, from **observer-level** bus subscriptions.
5950
+ *
5951
+ * ### Why one collector and not one per call
5952
+ *
5953
+ * The obvious implementation subscribes each call's summary to the events it needs. But the bus is
5954
+ * observer-wide: a listener attached for call A is invoked for every event of every call, so that
5955
+ * design costs `calls × events` listeners *and* `calls` invocations per event — quadratic in the
5956
+ * thing most likely to be large. At 500 concurrent calls and eight subscribed events that is 4 000
5957
+ * listeners doing 500 no-op calls each, per event.
5958
+ *
5959
+ * So the collector attaches **one listener per event type, once**, and routes each event to the
5960
+ * summary of the call it names. Cost is O(subscribed event types), independent of how many calls are
5961
+ * in flight, and an event for a call with no summary costs one `undefined` check.
5962
+ *
5963
+ * ### Only call-scoped events
5964
+ *
5965
+ * Routing needs `observedCall` on the payload, which is exactly what `CallScopedEventName` selects.
5966
+ * Observer-scoped events have no single call to attribute to; see that type for why fanning them out
5967
+ * to every open summary would be worse than refusing.
5968
+ */
5969
+ declare class CallSummaryCollector {
5970
+ private readonly _observer;
5971
+ private readonly _config;
5972
+ private readonly _scratch;
5973
+ private readonly _listeners;
5974
+ private _closed;
5975
+ constructor(_observer: Observer, _config: CallSummaryConfig);
5976
+ /**
5977
+ * Build a summary for `callId` and start tracking it.
5978
+ *
5979
+ * Creating it here, rather than letting the call create one and hand it over, keeps the resolved
5980
+ * configuration inside the single object that owns it — and makes it impossible to end up with a
5981
+ * summary whose sections nobody subscribed to fill.
5982
+ */
5983
+ createSummary(callId: string): CallSummary;
5984
+ /**
5985
+ * Finalise `call`'s summary: fold in what only makes sense once, and stamp the closing times.
5986
+ *
5987
+ * Percentiles are computed here rather than on every update — a median recomputed per tick over a
5988
+ * growing array is quadratic work to produce a number nobody reads until the end.
5989
+ */
5990
+ finalise(call: ObservedCall): void;
5991
+ /** Drop every bus subscription. Called when the observer closes. */
5992
+ close(): void;
5993
+ /**
5994
+ * Subscribe `listener` to `event`, routed to the summary of the call the event names.
5995
+ *
5996
+ * The `observedCall` is read off the payload rather than closed over, which is what lets one
5997
+ * subscription serve every call.
5998
+ */
5999
+ private _on;
6000
+ private _subscribeBuiltIns;
6001
+ private _subscribeEnrichers;
6002
+ }
6003
+
6004
+ type SampleRejectedReason = 'observer-closed' | 'missing-callId' | 'missing-clientId';
6005
+
6006
+ /**
6007
+ * Optional, free-form context supplied to `accept()`. A single context object is
6008
+ * threaded down the accept chain (Observer -> Client -> PeerConnection) and is
6009
+ * merged into the `appData` of entities created during the accept pass.
6010
+ */
6011
+ type AcceptContext = Record<string, unknown>;
6012
+ /** The payload threaded through `accept()` middlewares: the sample and its optional context. */
6013
+ type AcceptMiddlewarePayload = {
6014
+ sample: ClientSample;
6015
+ context?: AcceptContext;
6016
+ };
6017
+ /**
6018
+ * A global middleware run on every sample passed to `observer.accept()`, in registration order,
6019
+ * **before** the sample is dispatched to any call/client. It may inspect or mutate the sample
6020
+ * (e.g. set/normalize `callId`/`clientId`, enrich, redact) or the context, then call
6021
+ * `next(payload)` to continue the chain. Not calling `next` **drops** the sample.
6022
+ */
6023
+ type AcceptMiddleware = Middleware<AcceptMiddlewarePayload>;
6024
+ /** Produces the initial `appData` for a call created without an explicit `appData`. */
6025
+ type CallAppDataFactory = (params: {
6026
+ callId: string;
6027
+ observer: Observer;
6028
+ acceptCtx?: AcceptContext;
6029
+ }) => Record<string, unknown>;
6030
+ /** Produces the initial `appData` for a client created without an explicit `appData`. */
6031
+ type ClientAppDataFactory = (params: {
6032
+ clientId: string;
6033
+ observedCall: ObservedCall;
6034
+ acceptCtx?: AcceptContext;
6035
+ }) => Record<string, unknown>;
6036
+ type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = {
6037
+ /**
6038
+ * Application-owned data for the observer itself. Never read or modified by the library.
6039
+ *
6040
+ * For per-call / per-client data prefer `createCallAppData` / `createClientAppData`, which run at
6041
+ * creation time and can see the `accept()` context.
6042
+ */
6043
+ appData?: AppData;
6044
+ /**
6045
+ * Close a client that has not produced a sample for this long (ms). Default `60_000`.
6046
+ *
6047
+ * This is the **liveness timeout for a participant**, so set it from your client's sampling
6048
+ * period, not from taste: a client sampling every 5 s needs several missed samples to look gone.
6049
+ * Sensible range `3x`–`10x` the sampling period; `60_000` suits the usual 2–10 s collectors.
6050
+ *
6051
+ * Too low and a client that merely paused (tab backgrounded, brief network drop) is closed and
6052
+ * then re-created as a *new* client, which restarts its detectors and splits one participant into
6053
+ * two in any summary. Too high and left participants linger, inflating `peak`, the denominators of
6054
+ * every ratio-based detector, and memory. `undefined` disables the timeout — then nothing closes
6055
+ * an abandoned client but your own `client.close()`.
6056
+ */
6057
+ closeClientIfIdleForMs?: number;
6058
+ /**
6059
+ * Close a call once it has had zero clients for this long (ms). Default `60_000`.
6060
+ *
6061
+ * The grace period exists so a brief empty moment — everyone reconnecting after a network blip,
6062
+ * the last participant refreshing — does not end the call and start a new one under the same
6063
+ * `callId`. Sensible range `10_000`–`300_000`.
6064
+ *
6065
+ * Too low splits one meeting into several calls, and each split emits its own `call-summary`. Too
6066
+ * high keeps dead calls in `observedCalls`, holding their detectors and summaries in memory.
6067
+ * `undefined` disables it: the call then lives until you call `call.close()`.
6068
+ */
6069
+ closeCallIfEmptyForMs?: number;
6070
+ /**
6071
+ * When `true` (the default), every call update triggers an observer-wide `update()` pass.
6072
+ *
6073
+ * There is deliberately no timer and no separate policy object: a call is updated when any of its
6074
+ * clients is, and the observer is updated when any of its calls is — so the observer is updated
6075
+ * exactly when any client anywhere is. Set to `false` only if you drive `observer.update()`
6076
+ * yourself, and note that observer-scoped detectors and validators run *nowhere else*.
6077
+ */
6078
+ autoUpdateOnCallUpdate?: boolean;
6079
+ /**
6080
+ * Accumulate a {@link CallSummary} on every call this observer creates.
6081
+ *
6082
+ * **Absent or `null` means no summaries at all** — no accumulation, and not one bus subscription.
6083
+ * Pass an object (`{}` is valid) to switch it on; anything you leave out takes its default from
6084
+ * `defaultCallSummaryConfig`, including `include: []`, which collects *no* built-in section. A
6085
+ * summary that only runs `enrich` is a perfectly good summary.
6086
+ *
6087
+ * ```ts
6088
+ * const observer = new Observer({
6089
+ * callSummary: {
6090
+ * include: [ 'clients', 'issues' ],
6091
+ * enrich: {
6092
+ * 'client-joined': (summary, { observedClient }) => {
6093
+ * ((summary.attachments.regions ??= []) as string[]).push(String(observedClient.appData.region));
6094
+ * },
6095
+ * },
6096
+ * },
6097
+ * });
6098
+ *
6099
+ * observer.on('call-summary', ({ summary }) => archive(summary));
6100
+ * ```
6101
+ *
6102
+ * This is construction-time and fixed for the observer's life, unlike detectors, which are added
6103
+ * per call as an application decides what to watch. A summary is a record of what happened, and a
6104
+ * record you can turn on halfway through is a record with a hole in it — calls that started
6105
+ * earlier would carry different sections from calls that started later, with nothing on either to
6106
+ * say which. One shape for every call, or none.
6107
+ */
6108
+ callSummary?: Partial<CallSummaryConfig> | null;
6109
+ /**
6110
+ * Thresholds that mark a **received** track as degraded. Omitted (the default) means no track is
6111
+ * ever marked degraded — `inboundTrack.degraded` stays `false` and `degradationReasons` stays
6112
+ * empty, so this is opt-in and absence is *not* a clean bill of health.
6113
+ *
6114
+ * Each field is an **exclusive upper bound**: the reason is added when the measured value is
6115
+ * strictly greater. All are evaluated on every update and any number can fire at once;
6116
+ * `degradationReasons` lists the ones that did.
6117
+ *
6118
+ * These feed `CallHealthAggregator` and `outboundTrack.degradedRatio` (how many of a publisher's
6119
+ * receivers are unhappy), which is what `TrackDeliveryMismatchDetector` and
6120
+ * `PublisherFaultCorroborationDetector` read. They are **not** a substitute for client issues:
6121
+ * client-monitor already decides "this endpoint is in trouble" with hysteresis and multi-signal
6122
+ * confirmation. Treat these as a coarse per-track flag for cross-participant comparison, and keep
6123
+ * them loose enough that a single bad tick does not trip them.
6124
+ */
6125
+ inboundTrackDegradationThresholds?: {
6126
+ /**
6127
+ * Freezes counted **in one sampling period**, not since the start of the call. `1` means "any
6128
+ * freeze at all this tick", which is strict; `2`–`3` tolerates the odd frame hiccup.
6129
+ * Reason: `'freezes'`.
6130
+ */
6131
+ deltaFreezeCount: number;
6132
+ /**
6133
+ * Fraction of frames dropped, `0`–`1`. Typical `0.05`–`0.2`; below ~`0.02` you are inside
6134
+ * normal jitter for most decoders. Reason: `'frames-dropped'`.
6135
+ */
6136
+ framesDroppedRatio: number;
6137
+ /**
6138
+ * Jitter buffer delay in **ms**. Typical `150`–`500`: audio stays intelligible well past
6139
+ * `200`, while conversation turn-taking suffers beyond ~`400`. Reason:
6140
+ * `'jitter-buffer-delay'`.
6141
+ */
6142
+ jitterBufferDelayInMs: number;
6143
+ /**
6144
+ * Concealed (synthesised) audio samples as a fraction, `0`–`1`. Typical `0.05`–`0.15`; above
6145
+ * ~`0.1` is usually audible as robotic or clipped speech. Reason: `'concealment'`.
6146
+ */
6147
+ concealmentRatio: number;
6148
+ /**
6149
+ * Round-trip time in **ms**, as reported for this inbound stream. Typical `250`–`500`.
6150
+ * Remember this is absolute, not relative to the participant's own baseline — a genuinely
6151
+ * distant participant will sit permanently above any fixed bound, which is why "RTT got
6152
+ * worse" belongs to client-monitor and not here. Reason: `'rtt'`.
6153
+ */
6154
+ rttInMs: number;
6155
+ };
6156
+ /**
6157
+ * Thresholds that mark a **published** track as degraded, from what the receivers report back.
6158
+ * Omitted (the default) means neither threshold-based reason can fire.
6159
+ *
6160
+ * Note two reasons are added regardless of this setting, because they need no threshold:
6161
+ * `quality-limited-<reason>` when the encoder reports a `qualityLimitationReason`, and
6162
+ * `no-packets-sent` when an unmuted track sent nothing in a sampling period.
6163
+ *
6164
+ * Also an **exclusive upper bound** per field, evaluated on every update.
6165
+ */
6166
+ outboundTrackDegradationThresholds?: {
6167
+ /**
6168
+ * Fraction of packets lost as reported by the remote end, `0`–`1`. Typical `0.02`–`0.1`;
6169
+ * anything under ~`0.01` is normal on the open internet and will fire constantly. Reason:
6170
+ * `'remote-fraction-lost'`.
6171
+ */
6172
+ fractionLost: number;
6173
+ /**
6174
+ * Round-trip time in **ms** as reported by the remote end. Typical `250`–`500`, and absolute
6175
+ * rather than baseline-relative — see the note on the inbound `rttInMs`. Reason:
6176
+ * `'remote-rtt'`.
6177
+ */
6178
+ rttInMs: number;
6179
+ };
6180
+ /**
6181
+ * Optional factory invoked when a call is created without an explicit `appData`
6182
+ * (e.g. lazily by `accept()`), so apps can enrich appData without pre-creating the
6183
+ * entity. `appData` is application-owned; it is never modified by the `accept()` context.
6184
+ */
6185
+ createCallAppData?: CallAppDataFactory;
6186
+ /** Same as `createCallAppData`, for clients. Receives the (already-created) parent call. */
6187
+ createClientAppData?: ClientAppDataFactory;
6188
+ /**
6189
+ * Optional factory invoked when a client is created, producing a per-client sink that
6190
+ * receives every sample the client accepts (or `undefined` for no sink). The destination
6191
+ * can be derived from `callId` / `clientId`.
6192
+ */
6193
+ createClientSink?: ClientSampleSinkFactory;
6194
+ /**
6195
+ * Optional factory invoked when a call is created, producing the call's `RemoteTrackResolver`
6196
+ * (or `undefined` for none). Use the built-ins
6197
+ * (`createDefaultMediasoupRemoteTrackResolverFactory()` / `createP2pRemoteTrackResolverFactory()`)
6198
+ * or build a `RemoteTrackResolver` with custom publisher/subscriber id resolvers.
6199
+ */
6200
+ createRemoteTrackResolver?: RemoteTrackResolverFactory;
6201
+ };
6202
+ declare interface Observer {
6203
+ on<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
6204
+ off<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
6205
+ once<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
6206
+ emit<U extends keyof ObserverEvents>(event: U, ...args: ObserverEvents[U]): boolean;
6207
+ }
6208
+ declare class Observer<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
2817
6209
  readonly observedTURN: ObservedTURN;
2818
6210
  readonly observedCalls: Map<string, ObservedCall<Record<string, unknown>>>;
2819
6211
  readonly observedMediasoupRouters: Map<string, ObservedMediasoupRouter<Record<string, unknown>>>;
2820
- updater?: Updater;
2821
6212
  /** Ancestry base shared by all Observer-bus events originating at the observer. */
2822
6213
  readonly eventScope: ObserverEventBase;
6214
+ readonly config: ObserverConfig<AppData>;
2823
6215
  closed: boolean;
2824
6216
  totalAddedCall: number;
2825
6217
  totalRemovedCall: number;
@@ -2831,19 +6223,151 @@ declare class Observer<AppData extends Record<string, unknown> = Record<string,
2831
6223
  numberOfPeerConnections: number;
2832
6224
  /** Global, pre-dispatch middleware chain run on every accepted sample. */
2833
6225
  readonly acceptMiddlewares: MiddlewareProcessor<AcceptMiddlewarePayload>;
2834
- constructor(config?: ObserverConfig<AppData>);
6226
+ /**
6227
+ * Fleet-wide index of every open client issue, maintained incrementally as issues open and close.
6228
+ * Each call's index propagates into this one, so cross-call queries cost O(matching issues) rather
6229
+ * than a walk over every call and client. This is what observer-scoped detectors read.
6230
+ */
6231
+ readonly activeIssuesRegistry: ActiveIssuesRegistry;
6232
+ /**
6233
+ * Validators currently running. Each removes itself when it finishes, so this is normally empty —
6234
+ * a validator is a one-shot check, not a permanent fixture. Start one with {@link addValidator}.
6235
+ */
6236
+ readonly validators: Set<RunningValidator>;
6237
+ /**
6238
+ * Observer-scoped detector registry, run on every `observer.update()` — the place for findings
6239
+ * that span **calls**, e.g. "many calls on the same SFU degraded at once". Detectors raise
6240
+ * findings with `observer.addIssue(...)`, surfaced on the bus as `observer-issue`.
6241
+ * (For findings within a single call use `observedCall.detectors`.)
6242
+ *
6243
+ * Starts **empty**. Populate it with {@link addObserverDetector}, or `detectors.add(...)` for an
6244
+ * instance you built yourself.
6245
+ */
6246
+ readonly detectors: Detectors;
6247
+ /**
6248
+ * The call-scoped detectors to build on **every** call this observer creates, in registration
6249
+ * order. Written by {@link addCallDetector}; read by `createObservedCall`.
6250
+ *
6251
+ * Nothing is created implicitly. There is no detector configuration in `ObserverConfig` and no
6252
+ * default set, because a detector that nobody asked for is a detector nobody will act on: it costs
6253
+ * time on every tick and raises findings into a handler that was not written to expect them. An
6254
+ * application says what it wants to watch, or it watches nothing.
6255
+ *
6256
+ * ```ts
6257
+ * observer.addCallDetector('call-concurrent-issue-detector', {
6258
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
6259
+ * });
6260
+ * ```
6261
+ */
6262
+ readonly callDetectorConfigs: Map<keyof AvailableCallScopeDetectorsConfigs, Partial<UnconsumedTrackDetectorConfig | TrackDeliveryMismatchDetectorConfig | CallConcurrentIssueDetectorConfig | IssueFanOutDetectorConfig | PublisherFaultCorroborationDetectorConfig>>;
6263
+ /**
6264
+ * Owns every call's summary: the resolved `config.callSummary`, the bus subscriptions that keep
6265
+ * the summaries current (one per event type, not one per call), and the summaries themselves.
6266
+ *
6267
+ * `undefined` when `config.callSummary` was absent or `null` — so its presence *is* the answer to
6268
+ * "are summaries on", and nothing is subscribed to anything.
6269
+ */
6270
+ readonly callSummaryCollector?: CallSummaryCollector;
6271
+ constructor(config?: Partial<ObserverConfig<AppData>>);
2835
6272
  get numberOfCalls(): number;
2836
6273
  get appData(): AppData | undefined;
6274
+ /**
6275
+ * Build a cross-call detector onto `observer.detectors`. Chainable.
6276
+ *
6277
+ * To get a handle on what was built — to inspect it, or to remove that exact instance later — read
6278
+ * it back off the registry: `observer.detectors.getAll(name)`, or `observer.detectors.instances`.
6279
+ */
6280
+ addObserverDetector<K extends keyof AvailableObserverScopeDetectorsConfigs>(name: K, config?: Partial<AvailableObserverScopeDetectorsConfigs[K]>): this;
6281
+ /**
6282
+ * Enable a call-scoped detector for calls created **from now on**.
6283
+ *
6284
+ * This edits the config, not the live calls: calls already open keep the detector set they were
6285
+ * built with. To add one to an existing call, use `observedCall.addDetector(...)` directly.
6286
+ */
6287
+ addCallDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
6288
+ /**
6289
+ * Remove an observer-scoped detector by name, returning how many were removed.
6290
+ *
6291
+ * **Every** instance registered under the name goes, since a name can legitimately be registered
6292
+ * more than once (`ClientPopulationIssueDetector` is meant to be added once per `groupBy` axis).
6293
+ * When you want one of them specifically, go through the registry, which deals in instances:
6294
+ *
6295
+ * ```ts
6296
+ * const [ byBrowser, byOs ] = observer.detectors.getAll('client-population-issue-detector');
6297
+ *
6298
+ * observer.detectors.remove(byOs); // keeps the browser axis running
6299
+ * ```
6300
+ *
6301
+ * Either route `close()`s the detector, so it unsubscribes from the issue registry and drops any
6302
+ * timers or bus listeners it held.
6303
+ */
6304
+ removeObserverDetector(name: keyof AvailableObserverScopeDetectorsConfigs): number;
6305
+ /**
6306
+ * Stop building `name` on calls created from now on.
6307
+ *
6308
+ * By default this also removes it from the calls **already open**, so that "remove this detector"
6309
+ * means the same thing whether you say it before or after a call started — the alternative leaves
6310
+ * a fleet where the detector is live on some calls and not others, decided by join time. Pass
6311
+ * `{ includeOpenCalls: false }` to change only what future calls are built with.
6312
+ *
6313
+ * Returns the number of live detector instances removed (`0` when only the config changed).
6314
+ */
6315
+ removeCallDetector(name: keyof AvailableCallScopeDetectorsConfigs, { includeOpenCalls }?: {
6316
+ includeOpenCalls?: boolean;
6317
+ }): number;
6318
+ /**
6319
+ * Start a structural check. It runs on each `observer.update()` until it can decide, reports once
6320
+ * on `validation-ready`, and removes itself.
6321
+ *
6322
+ * ```ts
6323
+ * observer.validate('simulcast-receiver-validator', { minChecks: 5 });
6324
+ * ```
6325
+ *
6326
+ * Config keys are optional and merged over that validator's defaults. Call it again — after a
6327
+ * deploy, say — to check again; there is no revalidation timer, because a deploy rather than
6328
+ * elapsed time is what makes a structural verdict stale.
6329
+ */
6330
+ addValidator<K extends keyof AvailableValidatorConfigs>(name: K, config?: Partial<AvailableValidatorConfigs[K]>): this;
6331
+ /**
6332
+ * Stop a running validation, by name or by instance. Returns how many were cancelled.
6333
+ *
6334
+ * Cancelling is **not** silent discarding. The validator finishes with `inconclusive` and the given
6335
+ * `reason`, emits `validation-ready` like any other completion, and removes itself. That matters
6336
+ * because anything waiting on the verdict — a deploy gate, a dashboard, a promise — would otherwise
6337
+ * wait forever, and because "we stopped asking" is a materially different outcome from "we asked
6338
+ * and learned nothing", which is exactly what `inconclusive` with a reason records.
6339
+ *
6340
+ * ```ts
6341
+ * observer.cancelValidator('simulcast-receivers', 'sfu redeployed');
6342
+ *
6343
+ * // or one specific instance — `observer.validators` holds what is running
6344
+ * for (const validator of observer.validators) observer.cancelValidator(validator, 'shutting down');
6345
+ * ```
6346
+ *
6347
+ * Pass a real reason. The default tells the reader nothing they could not already infer.
6348
+ */
6349
+ cancelValidator(target: keyof AvailableValidatorConfigs | RunningValidator, reason?: string): number;
2837
6350
  getObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(callId: string): ObservedCall<T> | undefined;
2838
- createObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
2839
- getOrCreateObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
6351
+ /**
6352
+ * @param acceptCtx the `accept()` context, when this call is being created to receive a sample.
6353
+ * Passed on to `ObserverConfig.createCallAppData`, so the factory can read whatever the caller (or
6354
+ * an accept middleware) put there — a tenant, a region, a trace id.
6355
+ */
6356
+ createObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>, acceptCtx?: AcceptContext): ObservedCall<T> | undefined;
6357
+ getOrCreateObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>, acceptCtx?: AcceptContext): ObservedCall<T> | undefined;
2840
6358
  createObservedMediasoupRouter<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedMediasoupRouterSettings<T> & {
2841
- callId?: string;
2842
- bindCallByWebRtcTransportId?: boolean;
6359
+ matchPeerConnectionByWebRtcTransportId?: boolean;
2843
6360
  }): ObservedMediasoupRouter<Record<string, unknown>> | undefined;
2844
6361
  close(): void;
2845
6362
  accept(sample: ClientSample, context?: AcceptContext): void;
2846
6363
  update(): void;
6364
+ /**
6365
+ * Raise an observer-scoped (cross-call / SFU-wide) finding. Emitted on the bus as
6366
+ * `observer-issue`. Intended for `observer.detectors`, but the application may call it too.
6367
+ *
6368
+ * `payload` takes an **object**; see `ObserverIssue`.
6369
+ */
6370
+ addIssue(issue: Omit<ObserverIssue, 'scope'>): void;
2847
6371
  /** Emit an Observer-bus event. */
2848
6372
  private _notify;
2849
6373
  }
@@ -2882,6 +6406,286 @@ declare enum ClientEventTypes {
2882
6406
  DATA_CONSUMER_CLOSED = "DATA_CONSUMER_CLOSED"
2883
6407
  }
2884
6408
 
6409
+ /** Thresholds deciding when a client counts as degraded on the receiving / sending side. */
6410
+ type ClientHealthThresholds = {
6411
+ /** Inbound loss fraction across the client's received streams (0..1). */
6412
+ inboundFractionLost: number;
6413
+ /** Loss fraction reported back about the client's sent streams via RTCP (0..1). */
6414
+ outboundFractionLost: number;
6415
+ /** Round-trip time (ms). */
6416
+ rttInMs: number;
6417
+ /** Freezes observed across the client's inbound video in the tick. */
6418
+ freezeCount: number;
6419
+ /** Concealment fraction across the client's inbound audio (0..1). */
6420
+ concealmentRatio: number;
6421
+ };
6422
+ declare const defaultClientHealthThresholds: ClientHealthThresholds;
6423
+ /** The per-client health view, split by direction. */
6424
+ type ClientHealth = {
6425
+ observedClient: ObservedClient;
6426
+ clientId: string;
6427
+ /** Receiving (download) side is impaired. */
6428
+ inboundDegraded: boolean;
6429
+ /** Sending (upload) side is impaired. */
6430
+ outboundDegraded: boolean;
6431
+ /** `inboundDegraded || outboundDegraded`. */
6432
+ degraded: boolean;
6433
+ reasons: string[];
6434
+ inboundFractionLost?: number;
6435
+ outboundFractionLost?: number;
6436
+ rttInMs?: number;
6437
+ deltaFreezeCount: number;
6438
+ concealmentRatio?: number;
6439
+ /** Quality-limitation reasons seen on this client's outbound video ('cpu' | 'bandwidth' | …). */
6440
+ qualityLimitationReasons: string[];
6441
+ usingTURN: boolean;
6442
+ usingTCP: boolean;
6443
+ };
6444
+ /** Call-level rollup of the per-client health, using percentiles rather than means. */
6445
+ type CallHealth = {
6446
+ callId: string;
6447
+ clients: ClientHealth[];
6448
+ numberOfClients: number;
6449
+ numberOfDegradedClients: number;
6450
+ numberOfInboundDegradedClients: number;
6451
+ numberOfOutboundDegradedClients: number;
6452
+ /** degradedClients / clients (0..1). */
6453
+ degradedRatio: number;
6454
+ inboundDegradedRatio: number;
6455
+ outboundDegradedRatio: number;
6456
+ rttInMs?: StatsSummary;
6457
+ inboundFractionLost?: StatsSummary;
6458
+ concealmentRatio?: StatsSummary;
6459
+ /** How many clients reported each quality-limitation reason on their outbound video. */
6460
+ qualityLimitation: {
6461
+ cpu: number;
6462
+ bandwidth: number;
6463
+ other: number;
6464
+ };
6465
+ freezes: {
6466
+ affectedClients: number;
6467
+ total: number;
6468
+ };
6469
+ };
6470
+ /**
6471
+ * Aggregates a call along the **client** axis (as `TrackDistributionAggregator` does along the
6472
+ * publisher→subscriber axis): per-client health split into sending vs receiving, plus percentile
6473
+ * rollups and "affected ratio" counts for the whole call.
6474
+ *
6475
+ * Build once per call and call `aggregate()` on each `call.update()`.
6476
+ */
6477
+ declare class CallHealthAggregator {
6478
+ private readonly _call;
6479
+ readonly thresholds: ClientHealthThresholds;
6480
+ constructor(_call: ObservedCall, thresholds?: ClientHealthThresholds);
6481
+ aggregate(): CallHealth;
6482
+ private _clientHealth;
6483
+ }
6484
+
6485
+ /** An entry retained by {@link SlidingWindow}. */
6486
+ type SlidingWindowEntry<T> = {
6487
+ timestamp: number;
6488
+ value: T;
6489
+ };
6490
+ /**
6491
+ * A time-bounded buffer used by detectors that reason over a window ("N of M clients degraded within
6492
+ * 10 s"). Entries older than `windowMs` are evicted on write and on read.
6493
+ *
6494
+ * ### Ordering is enforced, not assumed
6495
+ *
6496
+ * Eviction walks from the front and stops at the first entry still inside the window, which is only
6497
+ * correct if entries are ordered by timestamp. Callers mostly pass `Date.now()` and are ordered by
6498
+ * construction — but not always: a timestamp taken from a client sample, or two calls inside the
6499
+ * same millisecond, can arrive out of order, and one such entry would park itself at the head and
6500
+ * stop eviction *permanently*, so the window would grow without bound and keep reporting symptoms
6501
+ * from hours ago.
6502
+ *
6503
+ * Rather than trust the caller, {@link add} inserts in timestamp order. Appending (the overwhelmingly
6504
+ * common case) stays O(1); an out-of-order insert costs a short backward scan, because such entries
6505
+ * are near the tail in practice.
6506
+ *
6507
+ * ### The window advances on the newest observation
6508
+ *
6509
+ * Eviction is relative to the largest timestamp seen, not the one just passed. A caller that reads
6510
+ * with a `now` behind the newest entry (a replayed sample, a clock that stepped back) would
6511
+ * otherwise un-evict nothing and, worse, a caller passing an old `now` to {@link add} would evict
6512
+ * everything newer.
6513
+ */
6514
+ declare class SlidingWindow<T> {
6515
+ readonly windowMs: number;
6516
+ /** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
6517
+ readonly maxEntries: number;
6518
+ private _entries;
6519
+ private _latest;
6520
+ constructor(windowMs: number,
6521
+ /** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
6522
+ maxEntries?: number);
6523
+ /** Add an entry (defaults to `Date.now()`), then evict anything outside the window. */
6524
+ add(value: T, timestamp?: number): void;
6525
+ /**
6526
+ * The entries still inside the window, oldest first.
6527
+ *
6528
+ * A **copy** — callers routinely map/sort what they get back, and handing out the live array let
6529
+ * them mutate the window from the outside.
6530
+ */
6531
+ entries(now?: number): SlidingWindowEntry<T>[];
6532
+ /** The values still inside the window, oldest first. */
6533
+ values(now?: number): T[];
6534
+ /**
6535
+ * How many entries are inside the window as of `now`.
6536
+ *
6537
+ * Prefer this to `values(now).length`: counting through {@link values} allocates an array of every
6538
+ * entry only to read its length, which on a hot path is the whole cost of the call.
6539
+ */
6540
+ count(now?: number): number;
6541
+ /** Retained entries, without evicting first. See {@link count} for the windowed answer. */
6542
+ get size(): number;
6543
+ clear(): void;
6544
+ private _evict;
6545
+ }
6546
+
6547
+ type TrendTesterConfig = {
6548
+ /**
6549
+ * How many of the most recent values to keep. Default `30`, floored at `2`.
6550
+ *
6551
+ * The one knob controlling how far back "trend" looks, for both tests — there is deliberately no
6552
+ * separate window per test. This counts *samples*, so the time it spans depends on how often you
6553
+ * push. Mann-Kendall needs roughly 8–10 points before its significance test is worth anything, so
6554
+ * below ~`10` it will mostly answer `no-trend`. Typical `20`–`60`: long enough for a stable
6555
+ * baseline, short enough that a sustained change eventually becomes the new normal instead of being
6556
+ * flagged forever.
6557
+ */
6558
+ size?: number;
6559
+ /**
6560
+ * Page-Hinkley's **drift tolerance** — change smaller than this is treated as noise and never
6561
+ * accumulated. Default `0`. See {@link pageHinkley}.
6562
+ *
6563
+ * Expressed in the units of whatever you push, so there is no universally good value: for RTT in ms
6564
+ * a few ms is a reasonable tolerance. The default `0` accumulates *every* deviation, which is the
6565
+ * most sensitive setting. Too low and ordinary fluctuation accumulates into a false step change; too
6566
+ * high and a real but gradual shift never accumulates at all.
6567
+ */
6568
+ pageHinkleyDelta?: number;
6569
+ /**
6570
+ * Page-Hinkley's **detection threshold** — how much accumulated drift counts as a step change.
6571
+ * Default `50`. See {@link pageHinkley}.
6572
+ *
6573
+ * Also in your units, and the direct sensitivity control: lower detects smaller or earlier steps and
6574
+ * false-positives more; higher waits for unmistakable ones. Worth tuning against a recorded series
6575
+ * rather than by intuition, because the right value depends entirely on the scale and noisiness of
6576
+ * the metric you feed it.
6577
+ */
6578
+ pageHinkleyLambda?: number;
6579
+ /**
6580
+ * Mann-Kendall's significance level. Default `0.05`. See {@link mannKendallVerdict}.
6581
+ *
6582
+ * Conventional values are `0.01`, `0.05` and `0.1`. This is the probability of claiming a monotonic
6583
+ * trend that is not really there: `0.01` is stricter and slower to call a trend, `0.1` more
6584
+ * sensitive and noisier.
6585
+ */
6586
+ mannKendallAlpha?: number;
6587
+ };
6588
+ /**
6589
+ * Streaming home for `stats.ts`'s two trend tests: feed it one value at a time via {@link add}
6590
+ * instead of re-running the batch functions over an array you manage yourself.
6591
+ *
6592
+ * ### The two tests answer different questions
6593
+ *
6594
+ * Take a client's RTT, sampled every couple of seconds. Two things can go wrong with it, and only
6595
+ * one of them looks like a spike:
6596
+ *
6597
+ * - **Mann-Kendall** asks *"is this drifting?"* — a monotonic trend, regardless of shape or scale.
6598
+ * `40, 45, 52, 61, 70, 84 ms` is a rising path with no single dramatic step; every jump is small
6599
+ * and plausible on its own. Mann-Kendall counts how many later samples exceed earlier ones and
6600
+ * reports whether that lopsidedness could plausibly be chance. It is rank-based, so one absurd
6601
+ * reading (a 4000 ms outlier from a stalled event loop) moves it by exactly one pair, not by the
6602
+ * 4000.
6603
+ * - **Page-Hinkley** asks *"did it change, and when?"* — a step. `40, 42, 39, 41, 180, 176, 182 ms`
6604
+ * is not a trend at all; it is one level followed by a different level, which is what a route
6605
+ * change or a TURN failover looks like. It accumulates the deviation from the running mean and
6606
+ * fires when the cumulative excess passes `lambda`.
6607
+ *
6608
+ * Neither subsumes the other, which is why both live here on one window. A slow climb toward
6609
+ * unusability shows up in Mann-Kendall and never trips Page-Hinkley; a hard failover trips
6610
+ * Page-Hinkley immediately while Mann-Kendall may read `no-trend`, because after the step the series
6611
+ * is flat again. `tests/trendTester.spec.ts` builds both RTT series and shows exactly this.
6612
+ *
6613
+ * ```ts
6614
+ * const rtt = new TrendTester({ size: 30, mannKendallAlpha: 0.05, pageHinkleyLambda: 50 });
6615
+ *
6616
+ * peerConnection.on('update', () => {
6617
+ * if (peerConnection.currentRttInMs === undefined) return; // no measurement is not a measurement
6618
+ * rtt.add(peerConnection.currentRttInMs);
6619
+ *
6620
+ * if (rtt.mannKendall().trend === 'increasing') warn('RTT is drifting up');
6621
+ * if (rtt.pageHinkley()?.changeDetected) {
6622
+ * warn('RTT stepped');
6623
+ * rtt.clear(); // the old level is no longer the baseline — judge the new one on its own
6624
+ * }
6625
+ * });
6626
+ * ```
6627
+ *
6628
+ * ### Both read the same window
6629
+ *
6630
+ * `size` is the one knob controlling how far back either test looks. They are kept incremental
6631
+ * differently, because they don't tolerate an evicted point the same way:
6632
+ *
6633
+ * - **Mann-Kendall**'s statistic is a sum over *pairs*, so evicting the oldest value only touches
6634
+ * the pairs it was part of — one pass over the (bounded) window corrects it in O(size) instead of
6635
+ * the O(size²) a full recompute costs.
6636
+ * - **Page-Hinkley**'s statistic is a running minimum of a cumulative sum, which has no cheap
6637
+ * correction for "forget this one old point" — the minimum may have depended on it. It is
6638
+ * recomputed from the window on every {@link add} rather than hand-rolling an incremental version
6639
+ * that would be easy to get subtly wrong. That recompute is O(size), the same order as above.
6640
+ *
6641
+ * ### Non-finite input is rejected, not absorbed
6642
+ *
6643
+ * See {@link add}. A single `NaN` would otherwise destroy the instance permanently.
6644
+ */
6645
+ declare class TrendTester {
6646
+ private readonly _size;
6647
+ private readonly _values;
6648
+ private readonly _tieCounts;
6649
+ private readonly _pageHinkleyDelta;
6650
+ private readonly _pageHinkleyLambda;
6651
+ private readonly _mannKendallAlpha;
6652
+ private _s;
6653
+ private _rejected;
6654
+ private _pageHinkleyResult?;
6655
+ constructor(config?: TrendTesterConfig);
6656
+ /** Values rejected by {@link add} for being non-finite. Non-zero means the caller has a bug. */
6657
+ get rejected(): number;
6658
+ /** How many values are currently in the window (`<= size`). */
6659
+ get length(): number;
6660
+ /** The configured window length. */
6661
+ get size(): number;
6662
+ /**
6663
+ * Add the next value in the stream, evicting the oldest once the window is full.
6664
+ *
6665
+ * **Non-finite values are rejected** rather than stored, and the rejection is counted in
6666
+ * {@link rejected}. This is not defensive noise — it is the difference between a bad reading and
6667
+ * a bad instance. `Math.sign(NaN)` is `NaN`, so a single `NaN` would poison the incremental
6668
+ * Mann-Kendall sum `_s` **permanently**: every later `add` and `_evictOldest` adds or subtracts
6669
+ * `NaN`, the z-score is `NaN`, every comparison against it is `false`, and the tester silently
6670
+ * reports `no-trend` forever after. It would also take a `NaN` key in `_tieCounts` that can never
6671
+ * be matched on eviction, since `NaN !== NaN`.
6672
+ *
6673
+ * `undefined` RTT (no measurement this tick) must not be coerced to `0` and passed in either —
6674
+ * "we didn't measure" is not "the trip took no time", and feeding zeros manufactures a downward
6675
+ * trend. Skip the sample instead.
6676
+ */
6677
+ add(value: number): void;
6678
+ /** The current Page-Hinkley read-out over the window. `undefined` before the first value. */
6679
+ pageHinkley(): PageHinkleyResult | undefined;
6680
+ /** The current Mann-Kendall read-out over the window. */
6681
+ mannKendall(): MannKendallResult;
6682
+ /** Drop everything, e.g. after a detected change point, to start judging the trend fresh. */
6683
+ clear(): void;
6684
+ /** Remove the oldest value from the window and correct `_s` for the pairs it was part of. */
6685
+ private _evictOldest;
6686
+ private _bumpTie;
6687
+ }
6688
+
2885
6689
  interface Logger {
2886
6690
  trace(...args: any[]): void;
2887
6691
  debug(...args: any[]): void;
@@ -2953,4 +6757,4 @@ declare function createInMemorySink(samples?: ClientSample[]): InMemorySink;
2953
6757
  declare function createDefaultMediasoupRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
2954
6758
  declare function createP2pRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
2955
6759
 
2956
- export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type CallAppDataFactory, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type Detector, Detectors, InMemorySink, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, type Logger, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, type ObserverEventBase, type ObserverEvents, type ObserverLogger, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolvers, type SampleRejectedReason, type ScoreCalculator, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, setObserverLogger };
6760
+ export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, type ActiveIssueTracker, ActiveIssuesRegistry, type AvailableCallScopeDetectorsConfigs, type AvailableDetectorsConfigs, type AvailableObserverScopeDetectorsConfigs, type AvailableValidatorConfigs, CODEC_MISMATCH_ISSUE, type CalculatedScore, type CallAppDataFactory, CallConcurrentIssueDetector, type CallConcurrentIssueDetectorConfig, type CallConcurrentIssueGroup, CallConcurrentIssueTypes, type CallHealth, CallHealthAggregator, type CallIssue, type CallIssueSpread, type CallScopedEventName, type CallSummary, type CallSummaryClients, CallSummaryCollector, type CallSummaryConfig, type CallSummaryEnricher, type CallSummaryEnrichers, type CallSummaryScores, type CallSummarySection, type CallSummaryTruncation, type CallSummaryTurnServers, type CertificateStats, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientLocation, type ClientLocationResolver, type ClientMetaData, ClientMetaTypes, type ClientPopulation, type ClientPopulationAxis, ClientPopulationIssueDetector, type ClientPopulationIssueDetectorConfig, ClientPopulationIssueTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type CodecConsistencyReportPayload, CodecConsistencyValidator, type CodecConsistencyValidatorConfig, type CodecEvidence, type CodecStats, type CorroboratedPublisherFault, type DataChannelStats, type Detector, Detectors, type ExtensionStat, GEOHASH_CELL_SIZES, type IceCandidatePairStats, type IceCandidateStats, type IceTransportStats, InMemorySink, type InboundRtpStats, type InboundTrackSample, type Issue, type IssueBase, type IssueConclusion, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, type IssueFaultDomain, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, LOWEST_COMMON_DENOMINATOR_ISSUE, type Logger, type MannKendallResult, type MediaKind, type MediaPlayoutStats, type MediaSourceStats, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, MiddlewareProcessor, ObservedCall, type ObservedCallScope, type ObservedCallSettings, ObservedCertificate, ObservedClient, ObservedClientIssueRegistry, type ObservedClientScope, type ObservedClientSettings, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, ObservedTURN, type ObservedTURNEventMap, ObservedTurnServer, Observer, ObserverConcurrentIssueDetector, type ObserverConcurrentIssueDetectorConfig, type ObserverConcurrentIssueGroup, ObserverConcurrentIssueTypes, type ObserverEventBase, type ObserverEvents, type ObserverIssue, type ObserverIssueSpread, type ObserverLogger, type OutboundRtpStats, type OutboundTrackSample, type PageHinkleyResult, type PeerConnectionSample, type PeerConnectionTransportStats, type PsnrSum, PublisherFaultCorroborationDetector, type PublisherFaultCorroborationDetectorConfig, PublisherFaultTypes, type QualityLimitationDurations, RESOLVED_ISSUE_SUFFIX, type RemoteInboundRtpStats, type RemoteOutboundRtpStats, type RemoteTrackLinkEvidence, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolverReportPayload, RemoteTrackResolverValidator, type RemoteTrackResolverValidatorConfig, type RemoteTrackResolvers, type ResolvedActiveClientIssue, type RunningValidator, type SampleRejectedReason, type ScoreCalculator, SfuCongestionDetector, type SfuCongestionDetectorBucket, type SfuCongestionDetectorConfig, type SfuCongestionDetectorEvaluation, type SfuCongestionDetectorReport, type SimulcastReceiverEvidence, type SimulcastReceiverReportPayload, SimulcastReceiverValidator, type SimulcastReceiverValidatorConfig, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrendTester, type TrendTesterConfig, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, TurnServerOutageDetector, type TurnServerOutageDetectorConfig, TurnServerOutageTypes, UNRESOLVED_TRACK_LINKS_ISSUE, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, type ValidationReport, type Validator, type ValidatorName, baseIssueType, concludeCallIssue, concludeObserverIssue, correlation, counterDelta, createCallSummary, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultCallSummaryConfig, defaultClientHealthThresholds, geohash, isClientIssueResolutionEntry, issuePayloadAsString, mannKendall, mannKendallVerdict, median, medianAbsoluteDeviation, pageHinkley, percentile, percentileOfSorted, robustZScore, schemaVersion, setObserverLogger, summarize };