@observertc/observer-js 1.0.0-beta.13 → 1.0.0-beta.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +643 -61
- package/dist/index.d.mts +2731 -74
- package/dist/index.d.ts +2731 -74
- package/dist/index.js +3471 -374
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +3421 -373
- package/dist/index.mjs.map +1 -1
- package/package.json +14 -4
- package/llms-full.txt +0 -1632
package/dist/index.d.ts
CHANGED
|
@@ -52,6 +52,12 @@ type ClientIssue = {
|
|
|
52
52
|
*/
|
|
53
53
|
type: string;
|
|
54
54
|
/**
|
|
55
|
+
* Identity of a **stateful** issue, shared by its raise entry and its `<type>-resolved`
|
|
56
|
+
* companion, so a server can open an active issue on the raise and close it on the match
|
|
57
|
+
* (client-monitor-js >= 4.6.0, `sendResolvedIssuesToServer`). One-shot issues have no key.
|
|
58
|
+
*/
|
|
59
|
+
key?: string;
|
|
60
|
+
/**
|
|
55
61
|
* The value associated with the event, if applicable.
|
|
56
62
|
*/
|
|
57
63
|
payload?: string;
|
|
@@ -1585,6 +1591,24 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
|
|
|
1585
1591
|
bitPerPixel: number;
|
|
1586
1592
|
deltaPacketsSent: number;
|
|
1587
1593
|
deltaBytesSent: number;
|
|
1594
|
+
deltaFramesSent: number;
|
|
1595
|
+
deltaFramesEncoded: number;
|
|
1596
|
+
deltaKeyFramesEncoded: number;
|
|
1597
|
+
deltaNackCount: number;
|
|
1598
|
+
deltaPliCount: number;
|
|
1599
|
+
deltaFirCount: number;
|
|
1600
|
+
deltaRetransmittedPacketsSent: number;
|
|
1601
|
+
deltaRetransmittedBytesSent: number;
|
|
1602
|
+
deltaEncodeTime: number;
|
|
1603
|
+
deltaQualityLimitationResolutionChanges: number;
|
|
1604
|
+
/**
|
|
1605
|
+
* `true` when the codec, encoder implementation or scalability mode changed in this tick.
|
|
1606
|
+
*
|
|
1607
|
+
* Chrome resets `packetsSent`/`bytesSent` on the SSRC when the codec switches
|
|
1608
|
+
* (crbug.com/webrtc/5361), producing sawtooth spikes and negative bitrates. All deltas for the
|
|
1609
|
+
* tick are suppressed so a codec change is never mistaken for a traffic event.
|
|
1610
|
+
*/
|
|
1611
|
+
counterResetBoundary: boolean;
|
|
1588
1612
|
remoteRttInMs?: number;
|
|
1589
1613
|
remoteFractionLost?: number;
|
|
1590
1614
|
remoteJitter?: number;
|
|
@@ -1599,6 +1623,241 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
|
|
|
1599
1623
|
update(stats: OutboundRtpStats): void;
|
|
1600
1624
|
}
|
|
1601
1625
|
|
|
1626
|
+
/**
|
|
1627
|
+
* Small statistics helpers used by the aggregators and detectors.
|
|
1628
|
+
*
|
|
1629
|
+
* Rationale: a mean is a poor summary for call telemetry — one participant with a 1500 ms RTT
|
|
1630
|
+
* skews the average for nine healthy ones. Detectors should reason with medians, high percentiles
|
|
1631
|
+
* and "affected ratios" instead.
|
|
1632
|
+
*/
|
|
1633
|
+
/** A distribution summary of a numeric sample set. */
|
|
1634
|
+
type StatsSummary = {
|
|
1635
|
+
count: number;
|
|
1636
|
+
min: number;
|
|
1637
|
+
max: number;
|
|
1638
|
+
mean: number;
|
|
1639
|
+
median: number;
|
|
1640
|
+
p25: number;
|
|
1641
|
+
p75: number;
|
|
1642
|
+
p95: number;
|
|
1643
|
+
};
|
|
1644
|
+
/**
|
|
1645
|
+
* The p-th percentile (0..1) using linear interpolation between closest ranks.
|
|
1646
|
+
* Returns `undefined` for an empty input.
|
|
1647
|
+
*/
|
|
1648
|
+
declare function percentile(values: number[], p: number): number | undefined;
|
|
1649
|
+
/** The median (50th percentile). `undefined` for an empty input. */
|
|
1650
|
+
declare function median(values: number[]): number | undefined;
|
|
1651
|
+
/**
|
|
1652
|
+
* Median absolute deviation: `median(|value - median(values)|)`. A robust alternative to standard
|
|
1653
|
+
* deviation for describing how spread out `values` are — one wild outlier shifts a mean-based
|
|
1654
|
+
* deviation a lot, but barely moves a median. `undefined` for an empty input.
|
|
1655
|
+
*/
|
|
1656
|
+
declare function medianAbsoluteDeviation(values: number[]): number | undefined;
|
|
1657
|
+
/**
|
|
1658
|
+
* A robust z-score: how many (MAD-based) standard deviations `value` sits above/below the median of
|
|
1659
|
+
* `baseline`.
|
|
1660
|
+
*
|
|
1661
|
+
* Uses the median and MAD instead of the mean and standard deviation so a handful of baseline
|
|
1662
|
+
* outliers can't inflate the "normal" spread and mask a genuine spike — the same reasoning behind
|
|
1663
|
+
* {@link summarize}'s percentiles applies here to a single scalar spread. `1.4826` is the constant
|
|
1664
|
+
* that makes MAD estimate the standard deviation of a normal distribution, so the result reads on
|
|
1665
|
+
* the same scale as a classic z-score.
|
|
1666
|
+
*
|
|
1667
|
+
* A classic z-score divides by zero once every baseline value is identical (`MAD === 0`). Here:
|
|
1668
|
+
* `value` strictly above that constant baseline reads as `Infinity` — a spike with no precedent
|
|
1669
|
+
* whatsoever, however small; `value` at or below it reads as `0`, indistinguishable from the (flat)
|
|
1670
|
+
* baseline rather than a division error.
|
|
1671
|
+
*
|
|
1672
|
+
* ### Worked example
|
|
1673
|
+
*
|
|
1674
|
+
* A baseline of "share of clients reporting congestion", one entry per 10 s bucket, on a healthy
|
|
1675
|
+
* fleet: `[0.02, 0.01, 0.03, 0.04, 0.02]`. Median `0.02`, MAD `0.01`, so the scale is
|
|
1676
|
+
* `1.4826 * 0.01 ≈ 0.0148`.
|
|
1677
|
+
*
|
|
1678
|
+
* ```ts
|
|
1679
|
+
* robustZScore(0.03, baseline); // ≈ 0.67 — an ordinary bucket
|
|
1680
|
+
* robustZScore(0.25, baseline); // ≈ 15.5 — a quarter of the fleet at once; nothing like it before
|
|
1681
|
+
* ```
|
|
1682
|
+
*
|
|
1683
|
+
* Now add one bad bucket to the *baseline* — `[0.02, 0.01, 0.03, 0.04, 0.02, 0.40]`. A mean/stddev
|
|
1684
|
+
* z-score would absorb it: the mean climbs to `0.087` and the stddev to `~0.14`, so a genuine `0.25`
|
|
1685
|
+
* spike scores about `1.2` and looks unremarkable. **One past incident would hide the next one.**
|
|
1686
|
+
* Median and MAD barely move (median `0.025`, MAD `0.01`), so `0.25` still scores `≈15.2`. That
|
|
1687
|
+
* resistance is the entire reason for this function.
|
|
1688
|
+
*
|
|
1689
|
+
* The `Infinity` case is not an edge case in practice — it is a fleet that has been perfectly quiet:
|
|
1690
|
+
*
|
|
1691
|
+
* ```ts
|
|
1692
|
+
* robustZScore(0.10, [ 0, 0, 0, 0, 0 ]); // Infinity — first congestion ever seen
|
|
1693
|
+
* robustZScore(0, [ 0, 0, 0, 0, 0 ]); // 0 — still nothing happening
|
|
1694
|
+
* ```
|
|
1695
|
+
*
|
|
1696
|
+
* `Infinity` clears any finite threshold, which is intended: "we have never seen this" *is* the
|
|
1697
|
+
* strongest possible statistical statement. It is also why a caller must gate on practical
|
|
1698
|
+
* significance too — see `SfuCongestionDetector`, which additionally requires a minimum number of
|
|
1699
|
+
* affected clients, so a single client on a quiet fleet cannot page anyone.
|
|
1700
|
+
*
|
|
1701
|
+
* `undefined` when `baseline` is empty — there is nothing to compare against.
|
|
1702
|
+
*/
|
|
1703
|
+
declare function robustZScore(value: number, baseline: number[]): number | undefined;
|
|
1704
|
+
/** Summarize a numeric sample set. Returns `undefined` for an empty input. */
|
|
1705
|
+
declare function summarize(values: number[]): StatsSummary | undefined;
|
|
1706
|
+
/**
|
|
1707
|
+
* Linear-interpolated percentile of an **already ascending** array. The building block behind
|
|
1708
|
+
* {@link percentile} and {@link summarize}; exported so callers computing several percentiles over
|
|
1709
|
+
* the same data can sort once themselves.
|
|
1710
|
+
*
|
|
1711
|
+
* Passing an unsorted array yields a meaningless number rather than an error — sort first.
|
|
1712
|
+
*/
|
|
1713
|
+
declare function percentileOfSorted(sorted: number[], p: number): number;
|
|
1714
|
+
/**
|
|
1715
|
+
* A counter-reset-safe delta: the increase of a cumulative counter between two observations.
|
|
1716
|
+
*
|
|
1717
|
+
* Returns `0` when the counter went backwards (reset / SSRC reuse) or when either side is missing.
|
|
1718
|
+
* NOTE the guard is `>=` on **defined** values — a previous value of `0` is a perfectly valid
|
|
1719
|
+
* baseline, so `0 -> 5` correctly yields `5` (using a truthiness check here silently drops the
|
|
1720
|
+
* first interval of every counter, which is exactly when the first loss/freeze event happens).
|
|
1721
|
+
*/
|
|
1722
|
+
declare function counterDelta(previous: number | undefined, current: number | undefined): number;
|
|
1723
|
+
/**
|
|
1724
|
+
* Pearson correlation of two equal-length series, clamped to `0..1`.
|
|
1725
|
+
*
|
|
1726
|
+
* Negative or undefined relationships read as `0`, because every caller here asks "does A follow B?"
|
|
1727
|
+
* — an inverse relationship is not a weaker yes, it is a no.
|
|
1728
|
+
*/
|
|
1729
|
+
declare function correlation(xs: number[], ys: number[]): number;
|
|
1730
|
+
/** Per-step result of {@link pageHinkley}. */
|
|
1731
|
+
type PageHinkleyResult = {
|
|
1732
|
+
/**
|
|
1733
|
+
* The Page-Hinkley statistic after each observation, same length as the input — one value per
|
|
1734
|
+
* `values[i]`, so `statistic[i]` is "how far the cumulative deviation has grown above its own
|
|
1735
|
+
* historical low, using only `values[0..i]`". Never negative (it's a gap to a *minimum*), and
|
|
1736
|
+
* `0` for as long as the process looks stable.
|
|
1737
|
+
*/
|
|
1738
|
+
statistic: number[];
|
|
1739
|
+
/**
|
|
1740
|
+
* Index of the **first** observation whose statistic exceeded `lambda`, else `undefined`.
|
|
1741
|
+
*
|
|
1742
|
+
* This is a one-shot latch over the call's whole input: once found, later observations are not
|
|
1743
|
+
* re-checked, even if the statistic subsequently falls back down (e.g. after a single transient
|
|
1744
|
+
* spike — see the class doc example). "Is it *still* elevated right now" is a different
|
|
1745
|
+
* question, answered by looking at the *tail* of {@link statistic}, or — for a live stream — by
|
|
1746
|
+
* re-running this over a recent window each time (which is what `TrendTester` does) rather than
|
|
1747
|
+
* over the whole history once.
|
|
1748
|
+
*/
|
|
1749
|
+
changePointIndex?: number;
|
|
1750
|
+
/** `true` when {@link changePointIndex} is defined. */
|
|
1751
|
+
changeDetected: boolean;
|
|
1752
|
+
};
|
|
1753
|
+
/**
|
|
1754
|
+
* Page-Hinkley test: sequential (online) detection of a sustained **increase** in the mean of
|
|
1755
|
+
* `values` — "has this metric settled onto a durably higher level", as opposed to "did one sample
|
|
1756
|
+
* come in high". A single noisy point should not read as a regression; ten points that are all a
|
|
1757
|
+
* bit higher than before should.
|
|
1758
|
+
*
|
|
1759
|
+
* ### How it works
|
|
1760
|
+
*
|
|
1761
|
+
* At step `i`, three numbers are tracked:
|
|
1762
|
+
*
|
|
1763
|
+
* 1. `mean` — the running average of `values[0..i]` (**not** a fixed baseline — it is recomputed
|
|
1764
|
+
* from everything seen so far, including `values[i]` itself, which is what makes this
|
|
1765
|
+
* *adaptive*: after a real shift, `mean` keeps drifting up to meet the new level, and the
|
|
1766
|
+
* signal below fades back out on its own rather than staying triggered forever).
|
|
1767
|
+
* 2. `cumulative` — running sum of `(values[i] - mean - delta)`. Subtracting `mean` centres each
|
|
1768
|
+
* term on "surprise relative to what we've seen so far"; subtracting `delta` on top means a
|
|
1769
|
+
* small positive surprise still nets to a *negative* contribution, so it doesn't accumulate.
|
|
1770
|
+
* 3. `runningMinimum` — the lowest `cumulative` has ever been.
|
|
1771
|
+
*
|
|
1772
|
+
* The **Page-Hinkley statistic** is `cumulative - runningMinimum`: how far the running sum has
|
|
1773
|
+
* climbed above its own historical floor. Pure noise pulls `cumulative` up and down around a flat
|
|
1774
|
+
* trend, so the gap to `runningMinimum` stays small. A sustained increase pushes `cumulative`
|
|
1775
|
+
* mostly one direction — up — so `runningMinimum` stops updating and the gap grows every step,
|
|
1776
|
+
* crossing `lambda` once the shift is large/long enough to be sure it isn't noise. That first
|
|
1777
|
+
* crossing is {@link PageHinkleyResult.changePointIndex}.
|
|
1778
|
+
*
|
|
1779
|
+
* ### Parameters
|
|
1780
|
+
*
|
|
1781
|
+
* - `delta` — the **drift tolerance** (in the same units as `values`, e.g. ms of RTT): how much of
|
|
1782
|
+
* a step-to-step increase is written off as noise rather than counted towards the cumulative sum.
|
|
1783
|
+
* `0` means even a razor-thin, consistent upward creep eventually accumulates enough to trigger;
|
|
1784
|
+
* raising it requires each observation to clear that bar above the running mean before it
|
|
1785
|
+
* contributes anything (see the class doc's hand-worked example — the same jump that triggers
|
|
1786
|
+
* with `delta: 0` is completely absorbed at `delta: 10`).
|
|
1787
|
+
* - `lambda` — the **detection threshold** the statistic must exceed. It is in "surprise units" (a
|
|
1788
|
+
* sum of deviations, not a single observation's units), so there's no shortcut for picking it
|
|
1789
|
+
* other than trying it against representative data — see the RTT examples in `stats.spec.ts` for
|
|
1790
|
+
* a worked comparison of a low vs. a high `lambda` on the same series. Raising it delays
|
|
1791
|
+
* detection but makes a false positive from a lucky run of noise less likely.
|
|
1792
|
+
*
|
|
1793
|
+
* O(n) — one pass, unlike {@link mannKendall}'s O(n²). To detect a **decrease** instead of an
|
|
1794
|
+
* increase, negate `values` before calling (or negate the result's meaning if you'd rather).
|
|
1795
|
+
*/
|
|
1796
|
+
declare function pageHinkley(values: number[], delta?: number, lambda?: number): PageHinkleyResult;
|
|
1797
|
+
/** Result of {@link mannKendall}. */
|
|
1798
|
+
type MannKendallResult = {
|
|
1799
|
+
/** Sum of pairwise signs (`sign(values[j] - values[i])` for every `i < j`). */
|
|
1800
|
+
s: number;
|
|
1801
|
+
/** Variance of {@link s}, corrected for tied values. */
|
|
1802
|
+
variance: number;
|
|
1803
|
+
/** Standard-normal score derived from `s`. `0` when `s` is `0` — no evidence either way. */
|
|
1804
|
+
z: number;
|
|
1805
|
+
/** Two-tailed p-value for the null hypothesis "no monotonic trend". */
|
|
1806
|
+
pValue: number;
|
|
1807
|
+
/** `'increasing'` / `'decreasing'` when significant at `alpha`, else `'no-trend'`. */
|
|
1808
|
+
trend: 'increasing' | 'decreasing' | 'no-trend';
|
|
1809
|
+
};
|
|
1810
|
+
/**
|
|
1811
|
+
* Mann-Kendall trend test: a non-parametric test for a **monotonic** trend in `values`, without
|
|
1812
|
+
* assuming a distribution or a constant rate of change — it only asks "are later values
|
|
1813
|
+
* consistently larger (or smaller) than earlier ones more often than chance would allow".
|
|
1814
|
+
*
|
|
1815
|
+
* Every pair `i < j` votes `+1` (`values[j] > values[i]`), `-1` (`values[j] < values[i]`) or `0`
|
|
1816
|
+
* (tie); `s` is the sum of those votes. Under the null hypothesis of no trend, `s` is
|
|
1817
|
+
* approximately normal with mean `0` and a known variance (corrected here for tied values), which
|
|
1818
|
+
* turns `s` into a Z score and a two-tailed p-value. `trend` is only `'increasing'` / `'decreasing'`
|
|
1819
|
+
* when that p-value clears `alpha` — a handful of mostly-ascending points is exactly what noise
|
|
1820
|
+
* looks like half the time, and this is what keeps that from reading as a trend.
|
|
1821
|
+
*
|
|
1822
|
+
* O(n²) (every pair is compared); fine for the small, per-tick sample counts detectors work with,
|
|
1823
|
+
* not for large historical series.
|
|
1824
|
+
*/
|
|
1825
|
+
declare function mannKendall(values: number[], alpha?: number): MannKendallResult;
|
|
1826
|
+
/**
|
|
1827
|
+
* Turns a Mann-Kendall `s` statistic and its `variance` into a Z score, p-value and verdict — the
|
|
1828
|
+
* back half of {@link mannKendall}, split out so an incremental caller that maintains `s` /
|
|
1829
|
+
* `variance` itself (e.g. over a sliding window, correcting for evicted points rather than
|
|
1830
|
+
* recomputing every pair from scratch) doesn't have to reimplement the normal approximation.
|
|
1831
|
+
*/
|
|
1832
|
+
declare function mannKendallVerdict(s: number, variance: number, alpha?: number): Pick<MannKendallResult, 'z' | 'pValue' | 'trend'>;
|
|
1833
|
+
|
|
1834
|
+
/** The per-receiver view the aggregator builds for one subscribed track. */
|
|
1835
|
+
type PublishedTrackReceivingDistributionEntry = {
|
|
1836
|
+
numberOfReceivers: number;
|
|
1837
|
+
numberOfHealthyReceivers: number;
|
|
1838
|
+
numberOfDegradedReceivers: number;
|
|
1839
|
+
/** degradedReceivers / receivers (0..1); `0` when there are no receivers. */
|
|
1840
|
+
degradedRatio: number;
|
|
1841
|
+
/** Distribution summaries across receivers (undefined when no receiver reported the metric). */
|
|
1842
|
+
bitrate?: StatsSummary;
|
|
1843
|
+
fractionLost?: StatsSummary;
|
|
1844
|
+
jitter?: StatsSummary;
|
|
1845
|
+
rttInMs?: StatsSummary;
|
|
1846
|
+
jitterBufferDelayInMs?: StatsSummary;
|
|
1847
|
+
concealmentRatio?: StatsSummary;
|
|
1848
|
+
/** Fan-out counters: how many receivers saw the symptom, and the total across them. */
|
|
1849
|
+
freezes: {
|
|
1850
|
+
affectedReceivers: number;
|
|
1851
|
+
total: number;
|
|
1852
|
+
};
|
|
1853
|
+
plis: {
|
|
1854
|
+
affectedReceivers: number;
|
|
1855
|
+
total: number;
|
|
1856
|
+
};
|
|
1857
|
+
concealment: {
|
|
1858
|
+
affectedReceivers: number;
|
|
1859
|
+
};
|
|
1860
|
+
};
|
|
1602
1861
|
declare class ObservedOutboundTrack implements OutboundTrackSample {
|
|
1603
1862
|
timestamp: number;
|
|
1604
1863
|
readonly id: string;
|
|
@@ -1609,18 +1868,28 @@ declare class ObservedOutboundTrack implements OutboundTrackSample {
|
|
|
1609
1868
|
private _visited;
|
|
1610
1869
|
appData?: Record<string, unknown>;
|
|
1611
1870
|
readonly remoteInboundTracks: Set<ObservedInboundTrack>;
|
|
1871
|
+
readonly detectors: Detectors;
|
|
1872
|
+
receivingDistribution?: PublishedTrackReceivingDistributionEntry;
|
|
1612
1873
|
readonly calculatedScore: CalculatedScore;
|
|
1613
1874
|
addedAt?: number | undefined;
|
|
1614
1875
|
removedAt?: number | undefined;
|
|
1615
1876
|
muted?: boolean;
|
|
1616
1877
|
attachments?: Record<string, unknown> | undefined;
|
|
1878
|
+
degradedReasons?: string[] | undefined;
|
|
1879
|
+
bitrate?: number | undefined;
|
|
1880
|
+
deltaPacketsSent?: number | undefined;
|
|
1881
|
+
remoteFractionLost?: number | undefined;
|
|
1882
|
+
remoteRttInMs?: number | undefined;
|
|
1883
|
+
qualityLimitationReason?: string | undefined;
|
|
1617
1884
|
constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _outboundRtps?: ObservedOutboundRtp[] | undefined, _mediaSource?: ObservedMediaSource | undefined);
|
|
1618
1885
|
get score(): number | undefined;
|
|
1619
1886
|
get visited(): boolean;
|
|
1887
|
+
get degraded(): boolean;
|
|
1620
1888
|
getPeerConnection(): ObservedPeerConnection;
|
|
1621
1889
|
getOutboundRtps(): ObservedOutboundRtp[] | undefined;
|
|
1622
1890
|
getMediaSource(): ObservedMediaSource | undefined;
|
|
1623
1891
|
update(stats: OutboundTrackSample): void;
|
|
1892
|
+
private createReceivingDistribution;
|
|
1624
1893
|
}
|
|
1625
1894
|
|
|
1626
1895
|
declare class ObservedInboundTrack implements InboundTrackSample {
|
|
@@ -1638,6 +1907,8 @@ declare class ObservedInboundTrack implements InboundTrackSample {
|
|
|
1638
1907
|
removedAt?: number | undefined;
|
|
1639
1908
|
muted?: boolean;
|
|
1640
1909
|
attachments?: Record<string, unknown> | undefined;
|
|
1910
|
+
degradationReasons: string[];
|
|
1911
|
+
get degraded(): boolean;
|
|
1641
1912
|
constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _inboundRtp?: ObservedInboundRtp | undefined, _mediaPlayout?: ObservedMediaPlayout | undefined);
|
|
1642
1913
|
get score(): number | undefined;
|
|
1643
1914
|
get visited(): boolean;
|
|
@@ -1645,6 +1916,7 @@ declare class ObservedInboundTrack implements InboundTrackSample {
|
|
|
1645
1916
|
getInboundRtp(): ObservedInboundRtp | undefined;
|
|
1646
1917
|
getMediaPlayout(): ObservedMediaPlayout | undefined;
|
|
1647
1918
|
update(stats: InboundTrackSample): void;
|
|
1919
|
+
private checkDegradation;
|
|
1648
1920
|
}
|
|
1649
1921
|
|
|
1650
1922
|
declare class ObservedRemoteOutboundRtp implements RemoteOutboundRtpStats {
|
|
@@ -1752,6 +2024,46 @@ declare class ObservedInboundRtp implements InboundRtpStats {
|
|
|
1752
2024
|
deltaBytesReceived: number;
|
|
1753
2025
|
deltaReceivedSamples: number;
|
|
1754
2026
|
deltaSilentConcealedSamples: number;
|
|
2027
|
+
deltaConcealedSamples: number;
|
|
2028
|
+
deltaConcealmentEvents: number;
|
|
2029
|
+
deltaFreezeCount: number;
|
|
2030
|
+
deltaFreezesDuration: number;
|
|
2031
|
+
deltaPliCount: number;
|
|
2032
|
+
deltaNackCount: number;
|
|
2033
|
+
deltaFirCount: number;
|
|
2034
|
+
deltaPacketsDiscarded: number;
|
|
2035
|
+
deltaFramesDecoded: number;
|
|
2036
|
+
deltaFramesReceived: number;
|
|
2037
|
+
deltaFramesRendered: number;
|
|
2038
|
+
deltaFramesDropped: number;
|
|
2039
|
+
deltaKeyFramesDecoded: number;
|
|
2040
|
+
deltaDecodeTime: number;
|
|
2041
|
+
deltaJitterBufferDelay: number;
|
|
2042
|
+
deltaJitterBufferEmittedCount: number;
|
|
2043
|
+
deltaRetransmittedPacketsReceived: number;
|
|
2044
|
+
deltaFecPacketsReceived: number;
|
|
2045
|
+
deltaFecPacketsDiscarded: number;
|
|
2046
|
+
deltaPausesDuration: number;
|
|
2047
|
+
/**
|
|
2048
|
+
* Mean jitter-buffer delay for the frames/samples emitted in this tick (seconds), derived from
|
|
2049
|
+
* the cumulative `jitterBufferDelay` / `jitterBufferEmittedCount` pair — the only correct way to
|
|
2050
|
+
* read those two counters.
|
|
2051
|
+
*/
|
|
2052
|
+
jitterBufferDelayInMs?: number;
|
|
2053
|
+
/** Fraction of the samples received in this tick that were concealed (0..1). */
|
|
2054
|
+
concealmentRatio?: number;
|
|
2055
|
+
/** Fraction of the frames received in this tick that were dropped before rendering (0..1). */
|
|
2056
|
+
framesDroppedRatio?: number;
|
|
2057
|
+
/**
|
|
2058
|
+
* `true` when the codec or decoder implementation changed in this tick.
|
|
2059
|
+
*
|
|
2060
|
+
* Chrome resets `packetsReceived`/`bytesReceived` on an SSRC when the codec switches
|
|
2061
|
+
* (crbug.com/webrtc/5361, open since 2015), which shows up as a sawtooth spike or a negative
|
|
2062
|
+
* rate. Every delta in this tick is therefore suppressed to `0` rather than reported as traffic
|
|
2063
|
+
* — otherwise a room-wide codec rollout produces a synchronized fake-degradation alert across
|
|
2064
|
+
* every participant at once.
|
|
2065
|
+
*/
|
|
2066
|
+
counterResetBoundary: boolean;
|
|
1755
2067
|
remoteRttInMs?: number;
|
|
1756
2068
|
remoteBytesSent?: number;
|
|
1757
2069
|
remotePacketsSent?: number;
|
|
@@ -1941,7 +2253,34 @@ declare class ObservedPeerConnection extends EventEmitter {
|
|
|
1941
2253
|
sendingVideoBitrate: number;
|
|
1942
2254
|
receivingAudioBitrate: number;
|
|
1943
2255
|
receivingVideoBitrate: number;
|
|
2256
|
+
/**
|
|
2257
|
+
* Median round-trip time of the tick, in ms.
|
|
2258
|
+
*
|
|
2259
|
+
* Prefers {@link rtcpRttInMs} and falls back to {@link iceRttInMs}, so within one tick it always
|
|
2260
|
+
* reports **one** kind of round trip. It used to be the median of both mixed together, which was
|
|
2261
|
+
* a bug: the mixing ratio changed as streams came and went, so the value moved for reasons that
|
|
2262
|
+
* had nothing to do with the network.
|
|
2263
|
+
*/
|
|
1944
2264
|
currentRttInMs?: number;
|
|
2265
|
+
/**
|
|
2266
|
+
* RTT measured by ICE/STUN consent checks, in ms — the trip to **whatever terminates ICE**. In an
|
|
2267
|
+
* SFU topology that is the SFU, so this is the client↔SFU leg, not client↔client.
|
|
2268
|
+
*/
|
|
2269
|
+
iceRttInMs?: number;
|
|
2270
|
+
/**
|
|
2271
|
+
* RTT reported by RTCP receiver reports, in ms — an **end-to-end** media-path round trip.
|
|
2272
|
+
*
|
|
2273
|
+
* Not the same trip as {@link iceRttInMs}; the difference between the two is roughly the far side
|
|
2274
|
+
* of the SFU (see {@link sfuHopRttInMs}).
|
|
2275
|
+
*/
|
|
2276
|
+
rtcpRttInMs?: number;
|
|
2277
|
+
/**
|
|
2278
|
+
* `rtcpRttInMs - iceRttInMs`, when both are known — an estimate of everything *past* the SFU.
|
|
2279
|
+
*
|
|
2280
|
+
* Useful for splitting "this client's own last mile is slow" (high `iceRttInMs`) from "the path
|
|
2281
|
+
* beyond the SFU is slow" (low ICE, high hop).
|
|
2282
|
+
*/
|
|
2283
|
+
sfuHopRttInMs?: number;
|
|
1945
2284
|
currentJitter?: number;
|
|
1946
2285
|
usingTCP: boolean;
|
|
1947
2286
|
usingTURN: boolean;
|
|
@@ -2036,6 +2375,179 @@ type OperationSystem = {
|
|
|
2036
2375
|
version: string;
|
|
2037
2376
|
};
|
|
2038
2377
|
|
|
2378
|
+
/**
|
|
2379
|
+
* A finding raised **by the observer** — `observedCall.addIssue()` / `observer.addIssue()`, surfaced
|
|
2380
|
+
* on the bus as `call-issue` / `observer-issue`.
|
|
2381
|
+
*
|
|
2382
|
+
* ### Why this is not `ClientIssue`
|
|
2383
|
+
*
|
|
2384
|
+
* `ClientIssue` is a **wire** type: it arrives inside a `ClientSample`, so its `payload` has to be a
|
|
2385
|
+
* string. Server-raised findings were reusing it, which forced every detector to `JSON.stringify` a
|
|
2386
|
+
* perfectly good object on the way out and every handler to `JSON.parse` it back on the way in —
|
|
2387
|
+
* paying serialisation on a path where nothing is ever serialised, and losing type information in
|
|
2388
|
+
* both directions.
|
|
2389
|
+
*
|
|
2390
|
+
* An observer issue goes straight to an in-process event handler, so it carries the object.
|
|
2391
|
+
*/
|
|
2392
|
+
type ObserverIssue = {
|
|
2393
|
+
/** What was found, e.g. `'CROSS_CALL_ISSUE_ONSET_BURST'`. */
|
|
2394
|
+
type: string;
|
|
2395
|
+
/** Observer clock, when the finding was raised. */
|
|
2396
|
+
timestamp: number;
|
|
2397
|
+
/**
|
|
2398
|
+
* The evidence behind the finding.
|
|
2399
|
+
*
|
|
2400
|
+
* Prefer an object — that is the point of this type. `string` is still accepted so an application
|
|
2401
|
+
* can forward a payload it already has serialised (e.g. relaying a `ClientIssue`) without a
|
|
2402
|
+
* pointless parse-then-restringify round trip. Use {@link issuePayloadOf} to read either form.
|
|
2403
|
+
*/
|
|
2404
|
+
payload?: string | Record<string, unknown>;
|
|
2405
|
+
};
|
|
2406
|
+
/**
|
|
2407
|
+
* The payload of an issue as an object, parsing it only if it happens to be a string.
|
|
2408
|
+
*
|
|
2409
|
+
* Handlers shouldn't have to care which form arrived. Returns `undefined` for a missing payload or a
|
|
2410
|
+
* string that isn't valid JSON — reading evidence must never throw inside an issue handler.
|
|
2411
|
+
*/
|
|
2412
|
+
declare function issuePayloadOf(issue: Pick<ObserverIssue, 'payload'>): Record<string, unknown> | undefined;
|
|
2413
|
+
/**
|
|
2414
|
+
* The payload as a JSON string, serialising it only if it is an object.
|
|
2415
|
+
*
|
|
2416
|
+
* For the boundaries that genuinely need text — a log line, an HTTP body, a message queue. Keep it at
|
|
2417
|
+
* the edge rather than in the detector, so in-process handlers never pay for it.
|
|
2418
|
+
*/
|
|
2419
|
+
declare function issuePayloadAsString(issue: Pick<ObserverIssue, 'payload'>): string | undefined;
|
|
2420
|
+
|
|
2421
|
+
/**
|
|
2422
|
+
* What a validator concluded, once it is done.
|
|
2423
|
+
*
|
|
2424
|
+
* `S` is the validator's own payload — a discriminated union on `verdict` plus whatever evidence
|
|
2425
|
+
* belongs to each outcome — so a report reads in the language of the thing being checked rather than
|
|
2426
|
+
* in generic pass/fail.
|
|
2427
|
+
*
|
|
2428
|
+
* Note this is a plain intersection with `S`, not `{ [K in keyof S]: S[K] }`. A mapped type over a
|
|
2429
|
+
* union collapses to the keys its members *share* (`keyof (A | B)` is the intersection), which would
|
|
2430
|
+
* silently drop every per-verdict evidence field and leave only `verdict` behind.
|
|
2431
|
+
*/
|
|
2432
|
+
type ValidationReport<S extends Record<string, unknown> = Record<string, unknown>> =
|
|
2433
|
+
/** Still running — it has not seen the conditions it needs to judge anything yet. */
|
|
2434
|
+
{
|
|
2435
|
+
ready: false;
|
|
2436
|
+
}
|
|
2437
|
+
/** Done. `verdict` is narrowed by `S` to that validator's own vocabulary. */
|
|
2438
|
+
| ({
|
|
2439
|
+
ready: true;
|
|
2440
|
+
verdict: string;
|
|
2441
|
+
decidedAt: number;
|
|
2442
|
+
} & S);
|
|
2443
|
+
/**
|
|
2444
|
+
* A **one-shot structural check**: something true of the deployment rather than of this moment.
|
|
2445
|
+
*
|
|
2446
|
+
* The distinction from a `Detector` is what changes over time. A detector answers "is something wrong
|
|
2447
|
+
* *right now*" — congestion, a dead relay, a track nobody receives — and the answer legitimately
|
|
2448
|
+
* differs every tick, so it runs every tick, forever. A validator answers "is this deployment built
|
|
2449
|
+
* correctly" — does the SFU pick layers per receiver — and that only changes when you deploy. So a
|
|
2450
|
+
* validator **runs until it knows, then finishes**: it calls {@link onDone} once, the observer drops
|
|
2451
|
+
* it, and nothing more is computed.
|
|
2452
|
+
*
|
|
2453
|
+
* To check again — after a deploy, say — start a new one with `observer.addValidator(...)`. There is
|
|
2454
|
+
* no revalidation timer, because a deploy, not the passage of time, is what makes a structural
|
|
2455
|
+
* verdict stale.
|
|
2456
|
+
*
|
|
2457
|
+
* Validators are held in `observer.validators` and driven by `observer.update()`.
|
|
2458
|
+
*/
|
|
2459
|
+
interface Validator<S extends Record<string, unknown> = Record<string, unknown>> {
|
|
2460
|
+
readonly name: string;
|
|
2461
|
+
/**
|
|
2462
|
+
* The conclusion so far. `{ ready: false }` until {@link onDone} fires — and **that is not a
|
|
2463
|
+
* pass**. A validator usually needs specific conditions to occur before it can judge anything,
|
|
2464
|
+
* and plenty of deployments never present them.
|
|
2465
|
+
*/
|
|
2466
|
+
readonly report: ValidationReport<S>;
|
|
2467
|
+
/** Called exactly once, when the validator finishes. The observer uses this to unregister it. */
|
|
2468
|
+
onDone: (report: ValidationReport<S>) => void;
|
|
2469
|
+
/** Gather evidence; decide if there is now enough. Called on every `observer.update()`. */
|
|
2470
|
+
update(): void;
|
|
2471
|
+
/** Give up without a verdict. Finishes with `inconclusive`, so a caller waiting on it is freed. */
|
|
2472
|
+
cancel: () => void;
|
|
2473
|
+
}
|
|
2474
|
+
/**
|
|
2475
|
+
* The part of a validator the observer needs in order to drive it.
|
|
2476
|
+
*
|
|
2477
|
+
* `Validator<S>` is invariant in `S` — `onDone` takes a `ValidationReport<S>` and `report` returns one
|
|
2478
|
+
* — so a `Set<Validator>` cannot hold validators with different payloads. Driving one only needs the
|
|
2479
|
+
* three members that don't mention `S`.
|
|
2480
|
+
*/
|
|
2481
|
+
type RunningValidator = Pick<Validator, 'name' | 'update' | 'cancel'>;
|
|
2482
|
+
|
|
2483
|
+
/** The suffix client-monitor-js appends to the type of a resolution entry. */
|
|
2484
|
+
declare const RESOLVED_ISSUE_SUFFIX = "-resolved";
|
|
2485
|
+
/**
|
|
2486
|
+
* A **stateful** client issue the server currently believes to be open.
|
|
2487
|
+
*
|
|
2488
|
+
* client-monitor-js (>= 4.6.0, `sendResolvedIssuesToServer`) puts an issue's whole lifecycle on the
|
|
2489
|
+
* wire as two entries sharing a `key`:
|
|
2490
|
+
*
|
|
2491
|
+
* ```
|
|
2492
|
+
* raise: { type: 'stuck-decoder', key, payload, timestamp: raisedAt }
|
|
2493
|
+
* resolution: { type: 'stuck-decoder-resolved', key, payload: { raisedAt, comment, …final }, timestamp: resolvedAt }
|
|
2494
|
+
* ```
|
|
2495
|
+
*
|
|
2496
|
+
* The observer opens an `ActiveClientIssue` on the raise and closes it on the matching key, which
|
|
2497
|
+
* turns a stream of point-in-time symptom reports into **intervals**. That is what makes the
|
|
2498
|
+
* difference between "several clients reported congestion in the last 10 seconds" (a guess built on
|
|
2499
|
+
* an arbitrary window) and "several clients are congested *right now, simultaneously*" — the latter
|
|
2500
|
+
* being real evidence of a shared cause.
|
|
2501
|
+
*
|
|
2502
|
+
* Issues still active when the client monitor closes are auto-resolved by the client, so a clean
|
|
2503
|
+
* departure does not leak.
|
|
2504
|
+
*/
|
|
2505
|
+
type ActiveClientIssue = {
|
|
2506
|
+
/** Identity shared by the raise and its resolution. Unique per client. */
|
|
2507
|
+
key: string;
|
|
2508
|
+
/** The issue type **without** the `-resolved` suffix (e.g. `'congestion'`). */
|
|
2509
|
+
type: string;
|
|
2510
|
+
/** The client that reported it. */
|
|
2511
|
+
clientId: string;
|
|
2512
|
+
/**
|
|
2513
|
+
* The call that client belongs to. The dimension that separates "one bad meeting" from "our
|
|
2514
|
+
* infrastructure": at observer scope, clients in *different* calls share nothing but the server.
|
|
2515
|
+
*/
|
|
2516
|
+
callId: string;
|
|
2517
|
+
/** When the client raised it (client clock). */
|
|
2518
|
+
raisedAt: number;
|
|
2519
|
+
/** When the observer first saw it (server clock) — skew-free, use this for cross-client timing. */
|
|
2520
|
+
observedAt: number;
|
|
2521
|
+
/** Parsed raise payload, when it was JSON. */
|
|
2522
|
+
payload?: Record<string, unknown>;
|
|
2523
|
+
/** `payload.peerConnectionId`, when present — most client detectors report it. */
|
|
2524
|
+
peerConnectionId?: string;
|
|
2525
|
+
/** `payload.trackId`, when present — the join key to a track (and thus to a publisher). */
|
|
2526
|
+
trackId?: string;
|
|
2527
|
+
};
|
|
2528
|
+
/** A closed interval: an {@link ActiveClientIssue} plus how it ended. */
|
|
2529
|
+
type ResolvedActiveClientIssue = ActiveClientIssue & {
|
|
2530
|
+
/** When the client resolved it (client clock). */
|
|
2531
|
+
resolvedAt: number;
|
|
2532
|
+
/** Observer-clock resolution time. */
|
|
2533
|
+
observedResolvedAt: number;
|
|
2534
|
+
/** `resolvedAt - raisedAt` as reported by the client, else derived from observer clocks. */
|
|
2535
|
+
durationInMs: number;
|
|
2536
|
+
/** Free-form note passed to `resolveIssue`. */
|
|
2537
|
+
comment?: string;
|
|
2538
|
+
/** Payload explicitly passed at resolution (the built-in detectors pass their final payload). */
|
|
2539
|
+
resolutionPayload?: Record<string, unknown>;
|
|
2540
|
+
/**
|
|
2541
|
+
* How the interval ended: the client said so, the observer expired it, or the client left
|
|
2542
|
+
* without resolving.
|
|
2543
|
+
*/
|
|
2544
|
+
resolvedBy: 'client' | 'timeout' | 'client-closed';
|
|
2545
|
+
};
|
|
2546
|
+
/** `true` when the entry is a resolution companion rather than a raise. */
|
|
2547
|
+
declare function isClientIssueResolutionEntry(issue: ClientIssue): boolean;
|
|
2548
|
+
/** Strip the `-resolved` suffix, so both entries of a lifecycle share one logical type. */
|
|
2549
|
+
declare function baseIssueType(type: string): string;
|
|
2550
|
+
|
|
2039
2551
|
/** The lifecycle events a sink may emit (a subset of Node's writable-stream events). */
|
|
2040
2552
|
type ClientSampleSinkEvents = {
|
|
2041
2553
|
/** The destination is fully written and closed (e.g. a file flushed and its fd closed). */
|
|
@@ -2103,11 +2615,11 @@ type RtpCodecParameters = {
|
|
|
2103
2615
|
parameter?: string;
|
|
2104
2616
|
}[];
|
|
2105
2617
|
};
|
|
2106
|
-
type SampleHistoryItem<T extends string> =
|
|
2618
|
+
type SampleHistoryItem<T extends string> = {
|
|
2107
2619
|
type: T;
|
|
2108
2620
|
timestamp: number;
|
|
2109
2621
|
};
|
|
2110
|
-
type MediasoupRouterSample =
|
|
2622
|
+
type MediasoupRouterSample = {
|
|
2111
2623
|
routerId: string;
|
|
2112
2624
|
attachments: Record<string, unknown>;
|
|
2113
2625
|
createdAt: number;
|
|
@@ -2151,7 +2663,9 @@ type MediasoupDirectTransportSample = {
|
|
|
2151
2663
|
type: 'direct';
|
|
2152
2664
|
history: MediasoupDirectTransportSampleEventMap[];
|
|
2153
2665
|
};
|
|
2154
|
-
type MediasoupTransportSample =
|
|
2666
|
+
type MediasoupTransportSample = {
|
|
2667
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
2668
|
+
attachments?: Record<string, unknown>;
|
|
2155
2669
|
id: string;
|
|
2156
2670
|
createdAt: number;
|
|
2157
2671
|
connectedAt?: number;
|
|
@@ -2167,7 +2681,9 @@ type MediasoupProducerSampleEventMap = {
|
|
|
2167
2681
|
type MediasoupProducerSampleEvent = {
|
|
2168
2682
|
[K in keyof MediasoupProducerSampleEventMap]: SampleHistoryItem<K>;
|
|
2169
2683
|
}[keyof MediasoupProducerSampleEventMap];
|
|
2170
|
-
type MediasoupProducerSample =
|
|
2684
|
+
type MediasoupProducerSample = {
|
|
2685
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
2686
|
+
attachments?: Record<string, unknown>;
|
|
2171
2687
|
id: string;
|
|
2172
2688
|
transportId: string;
|
|
2173
2689
|
createdAt: number;
|
|
@@ -2191,7 +2707,9 @@ type MediasoupConsumerSampleEventMap = {
|
|
|
2191
2707
|
type MediasoupConsumerSampleEvent = {
|
|
2192
2708
|
[K in keyof MediasoupConsumerSampleEventMap]: SampleHistoryItem<K>;
|
|
2193
2709
|
}[keyof MediasoupConsumerSampleEventMap];
|
|
2194
|
-
type MediasoupConsumerSample =
|
|
2710
|
+
type MediasoupConsumerSample = {
|
|
2711
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
2712
|
+
attachments?: Record<string, unknown>;
|
|
2195
2713
|
id: string;
|
|
2196
2714
|
producerId: string;
|
|
2197
2715
|
transportId: string;
|
|
@@ -2200,7 +2718,9 @@ type MediasoupConsumerSample = Record<string, unknown> & {
|
|
|
2200
2718
|
kind: 'audio' | 'video';
|
|
2201
2719
|
history: MediasoupConsumerSampleEvent[];
|
|
2202
2720
|
};
|
|
2203
|
-
type MediasoupDataProducerSample =
|
|
2721
|
+
type MediasoupDataProducerSample = {
|
|
2722
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
2723
|
+
attachments?: Record<string, unknown>;
|
|
2204
2724
|
id: string;
|
|
2205
2725
|
transportId: string;
|
|
2206
2726
|
createdAt: number;
|
|
@@ -2208,7 +2728,9 @@ type MediasoupDataProducerSample = Record<string, unknown> & {
|
|
|
2208
2728
|
label: string;
|
|
2209
2729
|
protocol: string;
|
|
2210
2730
|
};
|
|
2211
|
-
type MediasoupDataConsumerSample =
|
|
2731
|
+
type MediasoupDataConsumerSample = {
|
|
2732
|
+
/** Free-form application data. Attach anything here — see `ObservedMediasoupRouter`. */
|
|
2733
|
+
attachments?: Record<string, unknown>;
|
|
2212
2734
|
id: string;
|
|
2213
2735
|
dataProducerId: string;
|
|
2214
2736
|
transportId: string;
|
|
@@ -2218,13 +2740,86 @@ type MediasoupDataConsumerSample = Record<string, unknown> & {
|
|
|
2218
2740
|
protocol: string;
|
|
2219
2741
|
};
|
|
2220
2742
|
|
|
2743
|
+
/**
|
|
2744
|
+
* Declarative enrichment: return the `attachments` to stamp onto an entity's sample the moment it is
|
|
2745
|
+
* created. Called once per entity, before the corresponding `*-sample-added` event.
|
|
2746
|
+
*
|
|
2747
|
+
* The mediasoup object is handed in, so the common case — mirroring mediasoup's own `appData`, where
|
|
2748
|
+
* applications already keep `participantId`, `purpose` and friends — is a one-liner. Returning
|
|
2749
|
+
* `undefined` attaches nothing.
|
|
2750
|
+
*/
|
|
2751
|
+
type MediasoupSampleEnricher = {
|
|
2752
|
+
transport?: (transport: types.Transport) => Record<string, unknown> | undefined;
|
|
2753
|
+
producer?: (producer: types.Producer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
2754
|
+
consumer?: (consumer: types.Consumer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
2755
|
+
dataProducer?: (dataProducer: types.DataProducer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
2756
|
+
dataConsumer?: (dataConsumer: types.DataConsumer, transport: types.Transport) => Record<string, unknown> | undefined;
|
|
2757
|
+
};
|
|
2221
2758
|
type ObservedMediasoupRouterSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
2222
2759
|
router: types.Router;
|
|
2223
2760
|
appData?: AppData;
|
|
2224
2761
|
attachments?: Record<string, unknown>;
|
|
2762
|
+
/** Stamp `attachments` onto each entity sample as it is created. See {@link MediasoupSampleEnricher}. */
|
|
2763
|
+
enrich?: MediasoupSampleEnricher;
|
|
2225
2764
|
};
|
|
2765
|
+
/**
|
|
2766
|
+
* Lifecycle hooks for building your own report.
|
|
2767
|
+
*
|
|
2768
|
+
* Each entity announces itself as `<entity>-sample-added` when it appears and
|
|
2769
|
+
* `<entity>-sample-closed` when it goes away, carrying **the live sample object** plus the mediasoup
|
|
2770
|
+
* object it came from. Mutating `sample.attachments` inside a handler is the intended way to extend
|
|
2771
|
+
* a sample on the fly — the object you receive is the one held in `observedRouter.sample`, not a copy.
|
|
2772
|
+
*/
|
|
2226
2773
|
type ObservedMediasoupRouterEvents = {
|
|
2227
2774
|
close: [];
|
|
2775
|
+
'transport-sample-added': [{
|
|
2776
|
+
sample: MediasoupTransportSample;
|
|
2777
|
+
transport: types.Transport;
|
|
2778
|
+
}];
|
|
2779
|
+
'transport-sample-closed': [{
|
|
2780
|
+
sample: MediasoupTransportSample;
|
|
2781
|
+
transport: types.Transport;
|
|
2782
|
+
}];
|
|
2783
|
+
'producer-sample-added': [{
|
|
2784
|
+
sample: MediasoupProducerSample;
|
|
2785
|
+
producer: types.Producer;
|
|
2786
|
+
transport: types.Transport;
|
|
2787
|
+
}];
|
|
2788
|
+
'producer-sample-closed': [{
|
|
2789
|
+
sample: MediasoupProducerSample;
|
|
2790
|
+
producer: types.Producer;
|
|
2791
|
+
transport: types.Transport;
|
|
2792
|
+
}];
|
|
2793
|
+
'consumer-sample-added': [{
|
|
2794
|
+
sample: MediasoupConsumerSample;
|
|
2795
|
+
consumer: types.Consumer;
|
|
2796
|
+
transport: types.Transport;
|
|
2797
|
+
}];
|
|
2798
|
+
'consumer-sample-closed': [{
|
|
2799
|
+
sample: MediasoupConsumerSample;
|
|
2800
|
+
consumer: types.Consumer;
|
|
2801
|
+
transport: types.Transport;
|
|
2802
|
+
}];
|
|
2803
|
+
'data-producer-sample-added': [{
|
|
2804
|
+
sample: MediasoupDataProducerSample;
|
|
2805
|
+
dataProducer: types.DataProducer;
|
|
2806
|
+
transport: types.Transport;
|
|
2807
|
+
}];
|
|
2808
|
+
'data-producer-sample-closed': [{
|
|
2809
|
+
sample: MediasoupDataProducerSample;
|
|
2810
|
+
dataProducer: types.DataProducer;
|
|
2811
|
+
transport: types.Transport;
|
|
2812
|
+
}];
|
|
2813
|
+
'data-consumer-sample-added': [{
|
|
2814
|
+
sample: MediasoupDataConsumerSample;
|
|
2815
|
+
dataConsumer: types.DataConsumer;
|
|
2816
|
+
transport: types.Transport;
|
|
2817
|
+
}];
|
|
2818
|
+
'data-consumer-sample-closed': [{
|
|
2819
|
+
sample: MediasoupDataConsumerSample;
|
|
2820
|
+
dataConsumer: types.DataConsumer;
|
|
2821
|
+
transport: types.Transport;
|
|
2822
|
+
}];
|
|
2228
2823
|
};
|
|
2229
2824
|
declare interface ObservedMediasoupRouter {
|
|
2230
2825
|
on<U extends keyof ObservedMediasoupRouterEvents>(event: U, listener: (...args: ObservedMediasoupRouterEvents[U]) => void): this;
|
|
@@ -2250,9 +2845,36 @@ declare class ObservedMediasoupRouter<AppData extends Record<string, unknown> =
|
|
|
2250
2845
|
readonly sample: MediasoupRouterSample;
|
|
2251
2846
|
readonly webrtcTransportIds: Set<string>;
|
|
2252
2847
|
closed: boolean;
|
|
2848
|
+
private readonly _transportSamples;
|
|
2849
|
+
private readonly _producerSamples;
|
|
2850
|
+
private readonly _consumerSamples;
|
|
2851
|
+
private readonly _dataProducerSamples;
|
|
2852
|
+
private readonly _dataConsumerSamples;
|
|
2853
|
+
private readonly _enrich?;
|
|
2253
2854
|
constructor(settings: ObservedMediasoupRouterSettings<AppData>);
|
|
2254
2855
|
get id(): string;
|
|
2255
2856
|
get attachments(): Record<string, unknown>;
|
|
2857
|
+
getTransportSample(id: string): MediasoupTransportSample | undefined;
|
|
2858
|
+
getProducerSample(id: string): MediasoupProducerSample | undefined;
|
|
2859
|
+
getConsumerSample(id: string): MediasoupConsumerSample | undefined;
|
|
2860
|
+
getDataProducerSample(id: string): MediasoupDataProducerSample | undefined;
|
|
2861
|
+
getDataConsumerSample(id: string): MediasoupDataConsumerSample | undefined;
|
|
2862
|
+
/**
|
|
2863
|
+
* Merge `attachments` into an entity's sample, whichever kind it is.
|
|
2864
|
+
*
|
|
2865
|
+
* Ids are unique across mediasoup entity kinds, so one method covers all of them. Returns `false`
|
|
2866
|
+
* when the id is unknown — a real answer instead of failing quietly, which matters when the
|
|
2867
|
+
* annotation is driven by application events that may race the mediasoup ones.
|
|
2868
|
+
*/
|
|
2869
|
+
attachTo(id: string, attachments: Record<string, unknown>): boolean;
|
|
2870
|
+
/**
|
|
2871
|
+
* A **detached deep copy** of the current sample — the basis for building your own report.
|
|
2872
|
+
*
|
|
2873
|
+
* `this.sample` is live: its arrays grow and its `history` entries are appended as the router
|
|
2874
|
+
* runs, so a report built directly on it keeps changing after you think you're done. This returns
|
|
2875
|
+
* a snapshot that never moves.
|
|
2876
|
+
*/
|
|
2877
|
+
snapshot(): MediasoupRouterSample;
|
|
2256
2878
|
close(): void;
|
|
2257
2879
|
addTransport: (transport: types.Transport) => void;
|
|
2258
2880
|
addWebRtcTransport(transport: types.WebRtcTransport): void;
|
|
@@ -2264,6 +2886,18 @@ declare class ObservedMediasoupRouter<AppData extends Record<string, unknown> =
|
|
|
2264
2886
|
addDataProducer(transport: types.Transport, dataProducer: types.DataProducer): void;
|
|
2265
2887
|
addDataConsumer(transport: types.Transport, dataConsumer: types.DataConsumer): void;
|
|
2266
2888
|
private attachRouterListeners;
|
|
2889
|
+
private _addTransportSample;
|
|
2890
|
+
private _addProducerSample;
|
|
2891
|
+
private _addConsumerSample;
|
|
2892
|
+
private _addDataProducerSample;
|
|
2893
|
+
private _addDataConsumerSample;
|
|
2894
|
+
/**
|
|
2895
|
+
* Run an enricher and merge what it returns.
|
|
2896
|
+
*
|
|
2897
|
+
* Takes a thunk rather than a value so the **invocation** is inside the guard — application code
|
|
2898
|
+
* runs here, and a throwing enricher must not take the router's bookkeeping down with it.
|
|
2899
|
+
*/
|
|
2900
|
+
private _applyEnrichment;
|
|
2267
2901
|
private _attachTransportObserverListeners;
|
|
2268
2902
|
}
|
|
2269
2903
|
|
|
@@ -2304,6 +2938,18 @@ type ObserverEvents = {
|
|
|
2304
2938
|
reason: SampleRejectedReason;
|
|
2305
2939
|
sample: ClientSample;
|
|
2306
2940
|
}];
|
|
2941
|
+
/** An observer-scoped (cross-call / SFU-wide) finding raised by `observer.addIssue(...)`. */
|
|
2942
|
+
'observer-issue': [ObserverEventBase & {
|
|
2943
|
+
issue: ObserverIssue;
|
|
2944
|
+
}];
|
|
2945
|
+
/**
|
|
2946
|
+
* A validator decided. Fires once per settle, not per tick — the point of a validator is that it
|
|
2947
|
+
* stops talking once it knows.
|
|
2948
|
+
*/
|
|
2949
|
+
'validation-ready': [ObserverEventBase & {
|
|
2950
|
+
validator: string;
|
|
2951
|
+
report: ValidationReport;
|
|
2952
|
+
}];
|
|
2307
2953
|
'mediasoup-router-added': [ObservedMediasoupRouterScope];
|
|
2308
2954
|
'mediasoup-router-removed': [ObservedMediasoupRouterScope];
|
|
2309
2955
|
'mediasoup-router-matched-with-peer-connection': [ObservedMediasoupRouterScope & ObservedPeerConnectionScope];
|
|
@@ -2313,7 +2959,7 @@ type ObserverEvents = {
|
|
|
2313
2959
|
'call-empty': [ObservedCallScope];
|
|
2314
2960
|
'call-not-empty': [ObservedCallScope];
|
|
2315
2961
|
'call-issue': [ObservedCallScope & {
|
|
2316
|
-
issue:
|
|
2962
|
+
issue: ObserverIssue;
|
|
2317
2963
|
}];
|
|
2318
2964
|
'client-added': [ObservedClientScope];
|
|
2319
2965
|
'client-sink-created': [ObservedClientScope & {
|
|
@@ -2332,6 +2978,14 @@ type ObserverEvents = {
|
|
|
2332
2978
|
'client-issue': [ObservedClientScope & {
|
|
2333
2979
|
issue: ClientIssue;
|
|
2334
2980
|
}];
|
|
2981
|
+
/**
|
|
2982
|
+
* A stateful client issue ended — the client sent its `<type>-resolved` companion, or the
|
|
2983
|
+
* observer force-closed it because the client went away. Carries the finished **interval**
|
|
2984
|
+
* (`raisedAt` → `resolvedAt`, `durationInMs`).
|
|
2985
|
+
*/
|
|
2986
|
+
'client-issue-resolved': [ObservedClientScope & {
|
|
2987
|
+
resolvedIssue: ResolvedActiveClientIssue;
|
|
2988
|
+
}];
|
|
2335
2989
|
'client-metadata': [ObservedClientScope & {
|
|
2336
2990
|
metaData: ClientMetaData;
|
|
2337
2991
|
}];
|
|
@@ -2503,6 +3157,61 @@ type ObserverEvents = {
|
|
|
2503
3157
|
}];
|
|
2504
3158
|
};
|
|
2505
3159
|
|
|
3160
|
+
/**
|
|
3161
|
+
* Something that wants to be **handed** open client issues rather than to go looking for them.
|
|
3162
|
+
*
|
|
3163
|
+
* Register one with `activeIssuesRegistry.addIssueTracker(type, tracker)` and it receives every
|
|
3164
|
+
* issue of that type when it opens ({@link add}) and when it closes ({@link delete}). A detector
|
|
3165
|
+
* implementing this pays only for the issues it actually consumes.
|
|
3166
|
+
*
|
|
3167
|
+
* `ActiveIssuesRegistry` implements it too, which is how a call's registry feeds the observer's.
|
|
3168
|
+
*/
|
|
3169
|
+
interface ActiveIssueTracker {
|
|
3170
|
+
/** An issue of a subscribed type opened. */
|
|
3171
|
+
add(issue: ActiveClientIssue): void;
|
|
3172
|
+
/**
|
|
3173
|
+
* An issue this tracker was given has closed.
|
|
3174
|
+
*
|
|
3175
|
+
* Return `true` if it was actually held. Returning `false` is legitimate and not an error — a
|
|
3176
|
+
* tracker that counts *occurrences* (see `SfuCongestionDetector`) deliberately ignores
|
|
3177
|
+
* resolutions, because when a symptom ended says nothing about how many endpoints reported it.
|
|
3178
|
+
*/
|
|
3179
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3180
|
+
/** How many issues this tracker currently holds. */
|
|
3181
|
+
size: number;
|
|
3182
|
+
/** Drop everything. Called when the owning scope closes. */
|
|
3183
|
+
clear(): void;
|
|
3184
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3185
|
+
}
|
|
3186
|
+
|
|
3187
|
+
/**
|
|
3188
|
+
* One client's currently **open** stateful issues, keyed by `ClientIssue.key`.
|
|
3189
|
+
*
|
|
3190
|
+
* The server-side mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0):
|
|
3191
|
+
* a raise opens an entry, the matching `<type>-resolved` closes it, and the client's own close
|
|
3192
|
+
* force-resolves whatever is left. That turns point-in-time symptom reports into **intervals**,
|
|
3193
|
+
* which is what lets detectors ask "are these clients broken *at the same time*" rather than "did
|
|
3194
|
+
* they both report something recently".
|
|
3195
|
+
*
|
|
3196
|
+
* Keyed by `key` rather than by type on purpose: one client can have several issues of the same type
|
|
3197
|
+
* open at once (one per track), and they resolve independently.
|
|
3198
|
+
*
|
|
3199
|
+
* Every change is forwarded to the call's `ActiveIssuesRegistry`, which forwards to the observer's —
|
|
3200
|
+
* so the client owns the storage and the wider scopes get their views maintained as it happens.
|
|
3201
|
+
*/
|
|
3202
|
+
declare class ObservedClientIssueRegistry {
|
|
3203
|
+
private readonly registry?;
|
|
3204
|
+
private readonly issues;
|
|
3205
|
+
constructor(registry?: ActiveIssueTracker | undefined);
|
|
3206
|
+
get size(): number;
|
|
3207
|
+
keys(): IterableIterator<string>;
|
|
3208
|
+
values(): IterableIterator<ActiveClientIssue>;
|
|
3209
|
+
get(key: string): ActiveClientIssue | undefined;
|
|
3210
|
+
add(issue: ActiveClientIssue): this;
|
|
3211
|
+
remove(key: string): ActiveClientIssue | undefined;
|
|
3212
|
+
clear(): void;
|
|
3213
|
+
}
|
|
3214
|
+
|
|
2506
3215
|
type ObservedClientSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
2507
3216
|
clientId: string;
|
|
2508
3217
|
appData?: AppData;
|
|
@@ -2581,11 +3290,20 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
|
|
|
2581
3290
|
totalScoreSum: number;
|
|
2582
3291
|
numberOfScoreMeasurements: number;
|
|
2583
3292
|
readonly mediaDevices: MediaDeviceInfo[];
|
|
2584
|
-
|
|
3293
|
+
/**
|
|
3294
|
+
* The client's currently **open** stateful issues, keyed by `ClientIssue.key` — the server-side
|
|
3295
|
+
* mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0).
|
|
3296
|
+
*
|
|
3297
|
+
* This turns point-in-time symptom reports into intervals, which is what lets detectors ask
|
|
3298
|
+
* "are these clients broken *at the same time*" instead of "did they both report something
|
|
3299
|
+
* recently". Entries are opened by a raise, closed by the matching `<type>-resolved` entry, and
|
|
3300
|
+
* force-closed when the client closes.
|
|
3301
|
+
*/
|
|
3302
|
+
readonly activeIssues: ObservedClientIssueRegistry;
|
|
2585
3303
|
private _pendingInjections;
|
|
2586
3304
|
private _activeSample?;
|
|
2587
3305
|
private closeTimer?;
|
|
2588
|
-
constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall);
|
|
3306
|
+
constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall, activeIssues: ObservedClientIssueRegistry);
|
|
2589
3307
|
get numberOfPeerConnections(): number;
|
|
2590
3308
|
get score(): number | undefined;
|
|
2591
3309
|
close(): void;
|
|
@@ -2596,8 +3314,22 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
|
|
|
2596
3314
|
injectExtensionStat(stat: ExtensionStat): void;
|
|
2597
3315
|
injectAttachment(attachments: Record<string, unknown>): void;
|
|
2598
3316
|
addMetadata(metadata: ClientMetaData): void;
|
|
3317
|
+
/**
|
|
3318
|
+
* Process one `clientIssues[]` entry.
|
|
3319
|
+
*
|
|
3320
|
+
* Entries come in two flavours (client-monitor-js >= 4.6.0):
|
|
3321
|
+
*
|
|
3322
|
+
* - a **raise** — opens an {@link ActiveClientIssue} under `issue.key` and emits `client-issue`;
|
|
3323
|
+
* - a **resolution** — `type` ends in `-resolved` and carries the same `key`; it closes the
|
|
3324
|
+
* matching active issue and emits `client-issue-resolved`.
|
|
3325
|
+
*
|
|
3326
|
+
* Keyless entries are one-shot: reported, never tracked. A re-raise of a key already active
|
|
3327
|
+
* refreshes the payload rather than opening a second interval.
|
|
3328
|
+
*/
|
|
2599
3329
|
addIssue(issue: ClientIssue): void;
|
|
2600
3330
|
addExtensionStats(stats: ExtensionStat): void;
|
|
3331
|
+
/** Close the active issue a `<type>-resolved` entry refers to, and announce the finished interval. */
|
|
3332
|
+
private _resolveIssue;
|
|
2601
3333
|
private _processClientEvent;
|
|
2602
3334
|
private _updatePeerConnection;
|
|
2603
3335
|
private _mergePendingInjections;
|
|
@@ -2655,82 +3387,1192 @@ declare class RemoteTrackResolver {
|
|
|
2655
3387
|
private _removeOutboundTrack;
|
|
2656
3388
|
}
|
|
2657
3389
|
|
|
2658
|
-
interface Updater {
|
|
2659
|
-
readonly name: string;
|
|
2660
|
-
readonly description?: string;
|
|
2661
|
-
close(): void;
|
|
2662
|
-
}
|
|
2663
|
-
|
|
2664
3390
|
interface Detector {
|
|
2665
3391
|
readonly name: string;
|
|
2666
3392
|
/** Called on every entity update; may raise issues via the entity it observes. */
|
|
2667
3393
|
update(): void;
|
|
3394
|
+
/**
|
|
3395
|
+
* Optional teardown, called when the detector is removed from its registry (or the registry is
|
|
3396
|
+
* cleared, which happens when the owning call/observer closes). Implement it when the detector
|
|
3397
|
+
* subscribes to events or holds timers, so it doesn't leak.
|
|
3398
|
+
*/
|
|
3399
|
+
close?(): void;
|
|
2668
3400
|
}
|
|
2669
3401
|
|
|
2670
|
-
declare
|
|
2671
|
-
|
|
2672
|
-
|
|
2673
|
-
|
|
2674
|
-
|
|
2675
|
-
|
|
2676
|
-
|
|
3402
|
+
declare const CallConcurrentIssueTypes: {
|
|
3403
|
+
/** Several participants of this call have the same issue open **at the same time**. */
|
|
3404
|
+
readonly concurrentClientIssues: "CONCURRENT_CLIENT_ISSUES";
|
|
3405
|
+
/** Those issues also *began* together — the signature of one shared event. */
|
|
3406
|
+
readonly issueOnsetBurst: "ISSUE_ONSET_BURST";
|
|
3407
|
+
};
|
|
3408
|
+
type CallConcurrentIssueDetectorConfig = {
|
|
3409
|
+
/**
|
|
3410
|
+
* The issue types to watch. **Required, and must not be empty** — the detector subscribes to
|
|
3411
|
+
* exactly these and sees nothing else.
|
|
3412
|
+
*/
|
|
3413
|
+
issueTypes: string[];
|
|
3414
|
+
/** Minimum participants in the call before a ratio is meaningful. Default `3`. */
|
|
3415
|
+
minClients: number;
|
|
3416
|
+
/** Minimum distinct clients sharing the open issue. Default `3`. */
|
|
3417
|
+
minAffectedClients: number;
|
|
3418
|
+
/** Fraction of the call's participants that must share it. Default `0.5`. */
|
|
3419
|
+
affectedRatioThreshold: number;
|
|
3420
|
+
/**
|
|
3421
|
+
* When the onsets fall within this span (ms), the finding is escalated to `ISSUE_ONSET_BURST` —
|
|
3422
|
+
* they didn't just overlap, they started together. Default `2_000`.
|
|
3423
|
+
*/
|
|
3424
|
+
onsetBurstWindowInMs: number;
|
|
3425
|
+
/** Re-arm time (ms) per issue type. Default `60_000`. */
|
|
3426
|
+
cooldownMs: number;
|
|
3427
|
+
};
|
|
3428
|
+
/** What the detector currently knows about one issue type in this call. */
|
|
3429
|
+
type CallConcurrentIssueGroup = {
|
|
3430
|
+
type: string;
|
|
3431
|
+
issues: ActiveClientIssue[];
|
|
3432
|
+
clientIds: string[];
|
|
3433
|
+
affectedRatio: number;
|
|
3434
|
+
totalClients: number;
|
|
3435
|
+
/**
|
|
3436
|
+
* Spread of the onsets, in **observer** time (ms) — `max(observedAt) - min(observedAt)`.
|
|
3437
|
+
*
|
|
3438
|
+
* Measured on the observer clock on purpose: `raisedAt` comes from each client's own clock, and
|
|
3439
|
+
* comparing those across machines makes clock skew look like a shared event.
|
|
3440
|
+
*/
|
|
3441
|
+
onsetSpreadInMs: number;
|
|
3442
|
+
firstObservedAt: number;
|
|
3443
|
+
};
|
|
3444
|
+
/**
|
|
3445
|
+
* Answers **"is this meeting in trouble?"** — several participants of one call with the same issue
|
|
3446
|
+
* open simultaneously.
|
|
3447
|
+
*
|
|
3448
|
+
* The client already decides *what* is wrong for itself — `congestion`, `ice-disconnected`,
|
|
3449
|
+
* `audio-concealment`, `video-decoder-overloaded` — with hysteresis and multi-signal confirmation
|
|
3450
|
+
* behind each verdict. Re-deriving those server-side from raw counters would be strictly worse. What
|
|
3451
|
+
* the server uniquely knows is *how many other participants of the same call are in that state right
|
|
3452
|
+
* now*, which is the difference between "one person's Wi-Fi" and "this room is broken".
|
|
3453
|
+
*
|
|
3454
|
+
* Concurrency is judged from the **open interval set**, not a window of recent reports. A window has
|
|
3455
|
+
* to guess whether a symptom is still happening; an interval is closed by the client when the episode
|
|
3456
|
+
* actually ends (client-monitor-js >= 4.6.0 ships the `<type>-resolved` companion for exactly this).
|
|
3457
|
+
*
|
|
3458
|
+
* ```ts
|
|
3459
|
+
* observedCall.addDetector('call-concurrent-issue-detector', {
|
|
3460
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
3461
|
+
* });
|
|
3462
|
+
* ```
|
|
3463
|
+
*
|
|
3464
|
+
* For the cross-call version of this question — which is a different question, not this one with a
|
|
3465
|
+
* bigger denominator — see `ObserverConcurrentIssueDetector`.
|
|
3466
|
+
*/
|
|
3467
|
+
declare class CallConcurrentIssueDetector implements Detector, ActiveIssueTracker {
|
|
3468
|
+
private readonly _call;
|
|
3469
|
+
static readonly NAME: "call-concurrent-issue-detector";
|
|
3470
|
+
readonly name: "call-concurrent-issue-detector";
|
|
3471
|
+
private readonly _config;
|
|
3472
|
+
private readonly _lastRaisedAt;
|
|
3473
|
+
/** issue type -> the issues of that type currently open in this call. */
|
|
3474
|
+
private readonly _byType;
|
|
3475
|
+
private _size;
|
|
3476
|
+
/** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
|
|
3477
|
+
lastGroups: CallConcurrentIssueGroup[];
|
|
3478
|
+
constructor(_call: ObservedCall, config?: Partial<CallConcurrentIssueDetectorConfig>);
|
|
3479
|
+
get size(): number;
|
|
3480
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3481
|
+
add(issue: ActiveClientIssue): void;
|
|
3482
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
2677
3483
|
clear(): void;
|
|
3484
|
+
update(): void;
|
|
3485
|
+
close(): void;
|
|
3486
|
+
private _groupOf;
|
|
2678
3487
|
}
|
|
2679
3488
|
|
|
2680
|
-
|
|
2681
|
-
|
|
3489
|
+
declare const ClientPopulationIssueTypes: {
|
|
3490
|
+
/** One issue type is concentrated on one client population while the rest of the fleet is fine. */
|
|
3491
|
+
readonly clientPopulationIssue: "CLIENT_POPULATION_ISSUE";
|
|
2682
3492
|
};
|
|
2683
|
-
|
|
2684
|
-
|
|
2685
|
-
|
|
2686
|
-
|
|
3493
|
+
/** The client attribute to group by. One axis per detector — see the class description. */
|
|
3494
|
+
type ClientPopulationAxis = 'browser' | 'engine' | 'platform' | 'operationSystem';
|
|
3495
|
+
type ClientPopulationIssueDetectorConfig = {
|
|
3496
|
+
/**
|
|
3497
|
+
* The issue types to watch. **Required, and must not be empty.**
|
|
3498
|
+
*
|
|
3499
|
+
* The types worth grouping this way are the ones an endpoint owns: `cpulimitation`,
|
|
3500
|
+
* `encoder-bottleneck`, `capture-bottleneck`, `stuck-decoder`, `video-decoder-overloaded`.
|
|
3501
|
+
* Grouping a *network* symptom by browser is a category error — `congestion` clusters by ISP and
|
|
3502
|
+
* geography, neither of which this detector can see, and it would happily report a browser
|
|
3503
|
+
* correlation that is really a "most of our users are on Chrome" artefact.
|
|
3504
|
+
*/
|
|
3505
|
+
issueTypes: string[];
|
|
3506
|
+
/** Which client attribute to group by. Default `'browser'`. */
|
|
3507
|
+
groupBy: ClientPopulationAxis;
|
|
3508
|
+
/**
|
|
3509
|
+
* Group by `name` only, or by `name + version`. Default `true` (include version).
|
|
3510
|
+
*
|
|
3511
|
+
* Version is usually the point: "Chrome" is not actionable, "Chrome 141" is, because it names a
|
|
3512
|
+
* thing that changed on a date. Set to `false` when comparing whole engines.
|
|
3513
|
+
*/
|
|
3514
|
+
includeVersion: boolean;
|
|
3515
|
+
/** Minimum clients in a population before its rate means anything. Default `20`. */
|
|
3516
|
+
minPopulationSize: number;
|
|
3517
|
+
/** Minimum affected clients in the population. Default `5`. */
|
|
3518
|
+
minAffectedClients: number;
|
|
3519
|
+
/** Share of the population that must be affected. Default `0.3`. */
|
|
3520
|
+
affectedRatioThreshold: number;
|
|
3521
|
+
/**
|
|
3522
|
+
* How many times worse the suspect population must be than the rest of the fleet. Default `3`.
|
|
3523
|
+
*
|
|
3524
|
+
* This is the control, and it is what makes the finding mean anything. See the class description.
|
|
3525
|
+
*/
|
|
3526
|
+
minRelativeRisk: number;
|
|
3527
|
+
/** Minimum clients **outside** the suspect population before a comparison is possible. Default `20`. */
|
|
3528
|
+
minControlSize: number;
|
|
3529
|
+
/** Re-arm time (ms) per (population, issue type). Default `300_000`. */
|
|
3530
|
+
cooldownMs: number;
|
|
2687
3531
|
};
|
|
2688
|
-
|
|
2689
|
-
|
|
2690
|
-
|
|
2691
|
-
|
|
2692
|
-
|
|
2693
|
-
|
|
3532
|
+
/** The rollup for one population on one issue type. */
|
|
3533
|
+
type ClientPopulation = {
|
|
3534
|
+
/** e.g. `'Chrome 141'`, or `'Chrome'` when `includeVersion` is off. */
|
|
3535
|
+
population: string;
|
|
3536
|
+
axis: ClientPopulationAxis;
|
|
3537
|
+
issueType: string;
|
|
3538
|
+
clients: number;
|
|
3539
|
+
affectedClients: number;
|
|
3540
|
+
affectedRatio: number;
|
|
3541
|
+
affectedClientIds: string[];
|
|
3542
|
+
/** Everyone not in this population. */
|
|
3543
|
+
controlClients: number;
|
|
3544
|
+
controlAffectedClients: number;
|
|
3545
|
+
controlAffectedRatio: number;
|
|
3546
|
+
/** `affectedRatio / controlAffectedRatio`. `Infinity` when the control group is completely clean. */
|
|
3547
|
+
relativeRisk: number;
|
|
2694
3548
|
};
|
|
2695
|
-
|
|
2696
|
-
|
|
2697
|
-
|
|
2698
|
-
|
|
2699
|
-
|
|
3549
|
+
/**
|
|
3550
|
+
* Finds an issue that is concentrated on **one kind of client** — one browser, one browser version,
|
|
3551
|
+
* one OS — rather than on anything the servers own.
|
|
3552
|
+
*
|
|
3553
|
+
* ### Why this exists
|
|
3554
|
+
*
|
|
3555
|
+
* The other observer-scoped detectors all answer "who else has this open, and what do they share?"
|
|
3556
|
+
* with the answer *the infrastructure*, because clients in unrelated calls share nothing else. That
|
|
3557
|
+
* inference is right for network symptoms and **wrong for endpoint symptoms**, and the difference
|
|
3558
|
+
* matters at 3am. `cpulimitation` opening across six unrelated calls is not an SFU event: CPU is
|
|
3559
|
+
* owned by the endpoint, so what those endpoints have in common is a client release, a browser
|
|
3560
|
+
* update, or a fleet of identical VDI hosts. `IssueConclusion` already says exactly this — it maps
|
|
3561
|
+
* the endpoint-capacity family to a `client-population` fault domain instead of `infrastructure` —
|
|
3562
|
+
* but until now nothing in the library actually computed the grouping that claim refers to. This
|
|
3563
|
+
* detector is that computation.
|
|
3564
|
+
*
|
|
3565
|
+
* It is the one correlation in this library that is neither per-call nor per-server. A client knows
|
|
3566
|
+
* its own browser and nothing about anyone else's; only something sitting above the whole fleet can
|
|
3567
|
+
* notice that every complaint is coming from the same build.
|
|
3568
|
+
*
|
|
3569
|
+
* ### The control group is the whole point
|
|
3570
|
+
*
|
|
3571
|
+
* "30% of Chrome 141 users report encoder-bottleneck" is not a finding on its own. If 30% of
|
|
3572
|
+
* *everyone* reports it, Chrome 141 is not the story — you have a fleet-wide problem and this
|
|
3573
|
+
* detector would be pointing at the largest population rather than at a cause. Naive share-based
|
|
3574
|
+
* grouping always indicts whichever browser is most popular, which is why the gate here is
|
|
3575
|
+
* **relative risk**: the suspect population's rate divided by the rate among everyone else. A
|
|
3576
|
+
* population only qualifies when it is `minRelativeRisk` times worse than the rest of the fleet, and
|
|
3577
|
+
* only when the rest of the fleet is large enough (`minControlSize`) for "the rest of the fleet" to
|
|
3578
|
+
* be a real measurement.
|
|
3579
|
+
*
|
|
3580
|
+
* A completely clean control group gives `Infinity`, which is honest — nobody outside this
|
|
3581
|
+
* population has the problem at all — and is exactly why `minAffectedClients` and
|
|
3582
|
+
* `minPopulationSize` are checked independently, so a single unlucky user on a rare browser cannot
|
|
3583
|
+
* page anyone.
|
|
3584
|
+
*
|
|
3585
|
+
* ### One axis per detector
|
|
3586
|
+
*
|
|
3587
|
+
* `groupBy` takes a single attribute. Add a second instance if you want a second axis:
|
|
3588
|
+
*
|
|
3589
|
+
* ```ts
|
|
3590
|
+
* observer.addObserverDetector('client-population-issue-detector', {
|
|
3591
|
+
* issueTypes: [ 'cpulimitation', 'encoder-bottleneck', 'stuck-decoder' ],
|
|
3592
|
+
* groupBy: 'browser',
|
|
3593
|
+
* });
|
|
3594
|
+
*
|
|
3595
|
+
* observer.on('observer-issue', ({ issue }) => {
|
|
3596
|
+
* if (issue.type !== ClientPopulationIssueTypes.clientPopulationIssue) return;
|
|
3597
|
+
* // → { population: 'Chrome 141', issueType: 'encoder-bottleneck',
|
|
3598
|
+
* // affectedRatio: 0.34, controlAffectedRatio: 0.02, relativeRisk: 17 }
|
|
3599
|
+
* });
|
|
3600
|
+
* ```
|
|
3601
|
+
*
|
|
3602
|
+
* Deliberately not a cross-product of every axis at once: an issue that clusters on macOS *and* on
|
|
3603
|
+
* Safari is usually one fact reported twice, and a detector that emits both leaves the reader to
|
|
3604
|
+
* work out which one is causal. Pick the axis you want to reason about.
|
|
3605
|
+
*
|
|
3606
|
+
* ### Clients that never reported their metadata
|
|
3607
|
+
*
|
|
3608
|
+
* `browser` / `engine` / `platform` / `operationSystem` arrive as client metadata and may be absent —
|
|
3609
|
+
* a client that closed before sending them, or an application that does not collect them. Those
|
|
3610
|
+
* clients are excluded from **both** the population and the control group rather than bucketed as
|
|
3611
|
+
* `'unknown'`. A synthetic `'unknown'` population would be a mixture of every real one, so any rate
|
|
3612
|
+
* computed for it means nothing, and leaving those clients in the control group would dilute the
|
|
3613
|
+
* comparison with clients whose kind we cannot verify.
|
|
3614
|
+
*/
|
|
3615
|
+
declare class ClientPopulationIssueDetector implements Detector, ActiveIssueTracker {
|
|
3616
|
+
private readonly _observer;
|
|
3617
|
+
static readonly NAME: "client-population-issue-detector";
|
|
3618
|
+
readonly name: "client-population-issue-detector";
|
|
3619
|
+
private readonly _config;
|
|
3620
|
+
private readonly _lastRaisedAt;
|
|
3621
|
+
private readonly _issues;
|
|
3622
|
+
/** The populations that qualified on the most recent `update()`. Exposed for tests/dashboards. */
|
|
3623
|
+
lastPopulations: ClientPopulation[];
|
|
3624
|
+
constructor(_observer: Observer, config?: Partial<ClientPopulationIssueDetectorConfig>);
|
|
3625
|
+
get size(): number;
|
|
3626
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3627
|
+
add(issue: ActiveClientIssue): void;
|
|
3628
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3629
|
+
clear(): void;
|
|
3630
|
+
update(): void;
|
|
3631
|
+
close(): void;
|
|
3632
|
+
/** `undefined` when the client never reported this attribute — see the class description. */
|
|
3633
|
+
private _populationOf;
|
|
3634
|
+
private _rollupOf;
|
|
3635
|
+
private _riskText;
|
|
3636
|
+
/** Bigger populations and starker contrasts are harder to produce by chance. */
|
|
3637
|
+
private _confidenceOf;
|
|
2700
3638
|
}
|
|
2701
|
-
|
|
2702
|
-
|
|
2703
|
-
|
|
2704
|
-
|
|
2705
|
-
|
|
2706
|
-
|
|
2707
|
-
readonly
|
|
2708
|
-
|
|
2709
|
-
|
|
2710
|
-
|
|
2711
|
-
|
|
2712
|
-
|
|
2713
|
-
|
|
2714
|
-
|
|
2715
|
-
|
|
2716
|
-
|
|
2717
|
-
|
|
2718
|
-
|
|
2719
|
-
|
|
2720
|
-
|
|
2721
|
-
|
|
3639
|
+
|
|
3640
|
+
declare const PublisherFaultTypes: {
|
|
3641
|
+
/**
|
|
3642
|
+
* A publisher is reporting trouble on its own send path **while** its subscribers report trouble
|
|
3643
|
+
* receiving it. Both ends agree, so the source is implicated rather than inferred.
|
|
3644
|
+
*/
|
|
3645
|
+
readonly corroboratedPublisherFault: "CORROBORATED_PUBLISHER_FAULT";
|
|
3646
|
+
};
|
|
3647
|
+
type PublisherFaultCorroborationDetectorConfig = {
|
|
3648
|
+
/**
|
|
3649
|
+
* Issue types raised by the **publishing** client about its own outbound path. **Required.**
|
|
3650
|
+
*
|
|
3651
|
+
* The natural set from client-monitor-js: `encoder-bottleneck`, `capture-bottleneck`,
|
|
3652
|
+
* `dry-outbound-track`. All three mean "I am failing to produce or send this properly", which is
|
|
3653
|
+
* the half of the story the receivers cannot see.
|
|
3654
|
+
*/
|
|
3655
|
+
publisherIssueTypes: string[];
|
|
3656
|
+
/**
|
|
3657
|
+
* Issue types raised by the **subscribing** clients about the track they receive. **Required.**
|
|
3658
|
+
*
|
|
3659
|
+
* The natural set: `freezed-video-track`, `dry-inbound-track`, `video-recovery-failed`. These say
|
|
3660
|
+
* "I am not getting this properly", which is the half the publisher cannot see.
|
|
3661
|
+
*/
|
|
3662
|
+
receiverIssueTypes: string[];
|
|
3663
|
+
/** Minimum subscribers of the track that must be complaining. Default `2`. */
|
|
3664
|
+
minAffectedReceivers: number;
|
|
3665
|
+
/** Re-arm time (ms) per published track. Default `60_000`. */
|
|
3666
|
+
cooldownMs: number;
|
|
3667
|
+
};
|
|
3668
|
+
/** The two-sided evidence behind one finding. */
|
|
3669
|
+
type CorroboratedPublisherFault = {
|
|
3670
|
+
trackId: string;
|
|
3671
|
+
kind: string;
|
|
3672
|
+
publisherClientId: string;
|
|
3673
|
+
/** The publisher's own open issue types on this track. */
|
|
3674
|
+
publisherIssueTypes: string[];
|
|
3675
|
+
/** The receiver-side open issue types across this track's subscribers. */
|
|
3676
|
+
receiverIssueTypes: string[];
|
|
3677
|
+
receivers: number;
|
|
3678
|
+
affectedReceivers: number;
|
|
3679
|
+
affectedClientIds: string[];
|
|
3680
|
+
publisherBitrate?: number;
|
|
3681
|
+
};
|
|
3682
|
+
/**
|
|
3683
|
+
* Fires only when **both ends of one published track are complaining at the same time**: the
|
|
3684
|
+
* publisher about its own send path, and its subscribers about receiving it.
|
|
3685
|
+
*
|
|
3686
|
+
* ### How this differs from `IssueFanOutDetector`
|
|
3687
|
+
*
|
|
3688
|
+
* Fan-out sees one end. It observes that most of Alice's subscribers are unhappy and *infers* that
|
|
3689
|
+
* the fault is on Alice's side, because the affected clients share a publisher and nothing else. That
|
|
3690
|
+
* inference is sound, and it is still a inference: the same observation is produced by the SFU
|
|
3691
|
+
* mangling Alice's stream on the way out, with Alice herself perfectly healthy.
|
|
3692
|
+
*
|
|
3693
|
+
* This detector removes the inference. When Alice reports `encoder-bottleneck` *and* four of her six
|
|
3694
|
+
* subscribers report `freezed-video-track` in the same window, there is nothing left to deduce — the
|
|
3695
|
+
* source said it was struggling and the receivers confirmed the consequence. That is the strongest
|
|
3696
|
+
* statement this library can make about where a fault sits, and it is only available to something
|
|
3697
|
+
* holding both ends at once. Neither the publisher nor any receiver can reach this conclusion alone.
|
|
3698
|
+
*
|
|
3699
|
+
* Run both: fan-out is broader and catches the SFU-forwarding case where the publisher is fine;
|
|
3700
|
+
* this one is narrower and, when it fires, needs no interpretation.
|
|
3701
|
+
*
|
|
3702
|
+
* ### Silence here is not health
|
|
3703
|
+
*
|
|
3704
|
+
* A quiet detector means only that the two halves have not coincided — most commonly because the
|
|
3705
|
+
* publisher is genuinely fine and the fault is in forwarding, which is exactly the case `fan-out`
|
|
3706
|
+
* exists to report. Do not read "no corroborated fault" as "no publisher-side problem".
|
|
3707
|
+
*
|
|
3708
|
+
* ```ts
|
|
3709
|
+
* observedCall.addDetector('publisher-fault-corroboration-detector', {
|
|
3710
|
+
* publisherIssueTypes: [ 'encoder-bottleneck', 'capture-bottleneck', 'dry-outbound-track' ],
|
|
3711
|
+
* receiverIssueTypes: [ 'freezed-video-track', 'dry-inbound-track' ],
|
|
3712
|
+
* });
|
|
3713
|
+
* ```
|
|
3714
|
+
*
|
|
3715
|
+
* ### Requires a `RemoteTrackResolver`
|
|
3716
|
+
*
|
|
3717
|
+
* Matching a publisher's issue to its subscribers' issues needs the publisher↔subscriber links. With
|
|
3718
|
+
* no resolver the detector does nothing rather than guessing.
|
|
3719
|
+
*/
|
|
3720
|
+
declare class PublisherFaultCorroborationDetector implements Detector, ActiveIssueTracker {
|
|
3721
|
+
private readonly _call;
|
|
3722
|
+
static readonly NAME: "publisher-fault-corroboration-detector";
|
|
3723
|
+
readonly name: "publisher-fault-corroboration-detector";
|
|
3724
|
+
private readonly _config;
|
|
3725
|
+
private readonly _lastRaisedAt;
|
|
3726
|
+
/** Open publisher-side issues that name a track. */
|
|
3727
|
+
private readonly _publisherIssues;
|
|
3728
|
+
/** Open receiver-side issues that name a track. */
|
|
3729
|
+
private readonly _receiverIssues;
|
|
3730
|
+
/** The faults corroborated on the most recent `update()`. Exposed for tests/dashboards. */
|
|
3731
|
+
lastFaults: CorroboratedPublisherFault[];
|
|
3732
|
+
constructor(_call: ObservedCall, config?: Partial<PublisherFaultCorroborationDetectorConfig>);
|
|
3733
|
+
get size(): number;
|
|
3734
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3735
|
+
add(issue: ActiveClientIssue): void;
|
|
3736
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3737
|
+
clear(): void;
|
|
3738
|
+
update(): void;
|
|
3739
|
+
close(): void;
|
|
3740
|
+
/**
|
|
3741
|
+
* Resolve a publisher-side issue's `trackId` to the outbound track it is about.
|
|
3742
|
+
*
|
|
3743
|
+
* Looked up through the reporting client's own peer connections: the issue names its client, so
|
|
3744
|
+
* the search is bounded by that client's transports rather than by the size of the call.
|
|
3745
|
+
*/
|
|
3746
|
+
private _outboundTrackOf;
|
|
3747
|
+
}
|
|
3748
|
+
|
|
3749
|
+
declare const ObserverConcurrentIssueTypes: {
|
|
3750
|
+
/**
|
|
3751
|
+
* The same issue is open in **several unrelated calls at once**. Those clients share no meeting,
|
|
3752
|
+
* no publisher and no room — only the infrastructure serving them.
|
|
3753
|
+
*/
|
|
3754
|
+
readonly crossCallConcurrentIssues: "CROSS_CALL_CONCURRENT_ISSUES";
|
|
3755
|
+
/** The cross-call version that also started together. The strongest "it's us" signal available. */
|
|
3756
|
+
readonly crossCallIssueOnsetBurst: "CROSS_CALL_ISSUE_ONSET_BURST";
|
|
3757
|
+
};
|
|
3758
|
+
type ObserverConcurrentIssueDetectorConfig = {
|
|
3759
|
+
/**
|
|
3760
|
+
* The issue types to watch. **Required, and must not be empty** — the detector subscribes to
|
|
3761
|
+
* exactly these and sees nothing else.
|
|
3762
|
+
*/
|
|
3763
|
+
issueTypes: string[];
|
|
3764
|
+
/** Minimum distinct clients sharing the open issue. Default `3`. */
|
|
3765
|
+
minAffectedClients: number;
|
|
3766
|
+
/**
|
|
3767
|
+
* Minimum number of *distinct calls* the affected clients must span. Default `2`.
|
|
3768
|
+
*
|
|
3769
|
+
* This is what makes an observer-scoped finding mean something a call-scoped one doesn't. Without
|
|
3770
|
+
* it, one thirty-person meeting with congestion satisfies every client-count threshold and raises
|
|
3771
|
+
* a fleet-wide alert for what is really one bad room — which `CallConcurrentIssueDetector` has
|
|
3772
|
+
* already reported. Requiring two or more independent calls is the difference between a
|
|
3773
|
+
* coincidence and a shared cause.
|
|
3774
|
+
*/
|
|
3775
|
+
minAffectedCalls: number;
|
|
3776
|
+
/**
|
|
3777
|
+
* Fraction of the calls in flight that must be affected. Default `0` — off, because absolute
|
|
3778
|
+
* counts matter more than ratios here: three broken calls out of a thousand is still worth
|
|
3779
|
+
* knowing about, and a ratio threshold would hide it. Raise it if you only care about fleet-wide
|
|
3780
|
+
* events.
|
|
3781
|
+
*
|
|
3782
|
+
* Note there is deliberately **no participant ratio** at this scope. Six broken calls out of forty
|
|
3783
|
+
* is a handful of clients against the whole fleet, so any meaningful client ratio would suppress
|
|
3784
|
+
* exactly the finding this detector exists to produce.
|
|
3785
|
+
*/
|
|
3786
|
+
affectedCallRatioThreshold: number;
|
|
3787
|
+
/**
|
|
3788
|
+
* When the onsets fall within this span (ms), the finding is escalated to
|
|
3789
|
+
* `CROSS_CALL_ISSUE_ONSET_BURST`. Default `2_000`.
|
|
3790
|
+
*/
|
|
3791
|
+
onsetBurstWindowInMs: number;
|
|
3792
|
+
/** Re-arm time (ms) per issue type. Default `60_000`. */
|
|
3793
|
+
cooldownMs: number;
|
|
3794
|
+
};
|
|
3795
|
+
/** What the detector currently knows about one issue type across the fleet. */
|
|
3796
|
+
type ObserverConcurrentIssueGroup = {
|
|
3797
|
+
type: string;
|
|
3798
|
+
issues: ActiveClientIssue[];
|
|
3799
|
+
clientIds: string[];
|
|
3800
|
+
totalClients: number;
|
|
3801
|
+
affectedRatio: number;
|
|
3802
|
+
callIds: string[];
|
|
3803
|
+
totalCalls: number;
|
|
3804
|
+
affectedCallRatio: number;
|
|
3805
|
+
/** Per-call breakdown, largest first — the first question anyone asks is "which calls, how badly?". */
|
|
3806
|
+
perCall: {
|
|
3807
|
+
callId: string;
|
|
3808
|
+
affectedClients: number;
|
|
3809
|
+
totalClients: number;
|
|
3810
|
+
}[];
|
|
3811
|
+
/**
|
|
3812
|
+
* Spread of the onsets, in **observer** time (ms). Client clocks are never compared across
|
|
3813
|
+
* machines here — skew between them would masquerade as a synchronized event.
|
|
3814
|
+
*/
|
|
3815
|
+
onsetSpreadInMs: number;
|
|
3816
|
+
firstObservedAt: number;
|
|
3817
|
+
};
|
|
3818
|
+
/**
|
|
3819
|
+
* Answers **"is our infrastructure in trouble?"** — the same issue open across several *unrelated*
|
|
3820
|
+
* calls at the same moment.
|
|
3821
|
+
*
|
|
3822
|
+
* This is not the call-scoped question with a bigger denominator, which is why it is a separate
|
|
3823
|
+
* detector with separate gates and its own finding types. Participant count alone is a bad fleet
|
|
3824
|
+
* signal: one thirty-person meeting where everybody is congested clears every client threshold, yet
|
|
3825
|
+
* it has an obvious local explanation. Clients in *different* calls share no room, no publisher and
|
|
3826
|
+
* no host — only the servers and the network. When the same issue opens across several of them at
|
|
3827
|
+
* once, the infrastructure is the only remaining common factor, and that is the finding worth paging
|
|
3828
|
+
* someone about.
|
|
3829
|
+
*
|
|
3830
|
+
* ```ts
|
|
3831
|
+
* observer.addObserverDetector('observer-concurrent-issue-detector', {
|
|
3832
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
3833
|
+
* minAffectedCalls: 3,
|
|
3834
|
+
* });
|
|
3835
|
+
*
|
|
3836
|
+
* observer.on('observer-issue', ({ issue }) => {
|
|
3837
|
+
* if (issue.type === ObserverConcurrentIssueTypes.crossCallIssueOnsetBurst) page(issue);
|
|
3838
|
+
* });
|
|
3839
|
+
* // → CROSS_CALL_ISSUE_ONSET_BURST { issueType: 'congestion', calls: 40, affectedCalls: 6, … }
|
|
3840
|
+
* ```
|
|
3841
|
+
*
|
|
3842
|
+
* Onsets are compared on the **observer clock** (`observedAt`), never on client clocks: participants
|
|
3843
|
+
* degrading together within a couple of seconds is far more likely to be a deploy, a TURN failover or
|
|
3844
|
+
* a link flap than a coincidence — but only if the timestamps being compared came from one clock.
|
|
3845
|
+
*/
|
|
3846
|
+
declare class ObserverConcurrentIssueDetector implements Detector, ActiveIssueTracker {
|
|
3847
|
+
private readonly _observer;
|
|
3848
|
+
static readonly NAME: "observer-concurrent-issue-detector";
|
|
3849
|
+
readonly name: "observer-concurrent-issue-detector";
|
|
3850
|
+
private readonly _config;
|
|
3851
|
+
private readonly _lastRaisedAt;
|
|
3852
|
+
/** issue type -> the issues of that type currently open anywhere in the fleet. */
|
|
3853
|
+
private readonly _byType;
|
|
3854
|
+
private _size;
|
|
3855
|
+
/** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
|
|
3856
|
+
lastGroups: ObserverConcurrentIssueGroup[];
|
|
3857
|
+
constructor(_observer: Observer, config?: Partial<ObserverConcurrentIssueDetectorConfig>);
|
|
3858
|
+
get size(): number;
|
|
3859
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3860
|
+
add(issue: ActiveClientIssue): void;
|
|
3861
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3862
|
+
clear(): void;
|
|
3863
|
+
update(): void;
|
|
3864
|
+
close(): void;
|
|
3865
|
+
private _groupOf;
|
|
3866
|
+
}
|
|
3867
|
+
|
|
3868
|
+
declare const IssueFanOutTypes: {
|
|
3869
|
+
/** Most receivers of one published track have the same issue open → the fault follows the source. */
|
|
3870
|
+
readonly publishedTrackIssueFanOut: "PUBLISHED_TRACK_ISSUE_FAN_OUT";
|
|
3871
|
+
/** Exactly one receiver of a track has it → that receiver's own problem. */
|
|
3872
|
+
readonly singleReceiverIssue: "SINGLE_RECEIVER_ISSUE";
|
|
3873
|
+
};
|
|
3874
|
+
type IssueFanOutDetectorConfig = {
|
|
3875
|
+
/**
|
|
3876
|
+
* The receiver-side issue types to attribute to publishers. **Required, and must not be empty** —
|
|
3877
|
+
* the detector subscribes to exactly these.
|
|
3878
|
+
*
|
|
3879
|
+
* There is no "all types" option. Which of a receiver's complaints are worth blaming a publisher
|
|
3880
|
+
* for is application knowledge: `freezed-video-track` fanning out across a track's subscribers
|
|
3881
|
+
* implicates the source, `cpulimitation` fanning out the same way implicates the receivers'
|
|
3882
|
+
* hardware and would be a false accusation.
|
|
3883
|
+
*/
|
|
3884
|
+
issueTypes: string[];
|
|
3885
|
+
/** Minimum receivers of the track before a ratio is meaningful. Default `3`. */
|
|
3886
|
+
minReceivers: number;
|
|
3887
|
+
/** Fraction of a track's receivers that must share the issue. Default `0.6`. */
|
|
3888
|
+
affectedRatioThreshold: number;
|
|
3889
|
+
/** Also report the "only one receiver is affected" case. Default `true`. */
|
|
3890
|
+
reportSingleReceiver: boolean;
|
|
3891
|
+
/** Re-arm time (ms) per (track, issue type). Default `60_000`. */
|
|
3892
|
+
cooldownMs: number;
|
|
3893
|
+
};
|
|
3894
|
+
/**
|
|
3895
|
+
* Attributes **client-reported issues to the published track they are about**, then asks how far
|
|
3896
|
+
* the problem fans out across that track's receivers.
|
|
3897
|
+
*
|
|
3898
|
+
* The join is what makes this possible: a receiver-side issue payload carries `trackId` (the client
|
|
3899
|
+
* detectors report it for every track-scoped issue), the observer resolves that to an inbound track,
|
|
3900
|
+
* and `RemoteTrackResolver` links the inbound track to the `remoteOutboundTrack` that published it.
|
|
3901
|
+
* With the whole subscriber set of one source in hand, the verdict is straightforward and is the
|
|
3902
|
+
* single most useful thing a server can say:
|
|
3903
|
+
*
|
|
3904
|
+
* - **most receivers of Alice's track are affected** → the fault is on Alice's path — her uplink, the
|
|
3905
|
+
* SFU's ingress, or its forwarding of that stream.
|
|
3906
|
+
* - **one receiver of Alice's track is affected** → that receiver's downlink. Nothing to do with
|
|
3907
|
+
* Alice, even though the symptom is reported against her stream.
|
|
3908
|
+
*
|
|
3909
|
+
* Deliberately generic over the issue vocabulary: `freezed-video-track`, `keyframe-storm`,
|
|
3910
|
+
* `audio-concealment`, `video-decoder-overloaded`, `stuck-decoder` and anything a custom client
|
|
3911
|
+
* detector invents all fan out the same way, so one mechanism replaces a family of symptom-specific
|
|
3912
|
+
* detectors.
|
|
3913
|
+
*
|
|
3914
|
+
* ### It walks the affected tracks, never all of them
|
|
3915
|
+
*
|
|
3916
|
+
* The detector is fed open issues by the call's registry and keeps only those carrying a `trackId`.
|
|
3917
|
+
* Each tick it resolves *those* tracks to their publishers — never the published tracks of the call,
|
|
3918
|
+
* of which there are many more and almost all of them fine. A call with nothing wrong costs one
|
|
3919
|
+
* `size === 0` check.
|
|
3920
|
+
*
|
|
3921
|
+
* ### Requires a `RemoteTrackResolver`
|
|
3922
|
+
*
|
|
3923
|
+
* Without publisher↔subscriber links there is no way to know which receivers belong to one source,
|
|
3924
|
+
* so the detector does nothing when the call has no resolver. It does not fall back to guessing:
|
|
3925
|
+
* "one receiver of an unknown set" is not a statement worth raising.
|
|
3926
|
+
*/
|
|
3927
|
+
declare class IssueFanOutDetector implements Detector, ActiveIssueTracker {
|
|
3928
|
+
private readonly _call;
|
|
3929
|
+
static readonly NAME = "issue-fan-out-detector";
|
|
3930
|
+
readonly name = "issue-fan-out-detector";
|
|
3931
|
+
private readonly _config;
|
|
3932
|
+
private readonly _lastRaisedAt;
|
|
3933
|
+
/** Open issues that name a track. Issues without a `trackId` cannot be attributed and are dropped. */
|
|
3934
|
+
private readonly _trackIssues;
|
|
3935
|
+
constructor(_call: ObservedCall, config?: Partial<IssueFanOutDetectorConfig>);
|
|
3936
|
+
get size(): number;
|
|
3937
|
+
has(issue: ActiveClientIssue): boolean;
|
|
3938
|
+
add(issue: ActiveClientIssue): void;
|
|
3939
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
3940
|
+
clear(): void;
|
|
3941
|
+
update(): void;
|
|
3942
|
+
close(): void;
|
|
3943
|
+
/**
|
|
3944
|
+
* Resolve the issue's `trackId` to the outbound track that published it.
|
|
3945
|
+
*
|
|
3946
|
+
* Looked up through the reporting client's own peer connections rather than by scanning the call:
|
|
3947
|
+
* the issue names its client, so the search is bounded by that client's transports (typically one
|
|
3948
|
+
* or two) instead of by the size of the meeting.
|
|
3949
|
+
*/
|
|
3950
|
+
private _publisherOf;
|
|
3951
|
+
}
|
|
3952
|
+
|
|
3953
|
+
/**
|
|
3954
|
+
* One completed sampling bucket: how many/which clients reported congestion during it.
|
|
3955
|
+
*
|
|
3956
|
+
* `totalClients` (and therefore `congestedClientRatio`) is a snapshot of `observer.numberOfClients`
|
|
3957
|
+
* taken when the bucket closes — an approximation of "how many clients could have been congested",
|
|
3958
|
+
* not a claim that every client sent exactly one sample within the bucket. Good enough for a ratio
|
|
3959
|
+
* that only needs to be comparable bucket-to-bucket.
|
|
3960
|
+
*/
|
|
3961
|
+
type SfuCongestionDetectorBucket = {
|
|
3962
|
+
observedAt: number;
|
|
3963
|
+
totalClients: number;
|
|
3964
|
+
congestedClients: number;
|
|
3965
|
+
congestedClientRatio: number;
|
|
3966
|
+
affectedClientIds: string[];
|
|
3967
|
+
affectedCallIds: string[];
|
|
3968
|
+
};
|
|
3969
|
+
type SfuCongestionDetectorReport = {
|
|
3970
|
+
affectedCallIds: string[];
|
|
3971
|
+
affectedClientIds: string[];
|
|
3972
|
+
congestedClientRatio: number;
|
|
3973
|
+
totalNumberOfClients: number;
|
|
3974
|
+
numberOfCongestedClients: number;
|
|
3975
|
+
historySize: number;
|
|
3976
|
+
baselineCongestedClientRatio: number;
|
|
3977
|
+
robustZ: number;
|
|
3978
|
+
absoluteIncrease: number;
|
|
3979
|
+
relativeIncrease: number;
|
|
3980
|
+
};
|
|
3981
|
+
type SfuCongestionDetectorConfig = {
|
|
3982
|
+
consumedClientIssueTypes: string[];
|
|
3983
|
+
emittedObserverIssueType: string;
|
|
3984
|
+
samplesSendingTimeInMs: number;
|
|
3985
|
+
historySize: number;
|
|
3986
|
+
minHistorySize: number;
|
|
3987
|
+
minAffectedClients: number;
|
|
3988
|
+
minAbsoluteRatioIncrease: number;
|
|
3989
|
+
minRelativeRatioIncrease: number;
|
|
3990
|
+
robustZThreshold: number;
|
|
3991
|
+
};
|
|
3992
|
+
/** The statistical/practical-significance verdict for one candidate bucket against its baseline. */
|
|
3993
|
+
type SfuCongestionDetectorEvaluation = {
|
|
3994
|
+
isCongested: boolean;
|
|
3995
|
+
baselineCongestedClientRatio: number;
|
|
3996
|
+
robustZ: number;
|
|
3997
|
+
absoluteIncrease: number;
|
|
3998
|
+
relativeIncrease: number;
|
|
3999
|
+
};
|
|
4000
|
+
/**
|
|
4001
|
+
* Detects a **shared** congestion event: many clients, across different calls, reporting congestion
|
|
4002
|
+
* inside the same slice of time.
|
|
4003
|
+
*
|
|
4004
|
+
* Only add this when the observer's calls all come from the **same SFU** — the finding's whole
|
|
4005
|
+
* meaning is "these clients have nothing in common except that server", and that is only true if the
|
|
4006
|
+
* server really is the common factor.
|
|
4007
|
+
*
|
|
4008
|
+
* ### Why fixed-interval buckets, and not the update tick
|
|
4009
|
+
*
|
|
4010
|
+
* The obvious implementation counts congested clients on each `update()`. It is wrong here, for two
|
|
4011
|
+
* separate reasons:
|
|
4012
|
+
*
|
|
4013
|
+
* - **The tick is not evenly spaced.** `update()` fires when a client is updated, so its rate is a
|
|
4014
|
+
* function of how many clients are connected and how their sampling happens to interleave. Two
|
|
4015
|
+
* counts taken from windows of different length are not comparable, and this detector's entire
|
|
4016
|
+
* job is to compare a count against earlier counts.
|
|
4017
|
+
* - **Clients report on their own schedule.** A client sends a sample roughly every
|
|
4018
|
+
* `samplesSendingTimeInMs`, unsynchronised with every other client. A window shorter than that
|
|
4019
|
+
* systematically undercounts — half the congested clients simply hadn't spoken yet — and the
|
|
4020
|
+
* undercount varies with arrival phase, which is noise indistinguishable from signal.
|
|
4021
|
+
*
|
|
4022
|
+
* So the detector runs on a wall-clock interval and closes a bucket every
|
|
4023
|
+
* `samplesSendingTimeInMs`, giving every client a fair chance to be heard in each one. Buckets are
|
|
4024
|
+
* equal-length and equally lagged, which is what makes bucket-to-bucket comparison mean something.
|
|
4025
|
+
* {@link update} is deliberately empty: nothing here is driven by the update tick.
|
|
4026
|
+
*
|
|
4027
|
+
* ### Occurrences, not intervals
|
|
4028
|
+
*
|
|
4029
|
+
* Unlike `ConcurrentIssueDetector`, this one ignores resolutions — see {@link delete}. It counts how
|
|
4030
|
+
* many *distinct clients reported* congestion in a bucket, not how many are still congested. A
|
|
4031
|
+
* client that hits congestion and immediately drops its bitrate resolves the issue within seconds
|
|
4032
|
+
* and would vanish from an open-interval view, yet it is exactly the evidence wanted here.
|
|
4033
|
+
*/
|
|
4034
|
+
declare class SfuCongestionDetector implements Detector, ActiveIssueTracker {
|
|
4035
|
+
private readonly _observer;
|
|
4036
|
+
static readonly NAME = "sfu-congestion-detector";
|
|
4037
|
+
readonly name = "sfu-congestion-detector";
|
|
4038
|
+
private readonly _config;
|
|
4039
|
+
private readonly _history;
|
|
4040
|
+
private readonly _trackedIssues;
|
|
4041
|
+
private timer;
|
|
4042
|
+
private _lastEvaluatedBucket;
|
|
4043
|
+
constructor(_observer: Observer, config?: Partial<SfuCongestionDetectorConfig>);
|
|
4044
|
+
get size(): number;
|
|
4045
|
+
/** Record the issue against the bucket currently open. The timer, not this, closes the bucket. */
|
|
4046
|
+
add(issue: ActiveClientIssue): void;
|
|
4047
|
+
/**
|
|
4048
|
+
* Deliberately a no-op returning `false`.
|
|
4049
|
+
*
|
|
4050
|
+
* Resolutions are not interesting here. A congested client typically fixes its own symptom by
|
|
4051
|
+
* dropping bitrate hard, so the issue closes within seconds — but it still *happened*, and it is
|
|
4052
|
+
* evidence that the server was under pressure during this bucket. What matters is how many
|
|
4053
|
+
* distinct clients reported congestion within the bucket and whether that count suddenly jumps,
|
|
4054
|
+
* not how long any one client's issue stayed open.
|
|
4055
|
+
*
|
|
4056
|
+
* Nothing leaks: the tracked set is emptied wholesale every time a bucket closes.
|
|
4057
|
+
*/
|
|
4058
|
+
delete(_issue: ActiveClientIssue): boolean;
|
|
4059
|
+
clear(): void;
|
|
4060
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4061
|
+
close(): void;
|
|
4062
|
+
/** The completed buckets kept so far, oldest first. Read-only — for introspection/tests. */
|
|
4063
|
+
get history(): readonly SfuCongestionDetectorBucket[];
|
|
4064
|
+
/**
|
|
4065
|
+
* Intentionally empty — see the class description.
|
|
4066
|
+
*
|
|
4067
|
+
* Everything here is driven by the bucket timer, because the update tick is neither evenly spaced
|
|
4068
|
+
* nor long enough for every client to have reported. Counting on it would compare windows of
|
|
4069
|
+
* different lengths and call the difference a signal.
|
|
4070
|
+
*/
|
|
4071
|
+
update(): void;
|
|
4072
|
+
private _closeBucket;
|
|
4073
|
+
/**
|
|
4074
|
+
* Evaluate only the latest completed bucket (the candidate) against the buckets before it (the
|
|
4075
|
+
* baseline) — never against itself. Reached once per newly-closed bucket via {@link update}; the
|
|
4076
|
+
* identity check below additionally guards against evaluating the same bucket twice, in case
|
|
4077
|
+
* `update()` is ever called again before the next rotation.
|
|
4078
|
+
*/
|
|
4079
|
+
private _evaluateLatestBucket;
|
|
4080
|
+
/**
|
|
4081
|
+
* Is `candidate` — the latest completed bucket — abnormally high compared with the `baseline`
|
|
4082
|
+
* buckets before it?
|
|
4083
|
+
*
|
|
4084
|
+
* Requires both **statistical** significance (a robust z-score against a median+MAD baseline —
|
|
4085
|
+
* deliberately not Mann-Kendall, which asks "is this a monotonic trend", not "is the latest point
|
|
4086
|
+
* an outlier"; a single sudden spike on an otherwise flat series is exactly what should trigger
|
|
4087
|
+
* here and exactly what a trend test would miss) and **practical** significance (enough affected
|
|
4088
|
+
* clients, and a big enough absolute/relative jump — a statistically significant move in a tiny
|
|
4089
|
+
* or trivial ratio is not worth an alert).
|
|
4090
|
+
*/
|
|
4091
|
+
private _evaluateBucket;
|
|
4092
|
+
}
|
|
4093
|
+
|
|
4094
|
+
declare const TrackDeliveryMismatchTypes: {
|
|
4095
|
+
/**
|
|
4096
|
+
* The source is sending, but **none** of its subscribers are receiving → the media is being lost
|
|
4097
|
+
* between the publisher and the receivers. In an SFU that means the forwarding path.
|
|
4098
|
+
*/
|
|
4099
|
+
readonly publishedTrackNotDelivered: "PUBLISHED_TRACK_NOT_DELIVERED";
|
|
4100
|
+
/**
|
|
4101
|
+
* The source is sending and most subscribers are fine, but **some** are dry → those consumers are
|
|
4102
|
+
* broken individually (in mediasoup, the usual mitigation is recreating the consumer).
|
|
4103
|
+
*/
|
|
4104
|
+
readonly receiverTrackNotDelivered: "RECEIVER_TRACK_NOT_DELIVERED";
|
|
4105
|
+
/**
|
|
4106
|
+
* The source itself stopped producing, so its subscribers being dry is expected and **not** an
|
|
4107
|
+
* SFU fault. Reported so the other two verdicts can be trusted as *not* being this.
|
|
4108
|
+
*/
|
|
4109
|
+
readonly publisherTrackDry: "PUBLISHER_TRACK_DRY";
|
|
4110
|
+
};
|
|
4111
|
+
type TrackDeliveryMismatchDetectorConfig = {
|
|
4112
|
+
/** The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`. */
|
|
4113
|
+
dryInboundIssueType: string;
|
|
4114
|
+
/** The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`. */
|
|
4115
|
+
dryOutboundIssueType: string;
|
|
4116
|
+
/** Minimum subscribers before "all of them" means anything. Default `2`. */
|
|
4117
|
+
minReceivers: number;
|
|
4118
|
+
/** Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`. */
|
|
4119
|
+
allReceiversRatio: number;
|
|
4120
|
+
/** Re-arm time (ms) per (track, verdict). Default `60_000`. */
|
|
4121
|
+
cooldownMs: number;
|
|
4122
|
+
};
|
|
4123
|
+
/**
|
|
4124
|
+
* Answers **"is the media actually getting through?"** by joining the two ends of a published track.
|
|
4125
|
+
*
|
|
4126
|
+
* A dry track is the clearest possible symptom — no bytes are arriving — but on its own it is
|
|
4127
|
+
* ambiguous, and the ambiguity is precisely what a single endpoint cannot resolve. A receiver seeing
|
|
4128
|
+
* silence cannot tell whether the camera was switched off, the SFU stopped forwarding, or its own
|
|
4129
|
+
* consumer wedged. All three look identical from the browser.
|
|
4130
|
+
*
|
|
4131
|
+
* With the publisher↔subscriber links this becomes a three-way decision:
|
|
4132
|
+
*
|
|
4133
|
+
* | publisher | subscribers | verdict |
|
|
4134
|
+
* |---|---|---|
|
|
4135
|
+
* | sending | **all** dry | `PUBLISHED_TRACK_NOT_DELIVERED` — the SFU/forwarding path |
|
|
4136
|
+
* | sending | **some** dry | `RECEIVER_TRACK_NOT_DELIVERED` — those consumers (recreate them) |
|
|
4137
|
+
* | dry | any dry | `PUBLISHER_TRACK_DRY` — the source stopped; not an SFU fault |
|
|
4138
|
+
*
|
|
4139
|
+
* The publisher side is judged from **both** signals available: its own `dry-outbound-track` issue
|
|
4140
|
+
* when the client reports one, and — as the fallback, and the corroboration when it does not — the
|
|
4141
|
+
* observed outbound RTP (`deltaPacketsSent`). That combination is what makes the first row
|
|
4142
|
+
* trustworthy: the server can state that packets demonstrably left the publisher during the same
|
|
4143
|
+
* interval in which every receiver got nothing.
|
|
4144
|
+
*
|
|
4145
|
+
* This is the "SFU forwarding mismatch" check, and notably it needs **no** mediasoup instrumentation
|
|
4146
|
+
* — the client's own dry-track verdicts plus the resolver links are sufficient.
|
|
4147
|
+
*/
|
|
4148
|
+
declare class TrackDeliveryMismatchDetector implements Detector, ActiveIssueTracker {
|
|
4149
|
+
private readonly call;
|
|
4150
|
+
static readonly NAME: "track-delivery-mismatch-detector";
|
|
4151
|
+
readonly name: "track-delivery-mismatch-detector";
|
|
4152
|
+
private readonly _config;
|
|
4153
|
+
private readonly _lastRaisedAt;
|
|
4154
|
+
private readonly dryOutboundTracks;
|
|
4155
|
+
private readonly dryInboundTracks;
|
|
4156
|
+
constructor(call: ObservedCall, config?: Partial<TrackDeliveryMismatchDetectorConfig>);
|
|
4157
|
+
close(): void;
|
|
4158
|
+
add(issue: ActiveClientIssue): void;
|
|
4159
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4160
|
+
get size(): number;
|
|
4161
|
+
clear(): void;
|
|
4162
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4163
|
+
update(): void;
|
|
4164
|
+
}
|
|
4165
|
+
|
|
4166
|
+
declare const TurnServerHealthTypes: {
|
|
4167
|
+
/** One TURN server's clients are in trouble while other servers' clients are fine. */
|
|
4168
|
+
readonly turnServerDegraded: "TURN_SERVER_DEGRADED";
|
|
4169
|
+
};
|
|
4170
|
+
type TurnServerHealthDetectorConfig = {
|
|
4171
|
+
/** Minimum clients on a server before a ratio is meaningful. Default `5`. */
|
|
4172
|
+
minClientsPerServer: number;
|
|
4173
|
+
/** Fraction of a server's clients that must have an open issue. Default `0.5`. */
|
|
4174
|
+
degradedRatioThreshold: number;
|
|
4175
|
+
/**
|
|
4176
|
+
* Which client issue types count as "in trouble". Empty (default) means **any** open issue —
|
|
4177
|
+
* appropriate here, because the question is not *what* is wrong with each client but whether
|
|
4178
|
+
* trouble clusters on one relay.
|
|
4179
|
+
*/
|
|
4180
|
+
issueTypes: string[];
|
|
4181
|
+
/** Consecutive ticks the condition must hold before raising. Default `2`. */
|
|
4182
|
+
consecutiveTicks: number;
|
|
4183
|
+
/** Re-arm time (ms) before raising again for the same server. Default `60_000`. */
|
|
4184
|
+
cooldownMs: number;
|
|
4185
|
+
};
|
|
4186
|
+
/** The per-server view this detector builds. */
|
|
4187
|
+
type TurnServerHealth = {
|
|
4188
|
+
serverUrl: string;
|
|
4189
|
+
/** Distinct clients whose media is relayed through this server. */
|
|
4190
|
+
clients: number;
|
|
4191
|
+
/** Of those, how many currently have at least one open issue. */
|
|
4192
|
+
degradedClients: number;
|
|
4193
|
+
degradedRatio: number;
|
|
4194
|
+
affectedClientIds: string[];
|
|
4195
|
+
/** The open issue types seen on this server's clients, most common first. */
|
|
4196
|
+
issueTypes: string[];
|
|
4197
|
+
};
|
|
4198
|
+
/**
|
|
4199
|
+
* An **observer-level** detector that groups relayed clients by the TURN server carrying them and
|
|
4200
|
+
* compares the servers against each other.
|
|
4201
|
+
*
|
|
4202
|
+
* Counting TURN usage is not useful on its own; knowing that `turn-eu-1` has 22 of 30 clients in
|
|
4203
|
+
* trouble while `turn-eu-2` has 1 of 34 is. Because the comparison spans calls it lives on
|
|
4204
|
+
* `observer.detectors` and raises `observer-issue` — one actionable alert instead of fifty
|
|
4205
|
+
* per-client ones. Each finding carries the other servers' ratios as context, since "half the
|
|
4206
|
+
* clients here are unhappy" only means something relative to the rest of the fleet.
|
|
4207
|
+
*
|
|
4208
|
+
* Whether a client is in trouble comes from **its own reported issues**, not from thresholds applied
|
|
4209
|
+
* here. The client already decides that far better than a server-side rule could; the value this
|
|
4210
|
+
* adds is the grouping — the dimension no endpoint can see.
|
|
4211
|
+
*
|
|
4212
|
+
* For a relay that has stopped serving entirely, see `TurnServerOutageDetector`: this detector needs
|
|
4213
|
+
* clients *on* the server to ask how many are unhappy, and an outage takes them away.
|
|
4214
|
+
*/
|
|
4215
|
+
declare class TurnServerHealthDetector implements Detector {
|
|
4216
|
+
private readonly _observer;
|
|
4217
|
+
static readonly NAME = "turn-server-health-detector";
|
|
4218
|
+
readonly name = "turn-server-health-detector";
|
|
4219
|
+
private readonly _config;
|
|
4220
|
+
private readonly _streaks;
|
|
4221
|
+
private readonly _lastRaisedAt;
|
|
4222
|
+
/** The per-server rollup computed on the most recent `update()`. */
|
|
4223
|
+
lastServers: TurnServerHealth[];
|
|
4224
|
+
constructor(_observer: Observer, config?: Partial<TurnServerHealthDetectorConfig>);
|
|
4225
|
+
update(): void;
|
|
4226
|
+
close(): void;
|
|
4227
|
+
private _serverHealth;
|
|
4228
|
+
}
|
|
4229
|
+
|
|
4230
|
+
declare const TurnServerOutageTypes: {
|
|
4231
|
+
/** One TURN server's relayed population collapsed while the rest of the fleet is fine. */
|
|
4232
|
+
readonly turnServerOutage: "TURN_SERVER_OUTAGE";
|
|
4233
|
+
};
|
|
4234
|
+
type TurnServerOutageDetectorConfig = {
|
|
4235
|
+
/**
|
|
4236
|
+
* How many clients a server must have been carrying at its peak before its collapse means
|
|
4237
|
+
* anything. Below this, one or two people leaving looks like an outage. Default `5`.
|
|
4238
|
+
*/
|
|
4239
|
+
minClientsAtPeak: number;
|
|
4240
|
+
/**
|
|
4241
|
+
* Fraction of the peak population that must be gone or disrupted. Default `0.8` — an outage is
|
|
4242
|
+
* near-total by definition; partial degradation is `TurnServerHealthDetector`'s question.
|
|
4243
|
+
*/
|
|
4244
|
+
lossRatioThreshold: number;
|
|
4245
|
+
/**
|
|
4246
|
+
* Window (ms) the peak population is measured over. Long enough to span a real outage's onset,
|
|
4247
|
+
* short enough that yesterday's peak isn't held against today. Default `120_000`.
|
|
4248
|
+
*/
|
|
4249
|
+
peakWindowMs: number;
|
|
4250
|
+
/**
|
|
4251
|
+
* Require a healthy **control group** — clients not relayed through this server that are still
|
|
4252
|
+
* connected — before blaming the server. Without this, a call ending, a fleet-wide network
|
|
4253
|
+
* event, or the observer shutting down all look exactly like a TURN outage. Default `true`.
|
|
4254
|
+
*/
|
|
4255
|
+
requireControlGroup: boolean;
|
|
4256
|
+
/** Minimum clients elsewhere before the control group is statistically worth anything. Default `5`. */
|
|
4257
|
+
minControlGroupClients: number;
|
|
4258
|
+
/** Fraction of the control group that must still be healthy. Default `0.7`. */
|
|
4259
|
+
controlGroupHealthyRatio: number;
|
|
4260
|
+
/** Consecutive ticks the condition must hold before raising. Default `2`. */
|
|
4261
|
+
consecutiveTicks: number;
|
|
4262
|
+
/**
|
|
4263
|
+
* Re-arm time (ms) per server. Long by default (`300_000`) — an outage is one event, not one
|
|
4264
|
+
* per tick, and a server that stays down would otherwise alert forever.
|
|
4265
|
+
*/
|
|
4266
|
+
cooldownMs: number;
|
|
4267
|
+
};
|
|
4268
|
+
/**
|
|
4269
|
+
* Detects a **TURN server outage** — a relay that has stopped serving — by watching its client
|
|
4270
|
+
* population collapse while the rest of the fleet carries on.
|
|
4271
|
+
*
|
|
4272
|
+
* This is the case its sibling `TurnServerHealthDetector` structurally *cannot* see, and the
|
|
4273
|
+
* distinction is worth being precise about. That detector groups clients by the server relaying them
|
|
4274
|
+
* and asks how many are reporting issues. It needs clients on the server to ask the question. When a
|
|
4275
|
+
* TURN server goes down completely, allocation fails: existing sessions drop, and new clients never
|
|
4276
|
+
* obtain a relay candidate through it at all, so they are never attributed to it. The server's
|
|
4277
|
+
* population goes to zero and the health detector falls silent for the worst possible reason — it
|
|
4278
|
+
* has nobody left to ask. Degradation makes clients unhappy; an outage makes them *disappear*.
|
|
4279
|
+
*
|
|
4280
|
+
* So the signal here is absence, measured against the server's own recent peak:
|
|
4281
|
+
*
|
|
4282
|
+
* - clients gone entirely (their relayed peer connections closed, or they re-negotiated onto a
|
|
4283
|
+
* different path), plus
|
|
4284
|
+
* - clients still attributed to the server whose ICE or connection state is `disconnected` /
|
|
4285
|
+
* `failed` / `closed` — the ones mid-collapse, which is what you catch if you look during the
|
|
4286
|
+
* outage rather than after it.
|
|
4287
|
+
*
|
|
4288
|
+
* ### The control group is the whole design
|
|
4289
|
+
*
|
|
4290
|
+
* Absence is a dangerous signal: a call ending, everyone going home at 6pm, a fleet-wide network
|
|
4291
|
+
* event, and the observer itself shutting down all produce exactly the same collapse. The detector
|
|
4292
|
+
* therefore refuses to blame a server unless clients **not** relayed through it are demonstrably
|
|
4293
|
+
* still connected — `requireControlGroup`, on by default. "Everyone on `turn-eu-1` vanished" is
|
|
4294
|
+
* ambiguous; "everyone on `turn-eu-1` vanished while 200 clients elsewhere are fine" is an outage.
|
|
4295
|
+
*
|
|
4296
|
+
* That comparison is only available to something watching every call at once, which is why this is
|
|
4297
|
+
* an observer-level detector raising `observer-issue` — one alert for the fleet, not one per
|
|
4298
|
+
* abandoned call.
|
|
4299
|
+
*
|
|
4300
|
+
* ### Caveats worth knowing before you tune it
|
|
4301
|
+
*
|
|
4302
|
+
* Clients that fail over cleanly to a second TURN server still count as lost here, which is
|
|
4303
|
+
* correct — the server did stop serving them — but it means a well-configured fleet with automatic
|
|
4304
|
+
* failover reports outages that users never felt. That is the intended behaviour: the failover
|
|
4305
|
+
* worked *and* the server is down are both true, and you want to know the second one.
|
|
4306
|
+
*
|
|
4307
|
+
* A genuinely quiet server (last call of the day ends) is suppressed by the control group, not by
|
|
4308
|
+
* the collapse test. If you run a small deployment where the control group is routinely below
|
|
4309
|
+
* `minControlGroupClients`, this detector will stay quiet — prefer alerting on your TURN server's
|
|
4310
|
+
* own health checks there, since a handful of clients cannot distinguish these cases.
|
|
4311
|
+
*/
|
|
4312
|
+
declare class TurnServerOutageDetector implements Detector {
|
|
4313
|
+
private readonly _observer;
|
|
4314
|
+
static readonly NAME = "turn-server-outage-detector";
|
|
4315
|
+
readonly name = "turn-server-outage-detector";
|
|
4316
|
+
private readonly _config;
|
|
4317
|
+
/** serverUrl -> recent population observations, used to derive the windowed peak. */
|
|
4318
|
+
private readonly _peaks;
|
|
4319
|
+
private readonly _streaks;
|
|
4320
|
+
private readonly _lastRaisedAt;
|
|
4321
|
+
constructor(_observer: Observer, config?: Partial<TurnServerOutageDetectorConfig>);
|
|
4322
|
+
update(): void;
|
|
4323
|
+
close(): void;
|
|
4324
|
+
/** Distinct clients on a server, split by whether their relayed transport is actually up. */
|
|
4325
|
+
private _populationOf;
|
|
4326
|
+
/** Record this tick's population and return the peak across `peakWindowMs`. */
|
|
4327
|
+
private _recordAndPeak;
|
|
4328
|
+
/**
|
|
4329
|
+
* Everyone *not* relayed through `serverUrl`: clients on other TURN servers plus every client
|
|
4330
|
+
* the observer knows about that isn't relayed at all. The healthy share of that group is what
|
|
4331
|
+
* separates "this server broke" from "everything broke".
|
|
4332
|
+
*/
|
|
4333
|
+
private _controlGroup;
|
|
4334
|
+
}
|
|
4335
|
+
|
|
4336
|
+
declare const UnconsumedTrackTypes: {
|
|
4337
|
+
/** A track is being published to the SFU that nobody is subscribed to — pure wasted uplink. */
|
|
4338
|
+
readonly unconsumedPublishedTrack: "UNCONSUMED_PUBLISHED_TRACK";
|
|
4339
|
+
};
|
|
4340
|
+
type UnconsumedTrackDetectorConfig = {
|
|
4341
|
+
/** How long a track must stay unconsumed while sending before reporting (ms). Default `30_000`. */
|
|
4342
|
+
minUnconsumedDurationInMs: number;
|
|
4343
|
+
/** Ignore tracks below this bitrate — a trickle isn't worth an alert (bps). Default `50_000`. */
|
|
4344
|
+
minBitrate: number;
|
|
4345
|
+
/** Re-arm time (ms) per track. Default `300_000`. */
|
|
4346
|
+
cooldownMs: number;
|
|
4347
|
+
};
|
|
4348
|
+
/**
|
|
4349
|
+
* Finds tracks that are **published but consumed by nobody** — uplink and SFU ingress spent on media
|
|
4350
|
+
* that is never forwarded anywhere.
|
|
4351
|
+
*
|
|
4352
|
+
* This is the one detector that reads the resolver's *silence* as the signal: an outbound track with
|
|
4353
|
+
* an empty `remoteInboundTracks` set, still pushing packets. It reads `call.unconsumedOutboundTracks`,
|
|
4354
|
+
* which the resolver maintains as tracks gain and lose subscribers, so a healthy call costs one
|
|
4355
|
+
* `size === 0` check rather than a walk over every published track. The usual causes are a participant
|
|
4356
|
+
* publishing while everyone has them hidden or muted-in-UI, a simulcast layer no viewer's bandwidth
|
|
4357
|
+
* ever selects, or an application that forgot to stop a track after the last subscriber left.
|
|
4358
|
+
*
|
|
4359
|
+
* It is deliberately slow to fire: `minUnconsumedDurationInMs` must elapse with the track still
|
|
4360
|
+
* sending, because a brief gap between publishing and the first subscription is completely normal at
|
|
4361
|
+
* join time.
|
|
4362
|
+
*
|
|
4363
|
+
* ### Careful: this detector is only sound with a resolver
|
|
4364
|
+
*
|
|
4365
|
+
* "No subscribers" and "no resolver configured" produce the identical observation — an empty link
|
|
4366
|
+
* set. Without a `RemoteTrackResolver` this would report *every* published track in the call as
|
|
4367
|
+
* unconsumed, so it checks `call.remoteTrackResolver` at runtime and does nothing without one.
|
|
4368
|
+
*/
|
|
4369
|
+
declare class UnconsumedTrackDetector implements Detector {
|
|
4370
|
+
private readonly call;
|
|
4371
|
+
static readonly NAME = "unconsumed-track-detector";
|
|
4372
|
+
readonly name = "unconsumed-track-detector";
|
|
4373
|
+
readonly config: UnconsumedTrackDetectorConfig;
|
|
4374
|
+
/** trackId -> when it was first seen sending with no subscribers. */
|
|
4375
|
+
private readonly _unconsumedSince;
|
|
4376
|
+
private readonly _lastRaisedAt;
|
|
4377
|
+
constructor(call: ObservedCall, config?: Partial<UnconsumedTrackDetectorConfig>);
|
|
4378
|
+
update(): void;
|
|
4379
|
+
close(): void;
|
|
4380
|
+
}
|
|
4381
|
+
|
|
4382
|
+
/**
|
|
4383
|
+
* Detectors that reason **across calls**, created once on the observer.
|
|
4384
|
+
*
|
|
4385
|
+
* Adding one means: give the class a `static readonly NAME`, add its entry here, and add a `case` to
|
|
4386
|
+
* `Observer.addObserverDetector`. This map is what types the call site — the config is checked
|
|
4387
|
+
* against the right detector and an unknown name won't compile.
|
|
4388
|
+
*/
|
|
4389
|
+
type AvailableObserverScopeDetectorsConfigs = {
|
|
4390
|
+
[SfuCongestionDetector.NAME]: SfuCongestionDetectorConfig;
|
|
4391
|
+
[ObserverConcurrentIssueDetector.NAME]: ObserverConcurrentIssueDetectorConfig;
|
|
4392
|
+
[ClientPopulationIssueDetector.NAME]: ClientPopulationIssueDetectorConfig;
|
|
4393
|
+
[TurnServerHealthDetector.NAME]: TurnServerHealthDetectorConfig;
|
|
4394
|
+
[TurnServerOutageDetector.NAME]: TurnServerOutageDetectorConfig;
|
|
4395
|
+
};
|
|
4396
|
+
/**
|
|
4397
|
+
* Detectors that reason **within one call**, created for every call the observer opens.
|
|
4398
|
+
*
|
|
4399
|
+
* Note there is no detector in both maps. "Is this meeting in trouble?" and "is our infrastructure in
|
|
4400
|
+
* trouble?" are different questions with different gates and different findings, so they are separate
|
|
4401
|
+
* classes — `CallConcurrentIssueDetector` and `ObserverConcurrentIssueDetector` — rather than one
|
|
4402
|
+
* class branching on what it was handed.
|
|
4403
|
+
*/
|
|
4404
|
+
type AvailableCallScopeDetectorsConfigs = {
|
|
4405
|
+
[UnconsumedTrackDetector.NAME]: UnconsumedTrackDetectorConfig;
|
|
4406
|
+
[TrackDeliveryMismatchDetector.NAME]: TrackDeliveryMismatchDetectorConfig;
|
|
4407
|
+
[CallConcurrentIssueDetector.NAME]: CallConcurrentIssueDetectorConfig;
|
|
4408
|
+
[IssueFanOutDetector.NAME]: IssueFanOutDetectorConfig;
|
|
4409
|
+
[PublisherFaultCorroborationDetector.NAME]: PublisherFaultCorroborationDetectorConfig;
|
|
4410
|
+
};
|
|
4411
|
+
type AvailableDetectorsConfigs = AvailableObserverScopeDetectorsConfigs | AvailableCallScopeDetectorsConfigs;
|
|
4412
|
+
declare class Detectors {
|
|
4413
|
+
private _detectors;
|
|
4414
|
+
constructor(...detectors: Detector[]);
|
|
4415
|
+
get listOfNames(): string[];
|
|
4416
|
+
get size(): number;
|
|
4417
|
+
add(detector: Detector): void;
|
|
4418
|
+
get(name: string): Detector | undefined;
|
|
4419
|
+
remove(detector: Detector): void;
|
|
4420
|
+
update(): void;
|
|
4421
|
+
clear(): void;
|
|
4422
|
+
private _close;
|
|
4423
|
+
}
|
|
4424
|
+
|
|
4425
|
+
/**
|
|
4426
|
+
* The set of client issues currently believed to be **open**, plus the fan-out that pushes them to
|
|
4427
|
+
* whoever asked for them.
|
|
4428
|
+
*
|
|
4429
|
+
* ### Push, not poll
|
|
4430
|
+
*
|
|
4431
|
+
* A detector does not scan for the issues it cares about; it registers as an
|
|
4432
|
+
* {@link ActiveIssueTracker} for the types it consumes and is handed them as they open and close.
|
|
4433
|
+
* The cost of a detector is then proportional to the issues it actually receives, not to the number
|
|
4434
|
+
* of participants — a healthy 500-client fleet does no work per tick.
|
|
4435
|
+
*
|
|
4436
|
+
* ```ts
|
|
4437
|
+
* observer.activeIssuesRegistry.addIssueTracker('congestion', detector);
|
|
4438
|
+
* ```
|
|
4439
|
+
*
|
|
4440
|
+
* There is **no wildcard**. A tracker names the types it consumes, and nothing else reaches it. "Feed
|
|
4441
|
+
* me everything and I'll work out what matters" pushes the decision from the application — which
|
|
4442
|
+
* knows its client build and its issue vocabulary — onto a detector that has to guess, and it makes
|
|
4443
|
+
* the cost of a subscription unbounded and invisible. If a detector should watch five issue types,
|
|
4444
|
+
* the caller lists five issue types.
|
|
4445
|
+
*
|
|
4446
|
+
* ### Two levels
|
|
4447
|
+
*
|
|
4448
|
+
* Every call owns a registry constructed with the observer's as its `parent`. An add or delete
|
|
4449
|
+
* touches both, so a call-scoped tracker sees only that call's issues while an observer-scoped one
|
|
4450
|
+
* sees the fleet — without either side iterating the other. The child keeps **its own** storage:
|
|
4451
|
+
* `size` is this scope's count, and {@link clear} (called when the call closes) removes only this
|
|
4452
|
+
* scope's issues from the parent and never touches the parent's tracker registrations.
|
|
4453
|
+
*
|
|
4454
|
+
* ### Only keyed issues arrive here
|
|
4455
|
+
*
|
|
4456
|
+
* An issue without a `key` has no lifecycle — nothing can ever close it — so treating it as "active"
|
|
4457
|
+
* would mean holding a symptom that may have ended long ago. Keyless issues stay one-shot: emitted
|
|
4458
|
+
* as `client-issue`, never registered. `client-monitor-js` >= 4.6.0 sends `key` on everything
|
|
4459
|
+
* stateful.
|
|
4460
|
+
*/
|
|
4461
|
+
declare class ActiveIssuesRegistry implements ActiveIssueTracker {
|
|
4462
|
+
private readonly parent?;
|
|
4463
|
+
private readonly issues;
|
|
4464
|
+
private readonly typesToTrackers;
|
|
4465
|
+
constructor(parent?: ActiveIssueTracker | undefined);
|
|
4466
|
+
get size(): number;
|
|
4467
|
+
/**
|
|
4468
|
+
* The open issues in this scope, in insertion order.
|
|
4469
|
+
*
|
|
4470
|
+
* Insertion order is age order (`observedAt` is assigned on insert), which is what lets a consumer
|
|
4471
|
+
* stop at the first entry newer than its cutoff instead of scanning the whole set.
|
|
4472
|
+
*/
|
|
4473
|
+
values(): IterableIterator<ActiveClientIssue>;
|
|
4474
|
+
[Symbol.iterator](): IterableIterator<ActiveClientIssue>;
|
|
4475
|
+
has(issue: ActiveClientIssue): boolean;
|
|
4476
|
+
add(issue: ActiveClientIssue): this;
|
|
4477
|
+
delete(issue: ActiveClientIssue): boolean;
|
|
4478
|
+
/** Feed `tracker` every issue of `type` as it opens and closes. One call per type; no wildcard. */
|
|
4479
|
+
addIssueTracker(type: string, tracker: ActiveIssueTracker): this;
|
|
4480
|
+
removeIssueTracker(tracker: ActiveIssueTracker): this;
|
|
4481
|
+
/**
|
|
4482
|
+
* Drop every issue in this scope, e.g. because the call closed.
|
|
4483
|
+
*
|
|
4484
|
+
* Deletes through {@link delete} so the parent sheds exactly this scope's issues. Tracker
|
|
4485
|
+
* *registrations* survive: a detector subscribed to the observer's registry must keep receiving
|
|
4486
|
+
* issues after any one call ends.
|
|
4487
|
+
*/
|
|
4488
|
+
clear(): void;
|
|
4489
|
+
private _trackIssue;
|
|
4490
|
+
private _untrackIssue;
|
|
4491
|
+
/**
|
|
4492
|
+
* Apply `apply` to every tracker interested in `type`.
|
|
4493
|
+
*
|
|
4494
|
+
* A tracker throwing must not abort the fan-out: the issue has already been added to (or removed
|
|
4495
|
+
* from) this registry, so a partial dispatch would leave the remaining trackers permanently out of
|
|
4496
|
+
* step with it. One broken detector should not desynchronise the others.
|
|
4497
|
+
*/
|
|
4498
|
+
private _trackersOf;
|
|
4499
|
+
private _safely;
|
|
4500
|
+
}
|
|
4501
|
+
|
|
4502
|
+
type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
4503
|
+
callId: string;
|
|
4504
|
+
appData?: AppData;
|
|
4505
|
+
closeCallIfEmptyForMs?: number;
|
|
4506
|
+
/**
|
|
4507
|
+
* When `true`, the call's `update()` is invoked whenever a client accepts a sample. When `false`, it is not.
|
|
4508
|
+
*
|
|
4509
|
+
* DEFAULT: `true` — the call is updated on every client sample, which is the most common use case.
|
|
4510
|
+
*/
|
|
4511
|
+
autoUpdateOnClientUpdate?: boolean;
|
|
4512
|
+
};
|
|
4513
|
+
type ObservedCallEvents = {
|
|
4514
|
+
update: [];
|
|
4515
|
+
newclient: [ObservedClient];
|
|
4516
|
+
empty: [];
|
|
4517
|
+
'not-empty': [];
|
|
4518
|
+
close: [];
|
|
4519
|
+
};
|
|
4520
|
+
declare interface ObservedCall {
|
|
4521
|
+
on<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
|
|
4522
|
+
off<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
|
|
4523
|
+
once<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
|
|
4524
|
+
emit<U extends keyof ObservedCallEvents>(event: U, ...args: ObservedCallEvents[U]): boolean;
|
|
4525
|
+
}
|
|
4526
|
+
declare class ObservedCall<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
|
|
4527
|
+
readonly observer: Observer;
|
|
4528
|
+
readonly activeIssuesRegistry: ActiveIssuesRegistry;
|
|
4529
|
+
scoreCalculator: ScoreCalculator;
|
|
4530
|
+
readonly detectors: Detectors;
|
|
4531
|
+
readonly callId: string;
|
|
4532
|
+
readonly observedClients: Map<string, ObservedClient<Record<string, unknown>>>;
|
|
4533
|
+
readonly clientsUsedTurn: Set<string>;
|
|
4534
|
+
readonly calculatedScore: CalculatedScore;
|
|
4535
|
+
remoteTrackResolver?: RemoteTrackResolver;
|
|
4536
|
+
/**
|
|
4537
|
+
* Published tracks that currently have **no** subscriber linked to them.
|
|
4538
|
+
*
|
|
4539
|
+
* Maintained by the `RemoteTrackResolver` at the exact moments a track gains or loses its last
|
|
4540
|
+
* subscriber — the only moments the answer can change. `UnconsumedTrackDetector` reads this
|
|
4541
|
+
* instead of walking every published track in the call, so in a healthy call (where the set is
|
|
4542
|
+
* empty) it does no work at all.
|
|
4543
|
+
*
|
|
4544
|
+
* Empty when no resolver is configured: without links, "no subscribers" is unknowable.
|
|
4545
|
+
*/
|
|
4546
|
+
readonly unconsumedOutboundTracks: Set<ObservedOutboundTrack>;
|
|
4547
|
+
totalAddedClients: number;
|
|
4548
|
+
totalRemovedClients: number;
|
|
4549
|
+
numberOfIssues: number;
|
|
4550
|
+
numberOfPeerConnections: number;
|
|
4551
|
+
numberOfInboundRtpStreams: number;
|
|
4552
|
+
numberOfOutboundRtpStreams: number;
|
|
4553
|
+
numberOfDataChannels: number;
|
|
4554
|
+
maxNumberOfClients: number;
|
|
4555
|
+
deltaNumberOfIssues: number;
|
|
4556
|
+
appData: AppData;
|
|
4557
|
+
closed: boolean;
|
|
2722
4558
|
startedAt?: number;
|
|
2723
4559
|
endedAt?: number;
|
|
2724
4560
|
closedAt?: number;
|
|
2725
|
-
readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs'>;
|
|
4561
|
+
readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs' | 'autoUpdateOnClientUpdate'>;
|
|
2726
4562
|
/** Ancestry base shared by all Observer-bus events originating at this call. */
|
|
2727
4563
|
readonly eventScope: ObservedCallScope;
|
|
2728
4564
|
private closeTimer?;
|
|
2729
|
-
constructor(settings: ObservedCallSettings<AppData>, observer: Observer);
|
|
4565
|
+
constructor(settings: ObservedCallSettings<AppData>, observer: Observer, activeIssuesRegistry: ActiveIssuesRegistry);
|
|
2730
4566
|
get numberOfClients(): number;
|
|
2731
4567
|
get score(): number | undefined;
|
|
2732
|
-
|
|
2733
|
-
|
|
4568
|
+
addDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
|
|
4569
|
+
/**
|
|
4570
|
+
* Raise a call-level (server-side) finding; surfaced on the Observer bus as `call-issue`.
|
|
4571
|
+
*
|
|
4572
|
+
* `payload` takes an **object** — it is delivered to in-process handlers, so there is nothing to
|
|
4573
|
+
* serialise for. Pass a string only if you already have one.
|
|
4574
|
+
*/
|
|
4575
|
+
addIssue(issue: ObserverIssue): void;
|
|
2734
4576
|
close(): void;
|
|
2735
4577
|
getObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(clientId: string): ObservedClient<ClientAppData> | undefined;
|
|
2736
4578
|
createObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>): ObservedClient<ClientAppData> | undefined;
|
|
@@ -2759,6 +4601,378 @@ declare class MiddlewareProcessor<T> implements Processor<T> {
|
|
|
2759
4601
|
process(value: T): void;
|
|
2760
4602
|
}
|
|
2761
4603
|
|
|
4604
|
+
/** Raised once if one receiver turns out to be dragging a publisher down for everyone. */
|
|
4605
|
+
declare const LOWEST_COMMON_DENOMINATOR_ISSUE = "WORST_RECEIVER_CONTAGION";
|
|
4606
|
+
/** The measurements behind a decided verdict — everything needed to check the call yourself. */
|
|
4607
|
+
type SimulcastReceiverEvidence = {
|
|
4608
|
+
callId: string;
|
|
4609
|
+
trackId: string;
|
|
4610
|
+
publisherClientId: string;
|
|
4611
|
+
worstReceiverClientId: string;
|
|
4612
|
+
publisherBitrate: number;
|
|
4613
|
+
worstReceiverBitrate: number;
|
|
4614
|
+
medianReceiverBitrate: number;
|
|
4615
|
+
/** How closely the publisher's bitrate followed the **worst** receiver's, `0..1`. */
|
|
4616
|
+
trackingWithWorst: number;
|
|
4617
|
+
/** The same against the **median** receiver — the control. */
|
|
4618
|
+
trackingWithMedian: number;
|
|
4619
|
+
};
|
|
4620
|
+
type SimulcastReceiverReportPayload = ({
|
|
4621
|
+
/** The publisher held up while one receiver lagged: layers are being chosen per consumer. */
|
|
4622
|
+
verdict: 'layer-decided-per-receiver';
|
|
4623
|
+
evidence: SimulcastReceiverEvidence;
|
|
4624
|
+
} | {
|
|
4625
|
+
/** The publisher tracked its worst receiver: everyone is getting the lowest common denominator. */
|
|
4626
|
+
verdict: 'layer-decided-lowest-common-denominator';
|
|
4627
|
+
evidence: SimulcastReceiverEvidence;
|
|
4628
|
+
} | {
|
|
4629
|
+
/** Gave up without the conditions needed to judge. **Not a pass.** */
|
|
4630
|
+
verdict: 'inconclusive';
|
|
4631
|
+
reason: string;
|
|
4632
|
+
}) & {
|
|
4633
|
+
startedAt: number;
|
|
4634
|
+
/** How many times the check actually ran — i.e. how much the verdict is worth. */
|
|
4635
|
+
checks: number;
|
|
4636
|
+
};
|
|
4637
|
+
type SimulcastReceiverValidatorConfig = {
|
|
4638
|
+
/** Minimum receivers of a published track before the comparison means anything. */
|
|
4639
|
+
minReceivers: number;
|
|
4640
|
+
/** How long a track's bitrates are correlated over (ms). */
|
|
4641
|
+
windowMs: number;
|
|
4642
|
+
/** Samples needed inside the window before it can be judged. */
|
|
4643
|
+
minSamples: number;
|
|
4644
|
+
/** The worst receiver must be at most this share of the median, or there is nothing to be dragged by. */
|
|
4645
|
+
outlierRatioThreshold: number;
|
|
4646
|
+
/** How closely the publisher must follow the worst receiver to count as dragged. */
|
|
4647
|
+
trackingRatioThreshold: number;
|
|
4648
|
+
/** Clean checks required before concluding per-receiver adaptation. One could be luck. */
|
|
4649
|
+
minChecks: number;
|
|
4650
|
+
};
|
|
4651
|
+
/**
|
|
4652
|
+
* Answers one question: **does this SFU adapt each receiver on its own, or does one bad receiver
|
|
4653
|
+
* drag the publisher down for everyone?**
|
|
4654
|
+
*
|
|
4655
|
+
* That is what simulcast (or SVC) exists to prevent. With several encodings available the server can
|
|
4656
|
+
* hand the struggling participant a lower layer and leave everyone else alone. Without it — or with
|
|
4657
|
+
* a server that relays RTCP end to end instead of terminating it, so the publisher's bandwidth
|
|
4658
|
+
* estimate collapses to the minimum across all receivers — the only way to serve the slowest
|
|
4659
|
+
* participant is to make the source send less, and everybody gets the lowest common denominator.
|
|
4660
|
+
*
|
|
4661
|
+
* The two causes are worth naming because the *observation* cannot separate them: the publisher's
|
|
4662
|
+
* bitrate tracking its worst receiver looks identical either way. What the check establishes is
|
|
4663
|
+
* whether per-receiver adaptation is happening at all. If the verdict is
|
|
4664
|
+
* `layer-decided-lowest-common-denominator`, look at both — is simulcast/SVC actually enabled with
|
|
4665
|
+
* layers selected per consumer, and is the SFU terminating receiver reports rather than forwarding
|
|
4666
|
+
* them?
|
|
4667
|
+
*
|
|
4668
|
+
* ### The control matters more than the correlation
|
|
4669
|
+
*
|
|
4670
|
+
* "Publisher follows worst receiver" alone proves nothing: when the whole call degrades together,
|
|
4671
|
+
* the publisher follows *everyone*, and that is ordinary adaptation working correctly. The verdict
|
|
4672
|
+
* only goes against the deployment when the publisher tracks the worst receiver **more closely than
|
|
4673
|
+
* it tracks the median** — the worst receiver is leading, not merely coinciding.
|
|
4674
|
+
*
|
|
4675
|
+
* Likewise, a window with no outlier in it is not evidence of health, it is an untested SFU: if
|
|
4676
|
+
* nobody is struggling, there is nothing for per-receiver adaptation to do. Those windows are
|
|
4677
|
+
* skipped and never counted in `checks`.
|
|
4678
|
+
*
|
|
4679
|
+
* ### Why a validator, not a detector
|
|
4680
|
+
*
|
|
4681
|
+
* This is a property of the SFU build and configuration, not of this moment: a server doing
|
|
4682
|
+
* per-receiver layer selection at 09:00 still is at 17:00. Re-deriving it every tick cannot produce
|
|
4683
|
+
* new information — it would only keep a sliding window alive per published track for the life of
|
|
4684
|
+
* every call. So it decides once, reports, and releases that state.
|
|
4685
|
+
*
|
|
4686
|
+
* ```ts
|
|
4687
|
+
* observer.on('validation-ready', ({ validator, report }) => {
|
|
4688
|
+
* if (validator !== 'simulcast-receivers' || !report.ready) return;
|
|
4689
|
+
* console.log(report.verdict); // 'layer-decided-per-receiver' | ... | 'inconclusive'
|
|
4690
|
+
* });
|
|
4691
|
+
*
|
|
4692
|
+
* observer.addValidator('simulcast-receivers');
|
|
4693
|
+
* onDeploy(() => observer.addValidator('simulcast-receivers')); // check again
|
|
4694
|
+
* ```
|
|
4695
|
+
*
|
|
4696
|
+
* ### `inconclusive` is not a pass
|
|
4697
|
+
*
|
|
4698
|
+
* The check only runs when a publisher has several receivers and one of them is far behind the
|
|
4699
|
+
* median; plenty of healthy deployments never present that. Concluding from the absence of a failure
|
|
4700
|
+
* would verify nothing, so `checks` counts the times the check genuinely ran, and a validator that
|
|
4701
|
+
* is cancelled (or whose observer closes) finishes `inconclusive` with the reason why.
|
|
4702
|
+
*/
|
|
4703
|
+
declare class SimulcastReceiverValidator implements Validator<SimulcastReceiverReportPayload> {
|
|
4704
|
+
private readonly _observer;
|
|
4705
|
+
readonly onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void;
|
|
4706
|
+
static readonly NAME: "simulcast-receivers";
|
|
4707
|
+
readonly name: "simulcast-receivers";
|
|
4708
|
+
readonly startedAt: number;
|
|
4709
|
+
report: ValidationReport<SimulcastReceiverReportPayload>;
|
|
4710
|
+
private readonly _config;
|
|
4711
|
+
private readonly _windows;
|
|
4712
|
+
private _checks;
|
|
4713
|
+
private _done;
|
|
4714
|
+
constructor(_observer: Observer, onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void, config?: Partial<SimulcastReceiverValidatorConfig>);
|
|
4715
|
+
/** How many times the comparison actually ran. `0` means nothing was established. */
|
|
4716
|
+
get checks(): number;
|
|
4717
|
+
/** Give up without a verdict, freeing anything waiting on this validator. */
|
|
4718
|
+
cancel(reason?: string): void;
|
|
4719
|
+
update(): void;
|
|
4720
|
+
/** Returns `true` when a verdict was reached and the caller should stop iterating. */
|
|
4721
|
+
private _inspect;
|
|
4722
|
+
/**
|
|
4723
|
+
* Settle on a verdict, exactly once.
|
|
4724
|
+
*
|
|
4725
|
+
* The guard is not paranoia: `onDone` removes this validator from the observer, and a second call
|
|
4726
|
+
* would emit a second `validation-ready` for a validator that is no longer registered — e.g. when
|
|
4727
|
+
* `observer.close()` cancels a validator that decided earlier in the same tick.
|
|
4728
|
+
*/
|
|
4729
|
+
private _finish;
|
|
4730
|
+
private _windowOf;
|
|
4731
|
+
}
|
|
4732
|
+
|
|
4733
|
+
/** Raised once if the resolver turns out never to link anything. */
|
|
4734
|
+
declare const UNRESOLVED_TRACK_LINKS_ISSUE = "REMOTE_TRACK_LINKS_UNRESOLVED";
|
|
4735
|
+
/** What the check actually saw, whichever way it went. */
|
|
4736
|
+
type RemoteTrackLinkEvidence = {
|
|
4737
|
+
/** Calls that presented the conditions for linking: a resolver, ≥2 clients, and inbound tracks. */
|
|
4738
|
+
eligibleCalls: number;
|
|
4739
|
+
/** Inbound tracks seen across those calls. */
|
|
4740
|
+
inboundTracks: number;
|
|
4741
|
+
/** Of those, how many were linked to the outbound track that published them. */
|
|
4742
|
+
linkedInboundTracks: number;
|
|
4743
|
+
/** `linkedInboundTracks / inboundTracks`. */
|
|
4744
|
+
linkedRatio: number;
|
|
4745
|
+
/** A call that presented the conditions, for the reader to go and look at. */
|
|
4746
|
+
exampleCallId?: string;
|
|
4747
|
+
};
|
|
4748
|
+
type RemoteTrackResolverReportPayload = ({
|
|
4749
|
+
/** The resolver is linking subscribers to publishers. The detectors that need links will work. */
|
|
4750
|
+
verdict: 'links-resolved';
|
|
4751
|
+
evidence: RemoteTrackLinkEvidence;
|
|
4752
|
+
} | {
|
|
4753
|
+
/** Every condition for linking was met, repeatedly, and nothing was ever linked. */
|
|
4754
|
+
verdict: 'no-links-resolved';
|
|
4755
|
+
evidence: RemoteTrackLinkEvidence;
|
|
4756
|
+
} | {
|
|
4757
|
+
/** Never saw a call that could have been linked. **Not a pass.** */
|
|
4758
|
+
verdict: 'inconclusive';
|
|
4759
|
+
reason: string;
|
|
4760
|
+
}) & {
|
|
4761
|
+
startedAt: number;
|
|
4762
|
+
/** How many times the check genuinely ran — i.e. how much the verdict is worth. */
|
|
4763
|
+
checks: number;
|
|
4764
|
+
};
|
|
4765
|
+
type RemoteTrackResolverValidatorConfig = {
|
|
4766
|
+
/** Participants a call needs before it can plausibly have publisher↔subscriber links. Default `2`. */
|
|
4767
|
+
minClients: number;
|
|
4768
|
+
/** Inbound tracks that must be present in a call before it counts as a check. Default `2`. */
|
|
4769
|
+
minInboundTracks: number;
|
|
4770
|
+
/** Share of inbound tracks that must be linked to conclude the resolver works. Default `0.5`. */
|
|
4771
|
+
linkedRatioThreshold: number;
|
|
4772
|
+
/** Eligible calls to observe before concluding either way. One could be a race. Default `3`. */
|
|
4773
|
+
minChecks: number;
|
|
4774
|
+
};
|
|
4775
|
+
/**
|
|
4776
|
+
* Answers one question: **is the `RemoteTrackResolver` actually linking anything?**
|
|
4777
|
+
*
|
|
4778
|
+
* ### Why this is worth a validator
|
|
4779
|
+
*
|
|
4780
|
+
* Four things in this library are built on publisher↔subscriber links —
|
|
4781
|
+
* `IssueFanOutDetector`, `TrackDeliveryMismatchDetector`, `UnconsumedTrackDetector` and
|
|
4782
|
+
* `SimulcastReceiverValidator`. Every one of them checks `call.remoteTrackResolver` and, finding no
|
|
4783
|
+
* links, correctly does nothing rather than guessing.
|
|
4784
|
+
*
|
|
4785
|
+
* That is the right behaviour and it produces a nasty failure mode: a resolver wired to the wrong id
|
|
4786
|
+
* field, or a mediasoup `producerId` the application never attaches, leaves all four permanently
|
|
4787
|
+
* silent — and **silence is what a healthy deployment looks like too**. You would conclude your
|
|
4788
|
+
* calls were clean when in fact nothing was ever examined. This check exists to make that specific
|
|
4789
|
+
* mistake loud.
|
|
4790
|
+
*
|
|
4791
|
+
* ### `inconclusive` is not a pass
|
|
4792
|
+
*
|
|
4793
|
+
* A verdict is only reached from calls that *could* have been linked: a resolver configured, at
|
|
4794
|
+
* least `minClients` participants, and at least `minInboundTracks` inbound tracks present. A
|
|
4795
|
+
* one-to-one deployment, a lobby full of audio-only listeners, or a quiet period never presents
|
|
4796
|
+
* those conditions — and concluding "resolver works" from calls that had nothing to resolve would be
|
|
4797
|
+
* the very mistake this validator is here to catch. `checks` counts the eligible calls actually
|
|
4798
|
+
* seen; a validator cancelled before reaching `minChecks` finishes `inconclusive` and says so.
|
|
4799
|
+
*
|
|
4800
|
+
* ```ts
|
|
4801
|
+
* observer.on('validation-ready', ({ validator, report }) => {
|
|
4802
|
+
* if (validator !== 'remote-track-resolver' || !report.ready) return;
|
|
4803
|
+
* if (report.verdict === 'no-links-resolved') alert('resolver misconfigured — 4 detectors are inert');
|
|
4804
|
+
* });
|
|
4805
|
+
*
|
|
4806
|
+
* observer.addValidator('remote-track-resolver');
|
|
4807
|
+
* ```
|
|
4808
|
+
*
|
|
4809
|
+
* Run it once at start-up, and again after changing the resolver or the SFU's id scheme. Like every
|
|
4810
|
+
* validator it is one-shot: the answer is a property of the wiring, not of this moment.
|
|
4811
|
+
*/
|
|
4812
|
+
declare class RemoteTrackResolverValidator implements Validator<RemoteTrackResolverReportPayload> {
|
|
4813
|
+
private readonly _observer;
|
|
4814
|
+
readonly onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void;
|
|
4815
|
+
static readonly NAME: "remote-track-resolver";
|
|
4816
|
+
readonly name: "remote-track-resolver";
|
|
4817
|
+
readonly startedAt: number;
|
|
4818
|
+
report: ValidationReport<RemoteTrackResolverReportPayload>;
|
|
4819
|
+
private readonly _config;
|
|
4820
|
+
/** Accumulated across every eligible call seen, so one small call cannot decide alone. */
|
|
4821
|
+
private _eligibleCalls;
|
|
4822
|
+
private _inboundTracks;
|
|
4823
|
+
private _linkedInboundTracks;
|
|
4824
|
+
private _exampleCallId?;
|
|
4825
|
+
private _checks;
|
|
4826
|
+
private _done;
|
|
4827
|
+
constructor(_observer: Observer, onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void, config?: Partial<RemoteTrackResolverValidatorConfig>);
|
|
4828
|
+
/** How many eligible calls were actually examined. `0` means nothing was established. */
|
|
4829
|
+
get checks(): number;
|
|
4830
|
+
cancel(reason?: string): void;
|
|
4831
|
+
update(): void;
|
|
4832
|
+
private _evidence;
|
|
4833
|
+
private _finish;
|
|
4834
|
+
}
|
|
4835
|
+
|
|
4836
|
+
/** Raised once if the deployment is not actually delivering the codec it thinks it is. */
|
|
4837
|
+
declare const CODEC_MISMATCH_ISSUE = "CODEC_INCONSISTENCY";
|
|
4838
|
+
/** What the check saw across a call's participants. */
|
|
4839
|
+
type CodecEvidence = {
|
|
4840
|
+
callId: string;
|
|
4841
|
+
kind: 'audio' | 'video';
|
|
4842
|
+
/** Every mime type in use in that call, most common first — e.g. `[ 'video/VP8', 'video/H264' ]`. */
|
|
4843
|
+
mimeTypes: string[];
|
|
4844
|
+
/** How many clients used each, in the same order as {@link mimeTypes}. */
|
|
4845
|
+
clientsPerMimeType: number[];
|
|
4846
|
+
/** Clients considered — those that reported at least one codec of this kind. */
|
|
4847
|
+
clients: number;
|
|
4848
|
+
/** The codec the check was told to expect, when it was given one. */
|
|
4849
|
+
expected?: string;
|
|
4850
|
+
};
|
|
4851
|
+
type CodecConsistencyReportPayload = ({
|
|
4852
|
+
/** One codec per media kind, and it is the expected one if an expectation was given. */
|
|
4853
|
+
verdict: 'codec-consistent';
|
|
4854
|
+
evidence: CodecEvidence[];
|
|
4855
|
+
} | {
|
|
4856
|
+
/** Participants of one call are split across different codecs. */
|
|
4857
|
+
verdict: 'codec-split';
|
|
4858
|
+
evidence: CodecEvidence[];
|
|
4859
|
+
} | {
|
|
4860
|
+
/** Consistent, but not what the deployment believes it negotiated. */
|
|
4861
|
+
verdict: 'unexpected-codec';
|
|
4862
|
+
evidence: CodecEvidence[];
|
|
4863
|
+
} | {
|
|
4864
|
+
/** Never saw a call with enough participants reporting codecs. **Not a pass.** */
|
|
4865
|
+
verdict: 'inconclusive';
|
|
4866
|
+
reason: string;
|
|
4867
|
+
}) & {
|
|
4868
|
+
startedAt: number;
|
|
4869
|
+
checks: number;
|
|
4870
|
+
};
|
|
4871
|
+
type CodecConsistencyValidatorConfig = {
|
|
4872
|
+
/**
|
|
4873
|
+
* The mime type you believe you are delivering, per kind — e.g.
|
|
4874
|
+
* `{ video: 'video/VP8', audio: 'audio/opus' }`.
|
|
4875
|
+
*
|
|
4876
|
+
* Optional. Without it the check still reports a *split* (participants disagreeing with each
|
|
4877
|
+
* other), which needs no expectation to be a fact. With it, the check can additionally catch the
|
|
4878
|
+
* case where everyone agrees on the wrong thing — a silent fallback that nothing else notices.
|
|
4879
|
+
*/
|
|
4880
|
+
expected?: Partial<Record<'audio' | 'video', string>>;
|
|
4881
|
+
/** Which kinds to inspect. Default both. */
|
|
4882
|
+
kinds: ('audio' | 'video')[];
|
|
4883
|
+
/** Participants a call needs before disagreement is meaningful. Default `3`. */
|
|
4884
|
+
minClients: number;
|
|
4885
|
+
/** Calls to inspect before concluding. Default `3`. */
|
|
4886
|
+
minChecks: number;
|
|
4887
|
+
};
|
|
4888
|
+
/**
|
|
4889
|
+
* Answers: **is every participant of a call actually using the same codec — and is it the one you
|
|
4890
|
+
* think you negotiated?**
|
|
4891
|
+
*
|
|
4892
|
+
* ### Why the server has to answer this
|
|
4893
|
+
*
|
|
4894
|
+
* A client knows only its own codec. It cannot tell whether it is the odd one out, and an SFU that
|
|
4895
|
+
* forwards without transcoding cannot serve a call where participants disagree — so a split is a
|
|
4896
|
+
* real fault with a very confusing symptom: some pairs of participants see each other and some do
|
|
4897
|
+
* not, with no error anywhere. Only something holding every participant of a call at once can see
|
|
4898
|
+
* the split at all.
|
|
4899
|
+
*
|
|
4900
|
+
* The second half is the quieter failure. A deployment configured for VP9 or AV1 will fall back to
|
|
4901
|
+
* VP8 whenever one endpoint cannot negotiate the preferred codec, and nothing reports that — the
|
|
4902
|
+
* call works, the bitrate is higher than it should be, and the team believes it shipped AV1 months
|
|
4903
|
+
* ago. Give the check an `expected` mime type and it will say so.
|
|
4904
|
+
*
|
|
4905
|
+
* ### Why a validator and not a detector
|
|
4906
|
+
*
|
|
4907
|
+
* The answer is a property of the deployment — SDP munging, codec preferences, the SFU build — not
|
|
4908
|
+
* of this moment. A deployment that negotiates VP8 at 09:00 negotiates VP8 at 17:00. Re-deriving it
|
|
4909
|
+
* every tick would walk every codec of every peer connection of every call, forever, to re-learn a
|
|
4910
|
+
* constant. So it decides once and stops.
|
|
4911
|
+
*
|
|
4912
|
+
* Start it again after a deploy, or after changing codec preferences:
|
|
4913
|
+
*
|
|
4914
|
+
* ```ts
|
|
4915
|
+
* observer.addValidator('codec-consistency', {
|
|
4916
|
+
* expected: { video: 'video/VP9', audio: 'audio/opus' },
|
|
4917
|
+
* });
|
|
4918
|
+
*
|
|
4919
|
+
* observer.on('validation-ready', ({ validator, report }) => {
|
|
4920
|
+
* if (validator !== 'codec-consistency' || !report.ready) return;
|
|
4921
|
+
* // 'codec-consistent' | 'codec-split' | 'unexpected-codec' | 'inconclusive'
|
|
4922
|
+
* });
|
|
4923
|
+
* ```
|
|
4924
|
+
*
|
|
4925
|
+
* ### `inconclusive` is not a pass
|
|
4926
|
+
*
|
|
4927
|
+
* Only calls with at least `minClients` participants *reporting codecs of that kind* count as a
|
|
4928
|
+
* check. An audio-only deployment will never say anything about video, and concluding "video codecs
|
|
4929
|
+
* are consistent" from calls that carried no video would verify nothing.
|
|
4930
|
+
*
|
|
4931
|
+
* ### Comparison is by mime type only
|
|
4932
|
+
*
|
|
4933
|
+
* `sdpFmtpLine` carries profile and level — `profile-level-id` for H.264, `profile-id` for VP9 — and
|
|
4934
|
+
* two clients on the same mime type with different profiles are not truly interchangeable. That is
|
|
4935
|
+
* deliberately out of scope: fmtp differences are common, usually benign, and would make this check
|
|
4936
|
+
* noisy enough to ignore. It answers the coarse question, which is the one that is actually wrong in
|
|
4937
|
+
* practice.
|
|
4938
|
+
*/
|
|
4939
|
+
declare class CodecConsistencyValidator implements Validator<CodecConsistencyReportPayload> {
|
|
4940
|
+
private readonly _observer;
|
|
4941
|
+
readonly onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void;
|
|
4942
|
+
static readonly NAME: "codec-consistency";
|
|
4943
|
+
readonly name: "codec-consistency";
|
|
4944
|
+
readonly startedAt: number;
|
|
4945
|
+
report: ValidationReport<CodecConsistencyReportPayload>;
|
|
4946
|
+
private readonly _config;
|
|
4947
|
+
private readonly _inspectedCallIds;
|
|
4948
|
+
private _evidence;
|
|
4949
|
+
private _checks;
|
|
4950
|
+
private _done;
|
|
4951
|
+
constructor(_observer: Observer, onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void, config?: Partial<CodecConsistencyValidatorConfig>);
|
|
4952
|
+
get checks(): number;
|
|
4953
|
+
cancel(reason?: string): void;
|
|
4954
|
+
update(): void;
|
|
4955
|
+
/** One evidence entry per configured kind that the call actually carried. */
|
|
4956
|
+
private _tallyOf;
|
|
4957
|
+
private _finish;
|
|
4958
|
+
}
|
|
4959
|
+
|
|
4960
|
+
/**
|
|
4961
|
+
* The validators `observer.addValidator(name, config)` knows how to build, and the config each takes.
|
|
4962
|
+
*
|
|
4963
|
+
* Adding one means: write the class with a `static readonly NAME`, add its entry here, and add a
|
|
4964
|
+
* `case` to `addValidator`. The map is what gives the call site its types —
|
|
4965
|
+
* `addValidator('simulcast-receivers', { … })` type-checks the config against the right validator,
|
|
4966
|
+
* and an unknown name won't compile.
|
|
4967
|
+
*/
|
|
4968
|
+
type AvailableValidatorConfigs = {
|
|
4969
|
+
[SimulcastReceiverValidator.NAME]: SimulcastReceiverValidatorConfig;
|
|
4970
|
+
[RemoteTrackResolverValidator.NAME]: RemoteTrackResolverValidatorConfig;
|
|
4971
|
+
[CodecConsistencyValidator.NAME]: CodecConsistencyValidatorConfig;
|
|
4972
|
+
};
|
|
4973
|
+
/** A validator name that can be started. */
|
|
4974
|
+
type ValidatorName = keyof AvailableValidatorConfigs;
|
|
4975
|
+
|
|
2762
4976
|
type SampleRejectedReason = 'observer-closed' | 'missing-callId' | 'missing-clientId';
|
|
2763
4977
|
|
|
2764
4978
|
/**
|
|
@@ -2779,9 +4993,6 @@ type AcceptMiddlewarePayload = {
|
|
|
2779
4993
|
* `next(payload)` to continue the chain. Not calling `next` **drops** the sample.
|
|
2780
4994
|
*/
|
|
2781
4995
|
type AcceptMiddleware = Middleware<AcceptMiddlewarePayload>;
|
|
2782
|
-
type ObserverUpdateConfig = {
|
|
2783
|
-
updatePolicy?: 'update-on-any-call-updated' | 'update-when-all-call-updated' | 'none';
|
|
2784
|
-
};
|
|
2785
4996
|
/** Produces the initial `appData` for a call created without an explicit `appData`. */
|
|
2786
4997
|
type CallAppDataFactory = (params: {
|
|
2787
4998
|
callId: string;
|
|
@@ -2792,11 +5003,30 @@ type ClientAppDataFactory = (params: {
|
|
|
2792
5003
|
clientId: string;
|
|
2793
5004
|
observedCall: ObservedCall;
|
|
2794
5005
|
}) => Record<string, unknown>;
|
|
2795
|
-
type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> =
|
|
2796
|
-
defaultCallUpdatePolicy?: ObservedCallSettings['updatePolicy'];
|
|
5006
|
+
type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = {
|
|
2797
5007
|
appData?: AppData;
|
|
2798
5008
|
closeClientIfIdleForMs?: number;
|
|
2799
5009
|
closeCallIfEmptyForMs?: number;
|
|
5010
|
+
/**
|
|
5011
|
+
* When `true` (the default), every call update triggers an observer-wide `update()` pass.
|
|
5012
|
+
*
|
|
5013
|
+
* There is deliberately no timer and no separate policy object: a call is updated when any of its
|
|
5014
|
+
* clients is, and the observer is updated when any of its calls is — so the observer is updated
|
|
5015
|
+
* exactly when any client anywhere is. Set to `false` only if you drive `observer.update()`
|
|
5016
|
+
* yourself, and note that observer-scoped detectors and validators run *nowhere else*.
|
|
5017
|
+
*/
|
|
5018
|
+
autoUpdateOnCallUpdate?: boolean;
|
|
5019
|
+
inboundTrackDegradationThresholds?: {
|
|
5020
|
+
deltaFreezeCount: number;
|
|
5021
|
+
framesDroppedRatio: number;
|
|
5022
|
+
jitterBufferDelayInMs: number;
|
|
5023
|
+
concealmentRatio: number;
|
|
5024
|
+
rttInMs: number;
|
|
5025
|
+
};
|
|
5026
|
+
outboundTrackDegradationThresholds?: {
|
|
5027
|
+
fractionLost: number;
|
|
5028
|
+
rttInMs: number;
|
|
5029
|
+
};
|
|
2800
5030
|
/**
|
|
2801
5031
|
* Optional factory invoked when a call is created without an explicit `appData`
|
|
2802
5032
|
* (e.g. lazily by `accept()`), so apps can enrich appData without pre-creating the
|
|
@@ -2826,13 +5056,12 @@ declare interface Observer {
|
|
|
2826
5056
|
emit<U extends keyof ObserverEvents>(event: U, ...args: ObserverEvents[U]): boolean;
|
|
2827
5057
|
}
|
|
2828
5058
|
declare class Observer<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
|
|
2829
|
-
readonly config: ObserverConfig<AppData>;
|
|
2830
5059
|
readonly observedTURN: ObservedTURN;
|
|
2831
5060
|
readonly observedCalls: Map<string, ObservedCall<Record<string, unknown>>>;
|
|
2832
5061
|
readonly observedMediasoupRouters: Map<string, ObservedMediasoupRouter<Record<string, unknown>>>;
|
|
2833
|
-
updater?: Updater;
|
|
2834
5062
|
/** Ancestry base shared by all Observer-bus events originating at the observer. */
|
|
2835
5063
|
readonly eventScope: ObserverEventBase;
|
|
5064
|
+
readonly config: ObserverConfig<AppData>;
|
|
2836
5065
|
closed: boolean;
|
|
2837
5066
|
totalAddedCall: number;
|
|
2838
5067
|
totalRemovedCall: number;
|
|
@@ -2844,9 +5073,69 @@ declare class Observer<AppData extends Record<string, unknown> = Record<string,
|
|
|
2844
5073
|
numberOfPeerConnections: number;
|
|
2845
5074
|
/** Global, pre-dispatch middleware chain run on every accepted sample. */
|
|
2846
5075
|
readonly acceptMiddlewares: MiddlewareProcessor<AcceptMiddlewarePayload>;
|
|
2847
|
-
|
|
5076
|
+
/**
|
|
5077
|
+
* Fleet-wide index of every open client issue, maintained incrementally as issues open and close.
|
|
5078
|
+
* Each call's index propagates into this one, so cross-call queries cost O(matching issues) rather
|
|
5079
|
+
* than a walk over every call and client. This is what observer-scoped detectors read.
|
|
5080
|
+
*/
|
|
5081
|
+
readonly activeIssuesRegistry: ActiveIssuesRegistry;
|
|
5082
|
+
/**
|
|
5083
|
+
* Validators currently running. Each removes itself when it finishes, so this is normally empty —
|
|
5084
|
+
* a validator is a one-shot check, not a permanent fixture. Start one with {@link addValidator}.
|
|
5085
|
+
*/
|
|
5086
|
+
readonly validators: Set<RunningValidator>;
|
|
5087
|
+
/**
|
|
5088
|
+
* Observer-scoped detector registry, run on every `observer.update()` — the place for findings
|
|
5089
|
+
* that span **calls**, e.g. "many calls on the same SFU degraded at once". Detectors raise
|
|
5090
|
+
* findings with `observer.addIssue(...)`, surfaced on the bus as `observer-issue`.
|
|
5091
|
+
* (For findings within a single call use `observedCall.detectors`.)
|
|
5092
|
+
*
|
|
5093
|
+
* Starts **empty**. Populate it with {@link addObserverDetector}, or `detectors.add(...)` for an
|
|
5094
|
+
* instance you built yourself.
|
|
5095
|
+
*/
|
|
5096
|
+
readonly detectors: Detectors;
|
|
5097
|
+
/**
|
|
5098
|
+
* The call-scoped detectors to build on **every** call this observer creates, in registration
|
|
5099
|
+
* order. Written by {@link addCallDetector}; read by `createObservedCall`.
|
|
5100
|
+
*
|
|
5101
|
+
* Nothing is created implicitly. There is no detector configuration in `ObserverConfig` and no
|
|
5102
|
+
* default set, because a detector that nobody asked for is a detector nobody will act on: it costs
|
|
5103
|
+
* time on every tick and raises findings into a handler that was not written to expect them. An
|
|
5104
|
+
* application says what it wants to watch, or it watches nothing.
|
|
5105
|
+
*
|
|
5106
|
+
* ```ts
|
|
5107
|
+
* observer.addCallDetector('call-concurrent-issue-detector', {
|
|
5108
|
+
* issueTypes: [ 'congestion', 'ice-disconnected' ],
|
|
5109
|
+
* });
|
|
5110
|
+
* ```
|
|
5111
|
+
*/
|
|
5112
|
+
readonly callDetectorConfigs: Map<keyof AvailableCallScopeDetectorsConfigs, Partial<UnconsumedTrackDetectorConfig | TrackDeliveryMismatchDetectorConfig | CallConcurrentIssueDetectorConfig | IssueFanOutDetectorConfig | PublisherFaultCorroborationDetectorConfig>>;
|
|
5113
|
+
constructor(config?: Partial<ObserverConfig<AppData>>);
|
|
2848
5114
|
get numberOfCalls(): number;
|
|
2849
5115
|
get appData(): AppData | undefined;
|
|
5116
|
+
addObserverDetector<K extends keyof AvailableObserverScopeDetectorsConfigs>(name: K, config?: Partial<AvailableObserverScopeDetectorsConfigs[K]>): this;
|
|
5117
|
+
/**
|
|
5118
|
+
* Enable a call-scoped detector for calls created **from now on**.
|
|
5119
|
+
*
|
|
5120
|
+
* This edits the config, not the live calls: calls already open keep the detector set they were
|
|
5121
|
+
* built with. To add one to an existing call, use `observedCall.addDetector(...)` directly.
|
|
5122
|
+
*/
|
|
5123
|
+
addCallDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
|
|
5124
|
+
/** Stop building `name` on calls created from now on. Calls already open are untouched. */
|
|
5125
|
+
removeCallDetector(name: keyof AvailableCallScopeDetectorsConfigs): this;
|
|
5126
|
+
/**
|
|
5127
|
+
* Start a structural check. It runs on each `observer.update()` until it can decide, reports once
|
|
5128
|
+
* on `validation-ready`, and removes itself.
|
|
5129
|
+
*
|
|
5130
|
+
* ```ts
|
|
5131
|
+
* observer.validate('simulcast-receiver-validator', { minChecks: 5 });
|
|
5132
|
+
* ```
|
|
5133
|
+
*
|
|
5134
|
+
* Config keys are optional and merged over that validator's defaults. Call it again — after a
|
|
5135
|
+
* deploy, say — to check again; there is no revalidation timer, because a deploy rather than
|
|
5136
|
+
* elapsed time is what makes a structural verdict stale.
|
|
5137
|
+
*/
|
|
5138
|
+
addValidator<K extends keyof AvailableValidatorConfigs>(name: K, config?: Partial<AvailableValidatorConfigs[K]>): this;
|
|
2850
5139
|
getObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(callId: string): ObservedCall<T> | undefined;
|
|
2851
5140
|
createObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
|
|
2852
5141
|
getOrCreateObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
|
|
@@ -2856,6 +5145,13 @@ declare class Observer<AppData extends Record<string, unknown> = Record<string,
|
|
|
2856
5145
|
close(): void;
|
|
2857
5146
|
accept(sample: ClientSample, context?: AcceptContext): void;
|
|
2858
5147
|
update(): void;
|
|
5148
|
+
/**
|
|
5149
|
+
* Raise an observer-scoped (cross-call / SFU-wide) finding. Emitted on the bus as
|
|
5150
|
+
* `observer-issue`. Intended for `observer.detectors`, but the application may call it too.
|
|
5151
|
+
*
|
|
5152
|
+
* `payload` takes an **object**; see `ObserverIssue`.
|
|
5153
|
+
*/
|
|
5154
|
+
addIssue(issue: ObserverIssue): void;
|
|
2859
5155
|
/** Emit an Observer-bus event. */
|
|
2860
5156
|
private _notify;
|
|
2861
5157
|
}
|
|
@@ -2894,6 +5190,367 @@ declare enum ClientEventTypes {
|
|
|
2894
5190
|
DATA_CONSUMER_CLOSED = "DATA_CONSUMER_CLOSED"
|
|
2895
5191
|
}
|
|
2896
5192
|
|
|
5193
|
+
/** Thresholds deciding when a client counts as degraded on the receiving / sending side. */
|
|
5194
|
+
type ClientHealthThresholds = {
|
|
5195
|
+
/** Inbound loss fraction across the client's received streams (0..1). */
|
|
5196
|
+
inboundFractionLost: number;
|
|
5197
|
+
/** Loss fraction reported back about the client's sent streams via RTCP (0..1). */
|
|
5198
|
+
outboundFractionLost: number;
|
|
5199
|
+
/** Round-trip time (ms). */
|
|
5200
|
+
rttInMs: number;
|
|
5201
|
+
/** Freezes observed across the client's inbound video in the tick. */
|
|
5202
|
+
freezeCount: number;
|
|
5203
|
+
/** Concealment fraction across the client's inbound audio (0..1). */
|
|
5204
|
+
concealmentRatio: number;
|
|
5205
|
+
};
|
|
5206
|
+
declare const defaultClientHealthThresholds: ClientHealthThresholds;
|
|
5207
|
+
/** The per-client health view, split by direction. */
|
|
5208
|
+
type ClientHealth = {
|
|
5209
|
+
observedClient: ObservedClient;
|
|
5210
|
+
clientId: string;
|
|
5211
|
+
/** Receiving (download) side is impaired. */
|
|
5212
|
+
inboundDegraded: boolean;
|
|
5213
|
+
/** Sending (upload) side is impaired. */
|
|
5214
|
+
outboundDegraded: boolean;
|
|
5215
|
+
/** `inboundDegraded || outboundDegraded`. */
|
|
5216
|
+
degraded: boolean;
|
|
5217
|
+
reasons: string[];
|
|
5218
|
+
inboundFractionLost?: number;
|
|
5219
|
+
outboundFractionLost?: number;
|
|
5220
|
+
rttInMs?: number;
|
|
5221
|
+
deltaFreezeCount: number;
|
|
5222
|
+
concealmentRatio?: number;
|
|
5223
|
+
/** Quality-limitation reasons seen on this client's outbound video ('cpu' | 'bandwidth' | …). */
|
|
5224
|
+
qualityLimitationReasons: string[];
|
|
5225
|
+
usingTURN: boolean;
|
|
5226
|
+
usingTCP: boolean;
|
|
5227
|
+
};
|
|
5228
|
+
/** Call-level rollup of the per-client health, using percentiles rather than means. */
|
|
5229
|
+
type CallHealth = {
|
|
5230
|
+
callId: string;
|
|
5231
|
+
clients: ClientHealth[];
|
|
5232
|
+
numberOfClients: number;
|
|
5233
|
+
numberOfDegradedClients: number;
|
|
5234
|
+
numberOfInboundDegradedClients: number;
|
|
5235
|
+
numberOfOutboundDegradedClients: number;
|
|
5236
|
+
/** degradedClients / clients (0..1). */
|
|
5237
|
+
degradedRatio: number;
|
|
5238
|
+
inboundDegradedRatio: number;
|
|
5239
|
+
outboundDegradedRatio: number;
|
|
5240
|
+
rttInMs?: StatsSummary;
|
|
5241
|
+
inboundFractionLost?: StatsSummary;
|
|
5242
|
+
concealmentRatio?: StatsSummary;
|
|
5243
|
+
/** How many clients reported each quality-limitation reason on their outbound video. */
|
|
5244
|
+
qualityLimitation: {
|
|
5245
|
+
cpu: number;
|
|
5246
|
+
bandwidth: number;
|
|
5247
|
+
other: number;
|
|
5248
|
+
};
|
|
5249
|
+
freezes: {
|
|
5250
|
+
affectedClients: number;
|
|
5251
|
+
total: number;
|
|
5252
|
+
};
|
|
5253
|
+
};
|
|
5254
|
+
/**
|
|
5255
|
+
* Aggregates a call along the **client** axis (as `TrackDistributionAggregator` does along the
|
|
5256
|
+
* publisher→subscriber axis): per-client health split into sending vs receiving, plus percentile
|
|
5257
|
+
* rollups and "affected ratio" counts for the whole call.
|
|
5258
|
+
*
|
|
5259
|
+
* Build once per call and call `aggregate()` on each `call.update()`.
|
|
5260
|
+
*/
|
|
5261
|
+
declare class CallHealthAggregator {
|
|
5262
|
+
private readonly _call;
|
|
5263
|
+
readonly thresholds: ClientHealthThresholds;
|
|
5264
|
+
constructor(_call: ObservedCall, thresholds?: ClientHealthThresholds);
|
|
5265
|
+
aggregate(): CallHealth;
|
|
5266
|
+
private _clientHealth;
|
|
5267
|
+
}
|
|
5268
|
+
|
|
5269
|
+
/**
|
|
5270
|
+
* Turning a correlation into a **conclusion**.
|
|
5271
|
+
*
|
|
5272
|
+
* Every detector in this library ultimately reports the same shape of observation: *N clients have
|
|
5273
|
+
* issue X open at once, and here is what they have in common*. That is useful but not yet
|
|
5274
|
+
* actionable — someone still has to know that congestion spread across unrelated calls means the
|
|
5275
|
+
* server, while CPU limitation spread across unrelated calls means a bad client release. This module
|
|
5276
|
+
* holds that interpretation step so it is stated once, consistently, instead of being re-derived by
|
|
5277
|
+
* whoever reads the alert at 3am.
|
|
5278
|
+
*
|
|
5279
|
+
* ### Two functions, because there are two questions
|
|
5280
|
+
*
|
|
5281
|
+
* A detector already knows its scope — it was constructed with an `ObservedCall` or with the
|
|
5282
|
+
* `Observer`. Handing that scope back to a single generic function meant every caller supplied
|
|
5283
|
+
* fields the other scope needed and its own scope ignored: a call-scoped detector passing
|
|
5284
|
+
* `affectedCalls: 1, totalCalls: 1` forever, an observer-scoped one passing a participant ratio that
|
|
5285
|
+
* was deliberately never read. Placeholders like that are a standing invitation to read them as if
|
|
5286
|
+
* they meant something.
|
|
5287
|
+
*
|
|
5288
|
+
* So there are two entry points, each taking only the facts its scope actually has:
|
|
5289
|
+
*
|
|
5290
|
+
* - {@link concludeCallIssue} — within one call. The axis is *how much of the meeting*, and whether
|
|
5291
|
+
* the affected clients all subscribe to one published track.
|
|
5292
|
+
* - {@link concludeObserverIssue} — across calls. The axis is *how many independent calls*, which is
|
|
5293
|
+
* the only thing that separates "one bad room" from "our infrastructure".
|
|
5294
|
+
*
|
|
5295
|
+
* Neither the issue family nor the spread concludes anything alone: congestion in one call is a
|
|
5296
|
+
* meeting problem, congestion in six calls is an infrastructure problem, and the issue type is
|
|
5297
|
+
* identical in both.
|
|
5298
|
+
*/
|
|
5299
|
+
/** Where the fault most likely sits, given who is affected. */
|
|
5300
|
+
type IssueFaultDomain =
|
|
5301
|
+
/** Independent calls affected at once — they share only the servers and the network. */
|
|
5302
|
+
'infrastructure'
|
|
5303
|
+
/** One call, broadly affected — something that call shares (its SFU worker, room, or host). */
|
|
5304
|
+
| 'call'
|
|
5305
|
+
/** The subscribers of one published track — the publisher's path or the forwarding of it. */
|
|
5306
|
+
| 'published-track'
|
|
5307
|
+
/** A single endpoint — its own device or last mile. */
|
|
5308
|
+
| 'endpoint'
|
|
5309
|
+
/** Independent calls affected, but by something endpoints own — a client build, not a server. */
|
|
5310
|
+
| 'client-population'
|
|
5311
|
+
/** Not enough signal to attribute. */
|
|
5312
|
+
| 'unknown';
|
|
5313
|
+
/** A stated verdict, attached to the raised issue payload. */
|
|
5314
|
+
type IssueConclusion = {
|
|
5315
|
+
/** Where to look. */
|
|
5316
|
+
faultDomain: IssueFaultDomain;
|
|
5317
|
+
/** One line, written to be readable in an alert without opening a dashboard. */
|
|
5318
|
+
summary: string;
|
|
5319
|
+
/** What to check first. Omitted when the issue family is unknown to this module. */
|
|
5320
|
+
recommendation?: string;
|
|
5321
|
+
/**
|
|
5322
|
+
* How much the spread alone justifies the verdict, `0..1`. Not a probability — a coarse ranking
|
|
5323
|
+
* so alerting can threshold on it. More independent calls, or a tighter onset, means higher.
|
|
5324
|
+
*/
|
|
5325
|
+
confidence: number;
|
|
5326
|
+
};
|
|
5327
|
+
/** The facts a **call-scoped** conclusion is drawn from. */
|
|
5328
|
+
type CallIssueSpread = {
|
|
5329
|
+
issueType: string;
|
|
5330
|
+
/** Distinct clients of this call with the issue open. */
|
|
5331
|
+
affectedClients: number;
|
|
5332
|
+
/** Participants in the call — the denominator. */
|
|
5333
|
+
totalClients: number;
|
|
5334
|
+
/** True when the onsets clustered — a shared trigger rather than drift. */
|
|
5335
|
+
onsetBurst: boolean;
|
|
5336
|
+
/**
|
|
5337
|
+
* Set when the affected clients are the subscriber set of **one published track**.
|
|
5338
|
+
*
|
|
5339
|
+
* The strongest call-scoped statement available: those clients share a publisher and nothing
|
|
5340
|
+
* else, so the receivers are exonerated and the source's path is implicated.
|
|
5341
|
+
*/
|
|
5342
|
+
publishedTrackId?: string;
|
|
5343
|
+
};
|
|
5344
|
+
/** The facts an **observer-scoped** conclusion is drawn from. */
|
|
5345
|
+
type ObserverIssueSpread = {
|
|
5346
|
+
issueType: string;
|
|
5347
|
+
/** Distinct clients across the fleet with the issue open. */
|
|
5348
|
+
affectedClients: number;
|
|
5349
|
+
/** Clients in the fleet. Reported for context; it does not gate anything at this scope. */
|
|
5350
|
+
totalClients: number;
|
|
5351
|
+
/** Distinct calls containing at least one affected client. The dimension that matters here. */
|
|
5352
|
+
affectedCalls: number;
|
|
5353
|
+
/** Calls in flight. */
|
|
5354
|
+
totalCalls: number;
|
|
5355
|
+
/** True when the onsets clustered. */
|
|
5356
|
+
onsetBurst: boolean;
|
|
5357
|
+
};
|
|
5358
|
+
/**
|
|
5359
|
+
* Draw the conclusion for a group of clients **within one call**.
|
|
5360
|
+
*
|
|
5361
|
+
* Ordered most-to-least specific: a track-scoped group is a stronger statement than a call-wide one,
|
|
5362
|
+
* and a single affected endpoint is not a statement about the call at all.
|
|
5363
|
+
*/
|
|
5364
|
+
declare function concludeCallIssue(spread: CallIssueSpread): IssueConclusion;
|
|
5365
|
+
/**
|
|
5366
|
+
* Draw the conclusion for a group of clients spanning **several calls**.
|
|
5367
|
+
*
|
|
5368
|
+
* One affected call is not an observer-scoped finding — it has an obvious local explanation and the
|
|
5369
|
+
* call-scoped detector has already reported it — so that case returns `call` and says so rather than
|
|
5370
|
+
* dressing it up as a fleet event.
|
|
5371
|
+
*
|
|
5372
|
+
* Which domain breadth implicates depends on the family, and this is the whole reason the module
|
|
5373
|
+
* exists: `congestion` across unrelated calls points at the servers, `cpulimitation` across unrelated
|
|
5374
|
+
* calls points at what those *endpoints* share — a client release, a browser version, shared
|
|
5375
|
+
* virtualised hardware — and pointing an SFU team at the second one wastes a night.
|
|
5376
|
+
*/
|
|
5377
|
+
declare function concludeObserverIssue(spread: ObserverIssueSpread): IssueConclusion;
|
|
5378
|
+
|
|
5379
|
+
/** An entry retained by {@link SlidingWindow}. */
|
|
5380
|
+
type SlidingWindowEntry<T> = {
|
|
5381
|
+
timestamp: number;
|
|
5382
|
+
value: T;
|
|
5383
|
+
};
|
|
5384
|
+
/**
|
|
5385
|
+
* A time-bounded buffer used by detectors that reason over a window ("N of M clients degraded within
|
|
5386
|
+
* 10 s"). Entries older than `windowMs` are evicted on write and on read.
|
|
5387
|
+
*
|
|
5388
|
+
* ### Ordering is enforced, not assumed
|
|
5389
|
+
*
|
|
5390
|
+
* Eviction walks from the front and stops at the first entry still inside the window, which is only
|
|
5391
|
+
* correct if entries are ordered by timestamp. Callers mostly pass `Date.now()` and are ordered by
|
|
5392
|
+
* construction — but not always: a timestamp taken from a client sample, or two calls inside the
|
|
5393
|
+
* same millisecond, can arrive out of order, and one such entry would park itself at the head and
|
|
5394
|
+
* stop eviction *permanently*, so the window would grow without bound and keep reporting symptoms
|
|
5395
|
+
* from hours ago.
|
|
5396
|
+
*
|
|
5397
|
+
* Rather than trust the caller, {@link add} inserts in timestamp order. Appending (the overwhelmingly
|
|
5398
|
+
* common case) stays O(1); an out-of-order insert costs a short backward scan, because such entries
|
|
5399
|
+
* are near the tail in practice.
|
|
5400
|
+
*
|
|
5401
|
+
* ### The window advances on the newest observation
|
|
5402
|
+
*
|
|
5403
|
+
* Eviction is relative to the largest timestamp seen, not the one just passed. A caller that reads
|
|
5404
|
+
* with a `now` behind the newest entry (a replayed sample, a clock that stepped back) would
|
|
5405
|
+
* otherwise un-evict nothing and, worse, a caller passing an old `now` to {@link add} would evict
|
|
5406
|
+
* everything newer.
|
|
5407
|
+
*/
|
|
5408
|
+
declare class SlidingWindow<T> {
|
|
5409
|
+
readonly windowMs: number;
|
|
5410
|
+
/** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
|
|
5411
|
+
readonly maxEntries: number;
|
|
5412
|
+
private _entries;
|
|
5413
|
+
private _latest;
|
|
5414
|
+
constructor(windowMs: number,
|
|
5415
|
+
/** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
|
|
5416
|
+
maxEntries?: number);
|
|
5417
|
+
/** Add an entry (defaults to `Date.now()`), then evict anything outside the window. */
|
|
5418
|
+
add(value: T, timestamp?: number): void;
|
|
5419
|
+
/**
|
|
5420
|
+
* The entries still inside the window, oldest first.
|
|
5421
|
+
*
|
|
5422
|
+
* A **copy** — callers routinely map/sort what they get back, and handing out the live array let
|
|
5423
|
+
* them mutate the window from the outside.
|
|
5424
|
+
*/
|
|
5425
|
+
entries(now?: number): SlidingWindowEntry<T>[];
|
|
5426
|
+
/** The values still inside the window, oldest first. */
|
|
5427
|
+
values(now?: number): T[];
|
|
5428
|
+
/**
|
|
5429
|
+
* How many entries are inside the window as of `now`.
|
|
5430
|
+
*
|
|
5431
|
+
* Prefer this to `values(now).length`: counting through {@link values} allocates an array of every
|
|
5432
|
+
* entry only to read its length, which on a hot path is the whole cost of the call.
|
|
5433
|
+
*/
|
|
5434
|
+
count(now?: number): number;
|
|
5435
|
+
/** Retained entries, without evicting first. See {@link count} for the windowed answer. */
|
|
5436
|
+
get size(): number;
|
|
5437
|
+
clear(): void;
|
|
5438
|
+
private _evict;
|
|
5439
|
+
}
|
|
5440
|
+
|
|
5441
|
+
type TrendTesterConfig = {
|
|
5442
|
+
/**
|
|
5443
|
+
* How many of the most recent values to keep. The one knob that controls how far back "trend"
|
|
5444
|
+
* looks, for both tests — there is deliberately no separate window per test.
|
|
5445
|
+
*/
|
|
5446
|
+
size?: number;
|
|
5447
|
+
/** Page-Hinkley's drift tolerance and detection threshold — see {@link pageHinkley}. */
|
|
5448
|
+
pageHinkleyDelta?: number;
|
|
5449
|
+
pageHinkleyLambda?: number;
|
|
5450
|
+
/** Mann-Kendall's significance level — see {@link mannKendallVerdict}. */
|
|
5451
|
+
mannKendallAlpha?: number;
|
|
5452
|
+
};
|
|
5453
|
+
/**
|
|
5454
|
+
* Streaming home for `stats.ts`'s two trend tests: feed it one value at a time via {@link add}
|
|
5455
|
+
* instead of re-running the batch functions over an array you manage yourself.
|
|
5456
|
+
*
|
|
5457
|
+
* ### The two tests answer different questions
|
|
5458
|
+
*
|
|
5459
|
+
* Take a client's RTT, sampled every couple of seconds. Two things can go wrong with it, and only
|
|
5460
|
+
* one of them looks like a spike:
|
|
5461
|
+
*
|
|
5462
|
+
* - **Mann-Kendall** asks *"is this drifting?"* — a monotonic trend, regardless of shape or scale.
|
|
5463
|
+
* `40, 45, 52, 61, 70, 84 ms` is a rising path with no single dramatic step; every jump is small
|
|
5464
|
+
* and plausible on its own. Mann-Kendall counts how many later samples exceed earlier ones and
|
|
5465
|
+
* reports whether that lopsidedness could plausibly be chance. It is rank-based, so one absurd
|
|
5466
|
+
* reading (a 4000 ms outlier from a stalled event loop) moves it by exactly one pair, not by the
|
|
5467
|
+
* 4000.
|
|
5468
|
+
* - **Page-Hinkley** asks *"did it change, and when?"* — a step. `40, 42, 39, 41, 180, 176, 182 ms`
|
|
5469
|
+
* is not a trend at all; it is one level followed by a different level, which is what a route
|
|
5470
|
+
* change or a TURN failover looks like. It accumulates the deviation from the running mean and
|
|
5471
|
+
* fires when the cumulative excess passes `lambda`.
|
|
5472
|
+
*
|
|
5473
|
+
* Neither subsumes the other, which is why both live here on one window. A slow climb toward
|
|
5474
|
+
* unusability shows up in Mann-Kendall and never trips Page-Hinkley; a hard failover trips
|
|
5475
|
+
* Page-Hinkley immediately while Mann-Kendall may read `no-trend`, because after the step the series
|
|
5476
|
+
* is flat again. `tests/trendTester.spec.ts` builds both RTT series and shows exactly this.
|
|
5477
|
+
*
|
|
5478
|
+
* ```ts
|
|
5479
|
+
* const rtt = new TrendTester({ size: 30, mannKendallAlpha: 0.05, pageHinkleyLambda: 50 });
|
|
5480
|
+
*
|
|
5481
|
+
* peerConnection.on('update', () => {
|
|
5482
|
+
* if (peerConnection.currentRttInMs === undefined) return; // no measurement is not a measurement
|
|
5483
|
+
* rtt.add(peerConnection.currentRttInMs);
|
|
5484
|
+
*
|
|
5485
|
+
* if (rtt.mannKendall().trend === 'increasing') warn('RTT is drifting up');
|
|
5486
|
+
* if (rtt.pageHinkley()?.changeDetected) {
|
|
5487
|
+
* warn('RTT stepped');
|
|
5488
|
+
* rtt.clear(); // the old level is no longer the baseline — judge the new one on its own
|
|
5489
|
+
* }
|
|
5490
|
+
* });
|
|
5491
|
+
* ```
|
|
5492
|
+
*
|
|
5493
|
+
* ### Both read the same window
|
|
5494
|
+
*
|
|
5495
|
+
* `size` is the one knob controlling how far back either test looks. They are kept incremental
|
|
5496
|
+
* differently, because they don't tolerate an evicted point the same way:
|
|
5497
|
+
*
|
|
5498
|
+
* - **Mann-Kendall**'s statistic is a sum over *pairs*, so evicting the oldest value only touches
|
|
5499
|
+
* the pairs it was part of — one pass over the (bounded) window corrects it in O(size) instead of
|
|
5500
|
+
* the O(size²) a full recompute costs.
|
|
5501
|
+
* - **Page-Hinkley**'s statistic is a running minimum of a cumulative sum, which has no cheap
|
|
5502
|
+
* correction for "forget this one old point" — the minimum may have depended on it. It is
|
|
5503
|
+
* recomputed from the window on every {@link add} rather than hand-rolling an incremental version
|
|
5504
|
+
* that would be easy to get subtly wrong. That recompute is O(size), the same order as above.
|
|
5505
|
+
*
|
|
5506
|
+
* ### Non-finite input is rejected, not absorbed
|
|
5507
|
+
*
|
|
5508
|
+
* See {@link add}. A single `NaN` would otherwise destroy the instance permanently.
|
|
5509
|
+
*/
|
|
5510
|
+
declare class TrendTester {
|
|
5511
|
+
private readonly _size;
|
|
5512
|
+
private readonly _values;
|
|
5513
|
+
private readonly _tieCounts;
|
|
5514
|
+
private readonly _pageHinkleyDelta;
|
|
5515
|
+
private readonly _pageHinkleyLambda;
|
|
5516
|
+
private readonly _mannKendallAlpha;
|
|
5517
|
+
private _s;
|
|
5518
|
+
private _rejected;
|
|
5519
|
+
private _pageHinkleyResult?;
|
|
5520
|
+
constructor(config?: TrendTesterConfig);
|
|
5521
|
+
/** Values rejected by {@link add} for being non-finite. Non-zero means the caller has a bug. */
|
|
5522
|
+
get rejected(): number;
|
|
5523
|
+
/** How many values are currently in the window (`<= size`). */
|
|
5524
|
+
get length(): number;
|
|
5525
|
+
/** The configured window length. */
|
|
5526
|
+
get size(): number;
|
|
5527
|
+
/**
|
|
5528
|
+
* Add the next value in the stream, evicting the oldest once the window is full.
|
|
5529
|
+
*
|
|
5530
|
+
* **Non-finite values are rejected** rather than stored, and the rejection is counted in
|
|
5531
|
+
* {@link rejected}. This is not defensive noise — it is the difference between a bad reading and
|
|
5532
|
+
* a bad instance. `Math.sign(NaN)` is `NaN`, so a single `NaN` would poison the incremental
|
|
5533
|
+
* Mann-Kendall sum `_s` **permanently**: every later `add` and `_evictOldest` adds or subtracts
|
|
5534
|
+
* `NaN`, the z-score is `NaN`, every comparison against it is `false`, and the tester silently
|
|
5535
|
+
* reports `no-trend` forever after. It would also take a `NaN` key in `_tieCounts` that can never
|
|
5536
|
+
* be matched on eviction, since `NaN !== NaN`.
|
|
5537
|
+
*
|
|
5538
|
+
* `undefined` RTT (no measurement this tick) must not be coerced to `0` and passed in either —
|
|
5539
|
+
* "we didn't measure" is not "the trip took no time", and feeding zeros manufactures a downward
|
|
5540
|
+
* trend. Skip the sample instead.
|
|
5541
|
+
*/
|
|
5542
|
+
add(value: number): void;
|
|
5543
|
+
/** The current Page-Hinkley read-out over the window. `undefined` before the first value. */
|
|
5544
|
+
pageHinkley(): PageHinkleyResult | undefined;
|
|
5545
|
+
/** The current Mann-Kendall read-out over the window. */
|
|
5546
|
+
mannKendall(): MannKendallResult;
|
|
5547
|
+
/** Drop everything, e.g. after a detected change point, to start judging the trend fresh. */
|
|
5548
|
+
clear(): void;
|
|
5549
|
+
/** Remove the oldest value from the window and correct `_s` for the pairs it was part of. */
|
|
5550
|
+
private _evictOldest;
|
|
5551
|
+
private _bumpTie;
|
|
5552
|
+
}
|
|
5553
|
+
|
|
2897
5554
|
interface Logger {
|
|
2898
5555
|
trace(...args: any[]): void;
|
|
2899
5556
|
debug(...args: any[]): void;
|
|
@@ -2965,4 +5622,4 @@ declare function createInMemorySink(samples?: ClientSample[]): InMemorySink;
|
|
|
2965
5622
|
declare function createDefaultMediasoupRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
|
|
2966
5623
|
declare function createP2pRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
|
|
2967
5624
|
|
|
2968
|
-
export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type CallAppDataFactory, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type Detector, Detectors, InMemorySink, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, type Logger, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, type ObserverEventBase, type ObserverEvents, type ObserverLogger, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolvers, type SampleRejectedReason, type ScoreCalculator, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, setObserverLogger };
|
|
5625
|
+
export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, type ActiveIssueTracker, ActiveIssuesRegistry, type AvailableCallScopeDetectorsConfigs, type AvailableDetectorsConfigs, type AvailableObserverScopeDetectorsConfigs, type AvailableValidatorConfigs, CODEC_MISMATCH_ISSUE, type CallAppDataFactory, CallConcurrentIssueDetector, type CallConcurrentIssueDetectorConfig, type CallConcurrentIssueGroup, CallConcurrentIssueTypes, type CallHealth, CallHealthAggregator, type CallIssueSpread, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientPopulation, type ClientPopulationAxis, ClientPopulationIssueDetector, type ClientPopulationIssueDetectorConfig, ClientPopulationIssueTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type CodecConsistencyReportPayload, CodecConsistencyValidator, type CodecConsistencyValidatorConfig, type CodecEvidence, type CorroboratedPublisherFault, type Detector, Detectors, InMemorySink, type IssueConclusion, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, type IssueFaultDomain, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, LOWEST_COMMON_DENOMINATOR_ISSUE, type Logger, type MannKendallResult, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, ObservedClientIssueRegistry, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, ObserverConcurrentIssueDetector, type ObserverConcurrentIssueDetectorConfig, type ObserverConcurrentIssueGroup, ObserverConcurrentIssueTypes, type ObserverEventBase, type ObserverEvents, type ObserverIssue, type ObserverIssueSpread, type ObserverLogger, type PageHinkleyResult, PublisherFaultCorroborationDetector, type PublisherFaultCorroborationDetectorConfig, PublisherFaultTypes, RESOLVED_ISSUE_SUFFIX, type RemoteTrackLinkEvidence, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolverReportPayload, RemoteTrackResolverValidator, type RemoteTrackResolverValidatorConfig, type RemoteTrackResolvers, type ResolvedActiveClientIssue, type RunningValidator, type SampleRejectedReason, type ScoreCalculator, SfuCongestionDetector, type SfuCongestionDetectorBucket, type SfuCongestionDetectorConfig, type SfuCongestionDetectorEvaluation, type SfuCongestionDetectorReport, type SimulcastReceiverEvidence, type SimulcastReceiverReportPayload, SimulcastReceiverValidator, type SimulcastReceiverValidatorConfig, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrendTester, type TrendTesterConfig, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, TurnServerOutageDetector, type TurnServerOutageDetectorConfig, TurnServerOutageTypes, UNRESOLVED_TRACK_LINKS_ISSUE, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, type ValidationReport, type Validator, type ValidatorName, baseIssueType, concludeCallIssue, concludeObserverIssue, correlation, counterDelta, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultClientHealthThresholds, isClientIssueResolutionEntry, issuePayloadAsString, issuePayloadOf, mannKendall, mannKendallVerdict, median, medianAbsoluteDeviation, pageHinkley, percentile, percentileOfSorted, robustZScore, setObserverLogger, summarize };
|