@observertc/observer-js 1.0.0-beta.14 → 1.0.0-beta.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.mts CHANGED
@@ -1623,6 +1623,241 @@ declare class ObservedOutboundRtp implements OutboundRtpStats {
1623
1623
  update(stats: OutboundRtpStats): void;
1624
1624
  }
1625
1625
 
1626
+ /**
1627
+ * Small statistics helpers used by the aggregators and detectors.
1628
+ *
1629
+ * Rationale: a mean is a poor summary for call telemetry — one participant with a 1500 ms RTT
1630
+ * skews the average for nine healthy ones. Detectors should reason with medians, high percentiles
1631
+ * and "affected ratios" instead.
1632
+ */
1633
+ /** A distribution summary of a numeric sample set. */
1634
+ type StatsSummary = {
1635
+ count: number;
1636
+ min: number;
1637
+ max: number;
1638
+ mean: number;
1639
+ median: number;
1640
+ p25: number;
1641
+ p75: number;
1642
+ p95: number;
1643
+ };
1644
+ /**
1645
+ * The p-th percentile (0..1) using linear interpolation between closest ranks.
1646
+ * Returns `undefined` for an empty input.
1647
+ */
1648
+ declare function percentile(values: number[], p: number): number | undefined;
1649
+ /** The median (50th percentile). `undefined` for an empty input. */
1650
+ declare function median(values: number[]): number | undefined;
1651
+ /**
1652
+ * Median absolute deviation: `median(|value - median(values)|)`. A robust alternative to standard
1653
+ * deviation for describing how spread out `values` are — one wild outlier shifts a mean-based
1654
+ * deviation a lot, but barely moves a median. `undefined` for an empty input.
1655
+ */
1656
+ declare function medianAbsoluteDeviation(values: number[]): number | undefined;
1657
+ /**
1658
+ * A robust z-score: how many (MAD-based) standard deviations `value` sits above/below the median of
1659
+ * `baseline`.
1660
+ *
1661
+ * Uses the median and MAD instead of the mean and standard deviation so a handful of baseline
1662
+ * outliers can't inflate the "normal" spread and mask a genuine spike — the same reasoning behind
1663
+ * {@link summarize}'s percentiles applies here to a single scalar spread. `1.4826` is the constant
1664
+ * that makes MAD estimate the standard deviation of a normal distribution, so the result reads on
1665
+ * the same scale as a classic z-score.
1666
+ *
1667
+ * A classic z-score divides by zero once every baseline value is identical (`MAD === 0`). Here:
1668
+ * `value` strictly above that constant baseline reads as `Infinity` — a spike with no precedent
1669
+ * whatsoever, however small; `value` at or below it reads as `0`, indistinguishable from the (flat)
1670
+ * baseline rather than a division error.
1671
+ *
1672
+ * ### Worked example
1673
+ *
1674
+ * A baseline of "share of clients reporting congestion", one entry per 10 s bucket, on a healthy
1675
+ * fleet: `[0.02, 0.01, 0.03, 0.04, 0.02]`. Median `0.02`, MAD `0.01`, so the scale is
1676
+ * `1.4826 * 0.01 ≈ 0.0148`.
1677
+ *
1678
+ * ```ts
1679
+ * robustZScore(0.03, baseline); // ≈ 0.67 — an ordinary bucket
1680
+ * robustZScore(0.25, baseline); // ≈ 15.5 — a quarter of the fleet at once; nothing like it before
1681
+ * ```
1682
+ *
1683
+ * Now add one bad bucket to the *baseline* — `[0.02, 0.01, 0.03, 0.04, 0.02, 0.40]`. A mean/stddev
1684
+ * z-score would absorb it: the mean climbs to `0.087` and the stddev to `~0.14`, so a genuine `0.25`
1685
+ * spike scores about `1.2` and looks unremarkable. **One past incident would hide the next one.**
1686
+ * Median and MAD barely move (median `0.025`, MAD `0.01`), so `0.25` still scores `≈15.2`. That
1687
+ * resistance is the entire reason for this function.
1688
+ *
1689
+ * The `Infinity` case is not an edge case in practice — it is a fleet that has been perfectly quiet:
1690
+ *
1691
+ * ```ts
1692
+ * robustZScore(0.10, [ 0, 0, 0, 0, 0 ]); // Infinity — first congestion ever seen
1693
+ * robustZScore(0, [ 0, 0, 0, 0, 0 ]); // 0 — still nothing happening
1694
+ * ```
1695
+ *
1696
+ * `Infinity` clears any finite threshold, which is intended: "we have never seen this" *is* the
1697
+ * strongest possible statistical statement. It is also why a caller must gate on practical
1698
+ * significance too — see `SfuCongestionDetector`, which additionally requires a minimum number of
1699
+ * affected clients, so a single client on a quiet fleet cannot page anyone.
1700
+ *
1701
+ * `undefined` when `baseline` is empty — there is nothing to compare against.
1702
+ */
1703
+ declare function robustZScore(value: number, baseline: number[]): number | undefined;
1704
+ /** Summarize a numeric sample set. Returns `undefined` for an empty input. */
1705
+ declare function summarize(values: number[]): StatsSummary | undefined;
1706
+ /**
1707
+ * Linear-interpolated percentile of an **already ascending** array. The building block behind
1708
+ * {@link percentile} and {@link summarize}; exported so callers computing several percentiles over
1709
+ * the same data can sort once themselves.
1710
+ *
1711
+ * Passing an unsorted array yields a meaningless number rather than an error — sort first.
1712
+ */
1713
+ declare function percentileOfSorted(sorted: number[], p: number): number;
1714
+ /**
1715
+ * A counter-reset-safe delta: the increase of a cumulative counter between two observations.
1716
+ *
1717
+ * Returns `0` when the counter went backwards (reset / SSRC reuse) or when either side is missing.
1718
+ * NOTE the guard is `>=` on **defined** values — a previous value of `0` is a perfectly valid
1719
+ * baseline, so `0 -> 5` correctly yields `5` (using a truthiness check here silently drops the
1720
+ * first interval of every counter, which is exactly when the first loss/freeze event happens).
1721
+ */
1722
+ declare function counterDelta(previous: number | undefined, current: number | undefined): number;
1723
+ /**
1724
+ * Pearson correlation of two equal-length series, clamped to `0..1`.
1725
+ *
1726
+ * Negative or undefined relationships read as `0`, because every caller here asks "does A follow B?"
1727
+ * — an inverse relationship is not a weaker yes, it is a no.
1728
+ */
1729
+ declare function correlation(xs: number[], ys: number[]): number;
1730
+ /** Per-step result of {@link pageHinkley}. */
1731
+ type PageHinkleyResult = {
1732
+ /**
1733
+ * The Page-Hinkley statistic after each observation, same length as the input — one value per
1734
+ * `values[i]`, so `statistic[i]` is "how far the cumulative deviation has grown above its own
1735
+ * historical low, using only `values[0..i]`". Never negative (it's a gap to a *minimum*), and
1736
+ * `0` for as long as the process looks stable.
1737
+ */
1738
+ statistic: number[];
1739
+ /**
1740
+ * Index of the **first** observation whose statistic exceeded `lambda`, else `undefined`.
1741
+ *
1742
+ * This is a one-shot latch over the call's whole input: once found, later observations are not
1743
+ * re-checked, even if the statistic subsequently falls back down (e.g. after a single transient
1744
+ * spike — see the class doc example). "Is it *still* elevated right now" is a different
1745
+ * question, answered by looking at the *tail* of {@link statistic}, or — for a live stream — by
1746
+ * re-running this over a recent window each time (which is what `TrendTester` does) rather than
1747
+ * over the whole history once.
1748
+ */
1749
+ changePointIndex?: number;
1750
+ /** `true` when {@link changePointIndex} is defined. */
1751
+ changeDetected: boolean;
1752
+ };
1753
+ /**
1754
+ * Page-Hinkley test: sequential (online) detection of a sustained **increase** in the mean of
1755
+ * `values` — "has this metric settled onto a durably higher level", as opposed to "did one sample
1756
+ * come in high". A single noisy point should not read as a regression; ten points that are all a
1757
+ * bit higher than before should.
1758
+ *
1759
+ * ### How it works
1760
+ *
1761
+ * At step `i`, three numbers are tracked:
1762
+ *
1763
+ * 1. `mean` — the running average of `values[0..i]` (**not** a fixed baseline — it is recomputed
1764
+ * from everything seen so far, including `values[i]` itself, which is what makes this
1765
+ * *adaptive*: after a real shift, `mean` keeps drifting up to meet the new level, and the
1766
+ * signal below fades back out on its own rather than staying triggered forever).
1767
+ * 2. `cumulative` — running sum of `(values[i] - mean - delta)`. Subtracting `mean` centres each
1768
+ * term on "surprise relative to what we've seen so far"; subtracting `delta` on top means a
1769
+ * small positive surprise still nets to a *negative* contribution, so it doesn't accumulate.
1770
+ * 3. `runningMinimum` — the lowest `cumulative` has ever been.
1771
+ *
1772
+ * The **Page-Hinkley statistic** is `cumulative - runningMinimum`: how far the running sum has
1773
+ * climbed above its own historical floor. Pure noise pulls `cumulative` up and down around a flat
1774
+ * trend, so the gap to `runningMinimum` stays small. A sustained increase pushes `cumulative`
1775
+ * mostly one direction — up — so `runningMinimum` stops updating and the gap grows every step,
1776
+ * crossing `lambda` once the shift is large/long enough to be sure it isn't noise. That first
1777
+ * crossing is {@link PageHinkleyResult.changePointIndex}.
1778
+ *
1779
+ * ### Parameters
1780
+ *
1781
+ * - `delta` — the **drift tolerance** (in the same units as `values`, e.g. ms of RTT): how much of
1782
+ * a step-to-step increase is written off as noise rather than counted towards the cumulative sum.
1783
+ * `0` means even a razor-thin, consistent upward creep eventually accumulates enough to trigger;
1784
+ * raising it requires each observation to clear that bar above the running mean before it
1785
+ * contributes anything (see the class doc's hand-worked example — the same jump that triggers
1786
+ * with `delta: 0` is completely absorbed at `delta: 10`).
1787
+ * - `lambda` — the **detection threshold** the statistic must exceed. It is in "surprise units" (a
1788
+ * sum of deviations, not a single observation's units), so there's no shortcut for picking it
1789
+ * other than trying it against representative data — see the RTT examples in `stats.spec.ts` for
1790
+ * a worked comparison of a low vs. a high `lambda` on the same series. Raising it delays
1791
+ * detection but makes a false positive from a lucky run of noise less likely.
1792
+ *
1793
+ * O(n) — one pass, unlike {@link mannKendall}'s O(n²). To detect a **decrease** instead of an
1794
+ * increase, negate `values` before calling (or negate the result's meaning if you'd rather).
1795
+ */
1796
+ declare function pageHinkley(values: number[], delta?: number, lambda?: number): PageHinkleyResult;
1797
+ /** Result of {@link mannKendall}. */
1798
+ type MannKendallResult = {
1799
+ /** Sum of pairwise signs (`sign(values[j] - values[i])` for every `i < j`). */
1800
+ s: number;
1801
+ /** Variance of {@link s}, corrected for tied values. */
1802
+ variance: number;
1803
+ /** Standard-normal score derived from `s`. `0` when `s` is `0` — no evidence either way. */
1804
+ z: number;
1805
+ /** Two-tailed p-value for the null hypothesis "no monotonic trend". */
1806
+ pValue: number;
1807
+ /** `'increasing'` / `'decreasing'` when significant at `alpha`, else `'no-trend'`. */
1808
+ trend: 'increasing' | 'decreasing' | 'no-trend';
1809
+ };
1810
+ /**
1811
+ * Mann-Kendall trend test: a non-parametric test for a **monotonic** trend in `values`, without
1812
+ * assuming a distribution or a constant rate of change — it only asks "are later values
1813
+ * consistently larger (or smaller) than earlier ones more often than chance would allow".
1814
+ *
1815
+ * Every pair `i < j` votes `+1` (`values[j] > values[i]`), `-1` (`values[j] < values[i]`) or `0`
1816
+ * (tie); `s` is the sum of those votes. Under the null hypothesis of no trend, `s` is
1817
+ * approximately normal with mean `0` and a known variance (corrected here for tied values), which
1818
+ * turns `s` into a Z score and a two-tailed p-value. `trend` is only `'increasing'` / `'decreasing'`
1819
+ * when that p-value clears `alpha` — a handful of mostly-ascending points is exactly what noise
1820
+ * looks like half the time, and this is what keeps that from reading as a trend.
1821
+ *
1822
+ * O(n²) (every pair is compared); fine for the small, per-tick sample counts detectors work with,
1823
+ * not for large historical series.
1824
+ */
1825
+ declare function mannKendall(values: number[], alpha?: number): MannKendallResult;
1826
+ /**
1827
+ * Turns a Mann-Kendall `s` statistic and its `variance` into a Z score, p-value and verdict — the
1828
+ * back half of {@link mannKendall}, split out so an incremental caller that maintains `s` /
1829
+ * `variance` itself (e.g. over a sliding window, correcting for evicted points rather than
1830
+ * recomputing every pair from scratch) doesn't have to reimplement the normal approximation.
1831
+ */
1832
+ declare function mannKendallVerdict(s: number, variance: number, alpha?: number): Pick<MannKendallResult, 'z' | 'pValue' | 'trend'>;
1833
+
1834
+ /** The per-receiver view the aggregator builds for one subscribed track. */
1835
+ type PublishedTrackReceivingDistributionEntry = {
1836
+ numberOfReceivers: number;
1837
+ numberOfHealthyReceivers: number;
1838
+ numberOfDegradedReceivers: number;
1839
+ /** degradedReceivers / receivers (0..1); `0` when there are no receivers. */
1840
+ degradedRatio: number;
1841
+ /** Distribution summaries across receivers (undefined when no receiver reported the metric). */
1842
+ bitrate?: StatsSummary;
1843
+ fractionLost?: StatsSummary;
1844
+ jitter?: StatsSummary;
1845
+ rttInMs?: StatsSummary;
1846
+ jitterBufferDelayInMs?: StatsSummary;
1847
+ concealmentRatio?: StatsSummary;
1848
+ /** Fan-out counters: how many receivers saw the symptom, and the total across them. */
1849
+ freezes: {
1850
+ affectedReceivers: number;
1851
+ total: number;
1852
+ };
1853
+ plis: {
1854
+ affectedReceivers: number;
1855
+ total: number;
1856
+ };
1857
+ concealment: {
1858
+ affectedReceivers: number;
1859
+ };
1860
+ };
1626
1861
  declare class ObservedOutboundTrack implements OutboundTrackSample {
1627
1862
  timestamp: number;
1628
1863
  readonly id: string;
@@ -1633,18 +1868,28 @@ declare class ObservedOutboundTrack implements OutboundTrackSample {
1633
1868
  private _visited;
1634
1869
  appData?: Record<string, unknown>;
1635
1870
  readonly remoteInboundTracks: Set<ObservedInboundTrack>;
1871
+ readonly detectors: Detectors;
1872
+ receivingDistribution?: PublishedTrackReceivingDistributionEntry;
1636
1873
  readonly calculatedScore: CalculatedScore;
1637
1874
  addedAt?: number | undefined;
1638
1875
  removedAt?: number | undefined;
1639
1876
  muted?: boolean;
1640
1877
  attachments?: Record<string, unknown> | undefined;
1878
+ degradedReasons?: string[] | undefined;
1879
+ bitrate?: number | undefined;
1880
+ deltaPacketsSent?: number | undefined;
1881
+ remoteFractionLost?: number | undefined;
1882
+ remoteRttInMs?: number | undefined;
1883
+ qualityLimitationReason?: string | undefined;
1641
1884
  constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _outboundRtps?: ObservedOutboundRtp[] | undefined, _mediaSource?: ObservedMediaSource | undefined);
1642
1885
  get score(): number | undefined;
1643
1886
  get visited(): boolean;
1887
+ get degraded(): boolean;
1644
1888
  getPeerConnection(): ObservedPeerConnection;
1645
1889
  getOutboundRtps(): ObservedOutboundRtp[] | undefined;
1646
1890
  getMediaSource(): ObservedMediaSource | undefined;
1647
1891
  update(stats: OutboundTrackSample): void;
1892
+ private createReceivingDistribution;
1648
1893
  }
1649
1894
 
1650
1895
  declare class ObservedInboundTrack implements InboundTrackSample {
@@ -1662,6 +1907,8 @@ declare class ObservedInboundTrack implements InboundTrackSample {
1662
1907
  removedAt?: number | undefined;
1663
1908
  muted?: boolean;
1664
1909
  attachments?: Record<string, unknown> | undefined;
1910
+ degradationReasons: string[];
1911
+ get degraded(): boolean;
1665
1912
  constructor(timestamp: number, id: string, kind: MediaKind, _peerConnection: ObservedPeerConnection, _inboundRtp?: ObservedInboundRtp | undefined, _mediaPlayout?: ObservedMediaPlayout | undefined);
1666
1913
  get score(): number | undefined;
1667
1914
  get visited(): boolean;
@@ -1669,6 +1916,7 @@ declare class ObservedInboundTrack implements InboundTrackSample {
1669
1916
  getInboundRtp(): ObservedInboundRtp | undefined;
1670
1917
  getMediaPlayout(): ObservedMediaPlayout | undefined;
1671
1918
  update(stats: InboundTrackSample): void;
1919
+ private checkDegradation;
1672
1920
  }
1673
1921
 
1674
1922
  declare class ObservedRemoteOutboundRtp implements RemoteOutboundRtpStats {
@@ -2127,6 +2375,111 @@ type OperationSystem = {
2127
2375
  version: string;
2128
2376
  };
2129
2377
 
2378
+ /**
2379
+ * A finding raised **by the observer** — `observedCall.addIssue()` / `observer.addIssue()`, surfaced
2380
+ * on the bus as `call-issue` / `observer-issue`.
2381
+ *
2382
+ * ### Why this is not `ClientIssue`
2383
+ *
2384
+ * `ClientIssue` is a **wire** type: it arrives inside a `ClientSample`, so its `payload` has to be a
2385
+ * string. Server-raised findings were reusing it, which forced every detector to `JSON.stringify` a
2386
+ * perfectly good object on the way out and every handler to `JSON.parse` it back on the way in —
2387
+ * paying serialisation on a path where nothing is ever serialised, and losing type information in
2388
+ * both directions.
2389
+ *
2390
+ * An observer issue goes straight to an in-process event handler, so it carries the object.
2391
+ */
2392
+ type ObserverIssue = {
2393
+ /** What was found, e.g. `'CROSS_CALL_ISSUE_ONSET_BURST'`. */
2394
+ type: string;
2395
+ /** Observer clock, when the finding was raised. */
2396
+ timestamp: number;
2397
+ /**
2398
+ * The evidence behind the finding.
2399
+ *
2400
+ * Prefer an object — that is the point of this type. `string` is still accepted so an application
2401
+ * can forward a payload it already has serialised (e.g. relaying a `ClientIssue`) without a
2402
+ * pointless parse-then-restringify round trip. Use {@link issuePayloadOf} to read either form.
2403
+ */
2404
+ payload?: string | Record<string, unknown>;
2405
+ };
2406
+ /**
2407
+ * The payload of an issue as an object, parsing it only if it happens to be a string.
2408
+ *
2409
+ * Handlers shouldn't have to care which form arrived. Returns `undefined` for a missing payload or a
2410
+ * string that isn't valid JSON — reading evidence must never throw inside an issue handler.
2411
+ */
2412
+ declare function issuePayloadOf(issue: Pick<ObserverIssue, 'payload'>): Record<string, unknown> | undefined;
2413
+ /**
2414
+ * The payload as a JSON string, serialising it only if it is an object.
2415
+ *
2416
+ * For the boundaries that genuinely need text — a log line, an HTTP body, a message queue. Keep it at
2417
+ * the edge rather than in the detector, so in-process handlers never pay for it.
2418
+ */
2419
+ declare function issuePayloadAsString(issue: Pick<ObserverIssue, 'payload'>): string | undefined;
2420
+
2421
+ /**
2422
+ * What a validator concluded, once it is done.
2423
+ *
2424
+ * `S` is the validator's own payload — a discriminated union on `verdict` plus whatever evidence
2425
+ * belongs to each outcome — so a report reads in the language of the thing being checked rather than
2426
+ * in generic pass/fail.
2427
+ *
2428
+ * Note this is a plain intersection with `S`, not `{ [K in keyof S]: S[K] }`. A mapped type over a
2429
+ * union collapses to the keys its members *share* (`keyof (A | B)` is the intersection), which would
2430
+ * silently drop every per-verdict evidence field and leave only `verdict` behind.
2431
+ */
2432
+ type ValidationReport<S extends Record<string, unknown> = Record<string, unknown>> =
2433
+ /** Still running — it has not seen the conditions it needs to judge anything yet. */
2434
+ {
2435
+ ready: false;
2436
+ }
2437
+ /** Done. `verdict` is narrowed by `S` to that validator's own vocabulary. */
2438
+ | ({
2439
+ ready: true;
2440
+ verdict: string;
2441
+ decidedAt: number;
2442
+ } & S);
2443
+ /**
2444
+ * A **one-shot structural check**: something true of the deployment rather than of this moment.
2445
+ *
2446
+ * The distinction from a `Detector` is what changes over time. A detector answers "is something wrong
2447
+ * *right now*" — congestion, a dead relay, a track nobody receives — and the answer legitimately
2448
+ * differs every tick, so it runs every tick, forever. A validator answers "is this deployment built
2449
+ * correctly" — does the SFU pick layers per receiver — and that only changes when you deploy. So a
2450
+ * validator **runs until it knows, then finishes**: it calls {@link onDone} once, the observer drops
2451
+ * it, and nothing more is computed.
2452
+ *
2453
+ * To check again — after a deploy, say — start a new one with `observer.addValidator(...)`. There is
2454
+ * no revalidation timer, because a deploy, not the passage of time, is what makes a structural
2455
+ * verdict stale.
2456
+ *
2457
+ * Validators are held in `observer.validators` and driven by `observer.update()`.
2458
+ */
2459
+ interface Validator<S extends Record<string, unknown> = Record<string, unknown>> {
2460
+ readonly name: string;
2461
+ /**
2462
+ * The conclusion so far. `{ ready: false }` until {@link onDone} fires — and **that is not a
2463
+ * pass**. A validator usually needs specific conditions to occur before it can judge anything,
2464
+ * and plenty of deployments never present them.
2465
+ */
2466
+ readonly report: ValidationReport<S>;
2467
+ /** Called exactly once, when the validator finishes. The observer uses this to unregister it. */
2468
+ onDone: (report: ValidationReport<S>) => void;
2469
+ /** Gather evidence; decide if there is now enough. Called on every `observer.update()`. */
2470
+ update(): void;
2471
+ /** Give up without a verdict. Finishes with `inconclusive`, so a caller waiting on it is freed. */
2472
+ cancel: () => void;
2473
+ }
2474
+ /**
2475
+ * The part of a validator the observer needs in order to drive it.
2476
+ *
2477
+ * `Validator<S>` is invariant in `S` — `onDone` takes a `ValidationReport<S>` and `report` returns one
2478
+ * — so a `Set<Validator>` cannot hold validators with different payloads. Driving one only needs the
2479
+ * three members that don't mention `S`.
2480
+ */
2481
+ type RunningValidator = Pick<Validator, 'name' | 'update' | 'cancel'>;
2482
+
2130
2483
  /** The suffix client-monitor-js appends to the type of a resolution entry. */
2131
2484
  declare const RESOLVED_ISSUE_SUFFIX = "-resolved";
2132
2485
  /**
@@ -2147,8 +2500,7 @@ declare const RESOLVED_ISSUE_SUFFIX = "-resolved";
2147
2500
  * being real evidence of a shared cause.
2148
2501
  *
2149
2502
  * Issues still active when the client monitor closes are auto-resolved by the client, so a clean
2150
- * departure does not leak. An unclean one (crash, network death) can, which is why the registry
2151
- * also expires entries — see `IssueRegistry`.
2503
+ * departure does not leak.
2152
2504
  */
2153
2505
  type ActiveClientIssue = {
2154
2506
  /** Identity shared by the raise and its resolution. Unique per client. */
@@ -2157,6 +2509,11 @@ type ActiveClientIssue = {
2157
2509
  type: string;
2158
2510
  /** The client that reported it. */
2159
2511
  clientId: string;
2512
+ /**
2513
+ * The call that client belongs to. The dimension that separates "one bad meeting" from "our
2514
+ * infrastructure": at observer scope, clients in *different* calls share nothing but the server.
2515
+ */
2516
+ callId: string;
2160
2517
  /** When the client raised it (client clock). */
2161
2518
  raisedAt: number;
2162
2519
  /** When the observer first saw it (server clock) — skew-free, use this for cross-client timing. */
@@ -2169,7 +2526,7 @@ type ActiveClientIssue = {
2169
2526
  trackId?: string;
2170
2527
  };
2171
2528
  /** A closed interval: an {@link ActiveClientIssue} plus how it ended. */
2172
- type ResolvedClientIssue = ActiveClientIssue & {
2529
+ type ResolvedActiveClientIssue = ActiveClientIssue & {
2173
2530
  /** When the client resolved it (client clock). */
2174
2531
  resolvedAt: number;
2175
2532
  /** Observer-clock resolution time. */
@@ -2187,11 +2544,9 @@ type ResolvedClientIssue = ActiveClientIssue & {
2187
2544
  resolvedBy: 'client' | 'timeout' | 'client-closed';
2188
2545
  };
2189
2546
  /** `true` when the entry is a resolution companion rather than a raise. */
2190
- declare function isResolutionEntry(issue: ClientIssue): boolean;
2547
+ declare function isClientIssueResolutionEntry(issue: ClientIssue): boolean;
2191
2548
  /** Strip the `-resolved` suffix, so both entries of a lifecycle share one logical type. */
2192
2549
  declare function baseIssueType(type: string): string;
2193
- /** Best-effort parse of the JSON-string payload the schema carries. */
2194
- declare function parseIssuePayload(payload?: string): Record<string, unknown> | undefined;
2195
2550
 
2196
2551
  /** The lifecycle events a sink may emit (a subset of Node's writable-stream events). */
2197
2552
  type ClientSampleSinkEvents = {
@@ -2585,7 +2940,15 @@ type ObserverEvents = {
2585
2940
  }];
2586
2941
  /** An observer-scoped (cross-call / SFU-wide) finding raised by `observer.addIssue(...)`. */
2587
2942
  'observer-issue': [ObserverEventBase & {
2588
- issue: ClientIssue;
2943
+ issue: ObserverIssue;
2944
+ }];
2945
+ /**
2946
+ * A validator decided. Fires once per settle, not per tick — the point of a validator is that it
2947
+ * stops talking once it knows.
2948
+ */
2949
+ 'validation-ready': [ObserverEventBase & {
2950
+ validator: string;
2951
+ report: ValidationReport;
2589
2952
  }];
2590
2953
  'mediasoup-router-added': [ObservedMediasoupRouterScope];
2591
2954
  'mediasoup-router-removed': [ObservedMediasoupRouterScope];
@@ -2596,7 +2959,7 @@ type ObserverEvents = {
2596
2959
  'call-empty': [ObservedCallScope];
2597
2960
  'call-not-empty': [ObservedCallScope];
2598
2961
  'call-issue': [ObservedCallScope & {
2599
- issue: ClientIssue;
2962
+ issue: ObserverIssue;
2600
2963
  }];
2601
2964
  'client-added': [ObservedClientScope];
2602
2965
  'client-sink-created': [ObservedClientScope & {
@@ -2621,7 +2984,7 @@ type ObserverEvents = {
2621
2984
  * (`raisedAt` → `resolvedAt`, `durationInMs`).
2622
2985
  */
2623
2986
  'client-issue-resolved': [ObservedClientScope & {
2624
- resolvedIssue: ResolvedClientIssue;
2987
+ resolvedIssue: ResolvedActiveClientIssue;
2625
2988
  }];
2626
2989
  'client-metadata': [ObservedClientScope & {
2627
2990
  metaData: ClientMetaData;
@@ -2794,6 +3157,61 @@ type ObserverEvents = {
2794
3157
  }];
2795
3158
  };
2796
3159
 
3160
+ /**
3161
+ * Something that wants to be **handed** open client issues rather than to go looking for them.
3162
+ *
3163
+ * Register one with `activeIssuesRegistry.addIssueTracker(type, tracker)` and it receives every
3164
+ * issue of that type when it opens ({@link add}) and when it closes ({@link delete}). A detector
3165
+ * implementing this pays only for the issues it actually consumes.
3166
+ *
3167
+ * `ActiveIssuesRegistry` implements it too, which is how a call's registry feeds the observer's.
3168
+ */
3169
+ interface ActiveIssueTracker {
3170
+ /** An issue of a subscribed type opened. */
3171
+ add(issue: ActiveClientIssue): void;
3172
+ /**
3173
+ * An issue this tracker was given has closed.
3174
+ *
3175
+ * Return `true` if it was actually held. Returning `false` is legitimate and not an error — a
3176
+ * tracker that counts *occurrences* (see `SfuCongestionDetector`) deliberately ignores
3177
+ * resolutions, because when a symptom ended says nothing about how many endpoints reported it.
3178
+ */
3179
+ delete(issue: ActiveClientIssue): boolean;
3180
+ /** How many issues this tracker currently holds. */
3181
+ size: number;
3182
+ /** Drop everything. Called when the owning scope closes. */
3183
+ clear(): void;
3184
+ has(issue: ActiveClientIssue): boolean;
3185
+ }
3186
+
3187
+ /**
3188
+ * One client's currently **open** stateful issues, keyed by `ClientIssue.key`.
3189
+ *
3190
+ * The server-side mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0):
3191
+ * a raise opens an entry, the matching `<type>-resolved` closes it, and the client's own close
3192
+ * force-resolves whatever is left. That turns point-in-time symptom reports into **intervals**,
3193
+ * which is what lets detectors ask "are these clients broken *at the same time*" rather than "did
3194
+ * they both report something recently".
3195
+ *
3196
+ * Keyed by `key` rather than by type on purpose: one client can have several issues of the same type
3197
+ * open at once (one per track), and they resolve independently.
3198
+ *
3199
+ * Every change is forwarded to the call's `ActiveIssuesRegistry`, which forwards to the observer's —
3200
+ * so the client owns the storage and the wider scopes get their views maintained as it happens.
3201
+ */
3202
+ declare class ObservedClientIssueRegistry {
3203
+ private readonly registry?;
3204
+ private readonly issues;
3205
+ constructor(registry?: ActiveIssueTracker | undefined);
3206
+ get size(): number;
3207
+ keys(): IterableIterator<string>;
3208
+ values(): IterableIterator<ActiveClientIssue>;
3209
+ get(key: string): ActiveClientIssue | undefined;
3210
+ add(issue: ActiveClientIssue): this;
3211
+ remove(key: string): ActiveClientIssue | undefined;
3212
+ clear(): void;
3213
+ }
3214
+
2797
3215
  type ObservedClientSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
2798
3216
  clientId: string;
2799
3217
  appData?: AppData;
@@ -2872,7 +3290,6 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
2872
3290
  totalScoreSum: number;
2873
3291
  numberOfScoreMeasurements: number;
2874
3292
  readonly mediaDevices: MediaDeviceInfo[];
2875
- issues: ClientIssue[];
2876
3293
  /**
2877
3294
  * The client's currently **open** stateful issues, keyed by `ClientIssue.key` — the server-side
2878
3295
  * mirror of the client monitor's own active-issue map (client-monitor-js >= 4.6.0).
@@ -2882,11 +3299,11 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
2882
3299
  * recently". Entries are opened by a raise, closed by the matching `<type>-resolved` entry, and
2883
3300
  * force-closed when the client closes.
2884
3301
  */
2885
- readonly activeIssues: Map<string, ActiveClientIssue>;
3302
+ readonly activeIssues: ObservedClientIssueRegistry;
2886
3303
  private _pendingInjections;
2887
3304
  private _activeSample?;
2888
3305
  private closeTimer?;
2889
- constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall);
3306
+ constructor(settings: ObservedClientSettings<AppData>, call: ObservedCall, activeIssues: ObservedClientIssueRegistry);
2890
3307
  get numberOfPeerConnections(): number;
2891
3308
  get score(): number | undefined;
2892
3309
  close(): void;
@@ -2911,12 +3328,6 @@ declare class ObservedClient<AppData extends Record<string, unknown> = Record<st
2911
3328
  */
2912
3329
  addIssue(issue: ClientIssue): void;
2913
3330
  addExtensionStats(stats: ExtensionStat): void;
2914
- /**
2915
- * Close every still-open issue, e.g. because the client is going away. Emits
2916
- * `client-issue-resolved` for each with the given `resolvedBy`, so correlators relying on the
2917
- * active set don't keep a departed client's issues open forever.
2918
- */
2919
- resolveActiveIssues(resolvedBy?: ResolvedClientIssue['resolvedBy']): void;
2920
3331
  /** Close the active issue a `<type>-resolved` entry refers to, and announce the finished interval. */
2921
3332
  private _resolveIssue;
2922
3333
  private _processClientEvent;
@@ -2976,12 +3387,6 @@ declare class RemoteTrackResolver {
2976
3387
  private _removeOutboundTrack;
2977
3388
  }
2978
3389
 
2979
- interface Updater {
2980
- readonly name: string;
2981
- readonly description?: string;
2982
- close(): void;
2983
- }
2984
-
2985
3390
  interface Detector {
2986
3391
  readonly name: string;
2987
3392
  /** Called on every entity update; may raise issues via the entity it observes. */
@@ -2994,33 +3399,1125 @@ interface Detector {
2994
3399
  close?(): void;
2995
3400
  }
2996
3401
 
2997
- declare class Detectors {
2998
- private _detectors;
2999
- constructor(...detectors: Detector[]);
3000
- get listOfNames(): string[];
3001
- add(detector: Detector): void;
3002
- remove(detector: Detector): void;
3003
- update(): void;
3402
+ declare const CallConcurrentIssueTypes: {
3403
+ /** Several participants of this call have the same issue open **at the same time**. */
3404
+ readonly concurrentClientIssues: "CONCURRENT_CLIENT_ISSUES";
3405
+ /** Those issues also *began* together — the signature of one shared event. */
3406
+ readonly issueOnsetBurst: "ISSUE_ONSET_BURST";
3407
+ };
3408
+ type CallConcurrentIssueDetectorConfig = {
3409
+ /**
3410
+ * The issue types to watch. **Required, and must not be empty** — the detector subscribes to
3411
+ * exactly these and sees nothing else.
3412
+ */
3413
+ issueTypes: string[];
3414
+ /** Minimum participants in the call before a ratio is meaningful. Default `3`. */
3415
+ minClients: number;
3416
+ /** Minimum distinct clients sharing the open issue. Default `3`. */
3417
+ minAffectedClients: number;
3418
+ /** Fraction of the call's participants that must share it. Default `0.5`. */
3419
+ affectedRatioThreshold: number;
3420
+ /**
3421
+ * When the onsets fall within this span (ms), the finding is escalated to `ISSUE_ONSET_BURST` —
3422
+ * they didn't just overlap, they started together. Default `2_000`.
3423
+ */
3424
+ onsetBurstWindowInMs: number;
3425
+ /** Re-arm time (ms) per issue type. Default `60_000`. */
3426
+ cooldownMs: number;
3427
+ };
3428
+ /** What the detector currently knows about one issue type in this call. */
3429
+ type CallConcurrentIssueGroup = {
3430
+ type: string;
3431
+ issues: ActiveClientIssue[];
3432
+ clientIds: string[];
3433
+ affectedRatio: number;
3434
+ totalClients: number;
3435
+ /**
3436
+ * Spread of the onsets, in **observer** time (ms) — `max(observedAt) - min(observedAt)`.
3437
+ *
3438
+ * Measured on the observer clock on purpose: `raisedAt` comes from each client's own clock, and
3439
+ * comparing those across machines makes clock skew look like a shared event.
3440
+ */
3441
+ onsetSpreadInMs: number;
3442
+ firstObservedAt: number;
3443
+ };
3444
+ /**
3445
+ * Answers **"is this meeting in trouble?"** — several participants of one call with the same issue
3446
+ * open simultaneously.
3447
+ *
3448
+ * The client already decides *what* is wrong for itself — `congestion`, `ice-disconnected`,
3449
+ * `audio-concealment`, `video-decoder-overloaded` — with hysteresis and multi-signal confirmation
3450
+ * behind each verdict. Re-deriving those server-side from raw counters would be strictly worse. What
3451
+ * the server uniquely knows is *how many other participants of the same call are in that state right
3452
+ * now*, which is the difference between "one person's Wi-Fi" and "this room is broken".
3453
+ *
3454
+ * Concurrency is judged from the **open interval set**, not a window of recent reports. A window has
3455
+ * to guess whether a symptom is still happening; an interval is closed by the client when the episode
3456
+ * actually ends (client-monitor-js >= 4.6.0 ships the `<type>-resolved` companion for exactly this).
3457
+ *
3458
+ * ```ts
3459
+ * observedCall.addDetector('call-concurrent-issue-detector', {
3460
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
3461
+ * });
3462
+ * ```
3463
+ *
3464
+ * For the cross-call version of this question — which is a different question, not this one with a
3465
+ * bigger denominator — see `ObserverConcurrentIssueDetector`.
3466
+ */
3467
+ declare class CallConcurrentIssueDetector implements Detector, ActiveIssueTracker {
3468
+ private readonly _call;
3469
+ static readonly NAME: "call-concurrent-issue-detector";
3470
+ readonly name: "call-concurrent-issue-detector";
3471
+ private readonly _config;
3472
+ private readonly _lastRaisedAt;
3473
+ /** issue type -> the issues of that type currently open in this call. */
3474
+ private readonly _byType;
3475
+ private _size;
3476
+ /** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
3477
+ lastGroups: CallConcurrentIssueGroup[];
3478
+ constructor(_call: ObservedCall, config?: Partial<CallConcurrentIssueDetectorConfig>);
3479
+ get size(): number;
3480
+ has(issue: ActiveClientIssue): boolean;
3481
+ add(issue: ActiveClientIssue): void;
3482
+ delete(issue: ActiveClientIssue): boolean;
3004
3483
  clear(): void;
3005
- private _close;
3484
+ update(): void;
3485
+ close(): void;
3486
+ private _groupOf;
3006
3487
  }
3007
3488
 
3008
- type ObservedCallUpdateConfig = {
3009
- updatePolicy?: 'update-on-any-client-updated' | 'update-when-all-client-updated' | 'none';
3489
+ declare const ClientPopulationIssueTypes: {
3490
+ /** One issue type is concentrated on one client population while the rest of the fleet is fine. */
3491
+ readonly clientPopulationIssue: "CLIENT_POPULATION_ISSUE";
3010
3492
  };
3011
- type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = ObservedCallUpdateConfig & {
3012
- callId: string;
3013
- appData?: AppData;
3014
- closeCallIfEmptyForMs?: number;
3493
+ /** The client attribute to group by. One axis per detector — see the class description. */
3494
+ type ClientPopulationAxis = 'browser' | 'engine' | 'platform' | 'operationSystem';
3495
+ type ClientPopulationIssueDetectorConfig = {
3496
+ /**
3497
+ * The issue types to watch. **Required, and must not be empty.**
3498
+ *
3499
+ * The types worth grouping this way are the ones an endpoint owns: `cpulimitation`,
3500
+ * `encoder-bottleneck`, `capture-bottleneck`, `stuck-decoder`, `video-decoder-overloaded`.
3501
+ * Grouping a *network* symptom by browser is a category error — `congestion` clusters by ISP and
3502
+ * geography, neither of which this detector can see, and it would happily report a browser
3503
+ * correlation that is really a "most of our users are on Chrome" artefact.
3504
+ */
3505
+ issueTypes: string[];
3506
+ /** Which client attribute to group by. Default `'browser'`. */
3507
+ groupBy: ClientPopulationAxis;
3508
+ /**
3509
+ * Group by `name` only, or by `name + version`. Default `true` (include version).
3510
+ *
3511
+ * Version is usually the point: "Chrome" is not actionable, "Chrome 141" is, because it names a
3512
+ * thing that changed on a date. Set to `false` when comparing whole engines.
3513
+ */
3514
+ includeVersion: boolean;
3515
+ /** Minimum clients in a population before its rate means anything. Default `20`. */
3516
+ minPopulationSize: number;
3517
+ /** Minimum affected clients in the population. Default `5`. */
3518
+ minAffectedClients: number;
3519
+ /** Share of the population that must be affected. Default `0.3`. */
3520
+ affectedRatioThreshold: number;
3521
+ /**
3522
+ * How many times worse the suspect population must be than the rest of the fleet. Default `3`.
3523
+ *
3524
+ * This is the control, and it is what makes the finding mean anything. See the class description.
3525
+ */
3526
+ minRelativeRisk: number;
3527
+ /** Minimum clients **outside** the suspect population before a comparison is possible. Default `20`. */
3528
+ minControlSize: number;
3529
+ /** Re-arm time (ms) per (population, issue type). Default `300_000`. */
3530
+ cooldownMs: number;
3015
3531
  };
3016
- type ObservedCallEvents = {
3017
- update: [];
3018
- newclient: [ObservedClient];
3019
- empty: [];
3020
- 'not-empty': [];
3021
- close: [];
3532
+ /** The rollup for one population on one issue type. */
3533
+ type ClientPopulation = {
3534
+ /** e.g. `'Chrome 141'`, or `'Chrome'` when `includeVersion` is off. */
3535
+ population: string;
3536
+ axis: ClientPopulationAxis;
3537
+ issueType: string;
3538
+ clients: number;
3539
+ affectedClients: number;
3540
+ affectedRatio: number;
3541
+ affectedClientIds: string[];
3542
+ /** Everyone not in this population. */
3543
+ controlClients: number;
3544
+ controlAffectedClients: number;
3545
+ controlAffectedRatio: number;
3546
+ /** `affectedRatio / controlAffectedRatio`. `Infinity` when the control group is completely clean. */
3547
+ relativeRisk: number;
3022
3548
  };
3023
- declare interface ObservedCall {
3549
+ /**
3550
+ * Finds an issue that is concentrated on **one kind of client** — one browser, one browser version,
3551
+ * one OS — rather than on anything the servers own.
3552
+ *
3553
+ * ### Why this exists
3554
+ *
3555
+ * The other observer-scoped detectors all answer "who else has this open, and what do they share?"
3556
+ * with the answer *the infrastructure*, because clients in unrelated calls share nothing else. That
3557
+ * inference is right for network symptoms and **wrong for endpoint symptoms**, and the difference
3558
+ * matters at 3am. `cpulimitation` opening across six unrelated calls is not an SFU event: CPU is
3559
+ * owned by the endpoint, so what those endpoints have in common is a client release, a browser
3560
+ * update, or a fleet of identical VDI hosts. `IssueConclusion` already says exactly this — it maps
3561
+ * the endpoint-capacity family to a `client-population` fault domain instead of `infrastructure` —
3562
+ * but until now nothing in the library actually computed the grouping that claim refers to. This
3563
+ * detector is that computation.
3564
+ *
3565
+ * It is the one correlation in this library that is neither per-call nor per-server. A client knows
3566
+ * its own browser and nothing about anyone else's; only something sitting above the whole fleet can
3567
+ * notice that every complaint is coming from the same build.
3568
+ *
3569
+ * ### The control group is the whole point
3570
+ *
3571
+ * "30% of Chrome 141 users report encoder-bottleneck" is not a finding on its own. If 30% of
3572
+ * *everyone* reports it, Chrome 141 is not the story — you have a fleet-wide problem and this
3573
+ * detector would be pointing at the largest population rather than at a cause. Naive share-based
3574
+ * grouping always indicts whichever browser is most popular, which is why the gate here is
3575
+ * **relative risk**: the suspect population's rate divided by the rate among everyone else. A
3576
+ * population only qualifies when it is `minRelativeRisk` times worse than the rest of the fleet, and
3577
+ * only when the rest of the fleet is large enough (`minControlSize`) for "the rest of the fleet" to
3578
+ * be a real measurement.
3579
+ *
3580
+ * A completely clean control group gives `Infinity`, which is honest — nobody outside this
3581
+ * population has the problem at all — and is exactly why `minAffectedClients` and
3582
+ * `minPopulationSize` are checked independently, so a single unlucky user on a rare browser cannot
3583
+ * page anyone.
3584
+ *
3585
+ * ### One axis per detector
3586
+ *
3587
+ * `groupBy` takes a single attribute. Add a second instance if you want a second axis:
3588
+ *
3589
+ * ```ts
3590
+ * observer.addObserverDetector('client-population-issue-detector', {
3591
+ * issueTypes: [ 'cpulimitation', 'encoder-bottleneck', 'stuck-decoder' ],
3592
+ * groupBy: 'browser',
3593
+ * });
3594
+ *
3595
+ * observer.on('observer-issue', ({ issue }) => {
3596
+ * if (issue.type !== ClientPopulationIssueTypes.clientPopulationIssue) return;
3597
+ * // → { population: 'Chrome 141', issueType: 'encoder-bottleneck',
3598
+ * // affectedRatio: 0.34, controlAffectedRatio: 0.02, relativeRisk: 17 }
3599
+ * });
3600
+ * ```
3601
+ *
3602
+ * Deliberately not a cross-product of every axis at once: an issue that clusters on macOS *and* on
3603
+ * Safari is usually one fact reported twice, and a detector that emits both leaves the reader to
3604
+ * work out which one is causal. Pick the axis you want to reason about.
3605
+ *
3606
+ * ### Clients that never reported their metadata
3607
+ *
3608
+ * `browser` / `engine` / `platform` / `operationSystem` arrive as client metadata and may be absent —
3609
+ * a client that closed before sending them, or an application that does not collect them. Those
3610
+ * clients are excluded from **both** the population and the control group rather than bucketed as
3611
+ * `'unknown'`. A synthetic `'unknown'` population would be a mixture of every real one, so any rate
3612
+ * computed for it means nothing, and leaving those clients in the control group would dilute the
3613
+ * comparison with clients whose kind we cannot verify.
3614
+ */
3615
+ declare class ClientPopulationIssueDetector implements Detector, ActiveIssueTracker {
3616
+ private readonly _observer;
3617
+ static readonly NAME: "client-population-issue-detector";
3618
+ readonly name: "client-population-issue-detector";
3619
+ private readonly _config;
3620
+ private readonly _lastRaisedAt;
3621
+ private readonly _issues;
3622
+ /** The populations that qualified on the most recent `update()`. Exposed for tests/dashboards. */
3623
+ lastPopulations: ClientPopulation[];
3624
+ constructor(_observer: Observer, config?: Partial<ClientPopulationIssueDetectorConfig>);
3625
+ get size(): number;
3626
+ has(issue: ActiveClientIssue): boolean;
3627
+ add(issue: ActiveClientIssue): void;
3628
+ delete(issue: ActiveClientIssue): boolean;
3629
+ clear(): void;
3630
+ update(): void;
3631
+ close(): void;
3632
+ /** `undefined` when the client never reported this attribute — see the class description. */
3633
+ private _populationOf;
3634
+ private _rollupOf;
3635
+ private _riskText;
3636
+ /** Bigger populations and starker contrasts are harder to produce by chance. */
3637
+ private _confidenceOf;
3638
+ }
3639
+
3640
+ declare const PublisherFaultTypes: {
3641
+ /**
3642
+ * A publisher is reporting trouble on its own send path **while** its subscribers report trouble
3643
+ * receiving it. Both ends agree, so the source is implicated rather than inferred.
3644
+ */
3645
+ readonly corroboratedPublisherFault: "CORROBORATED_PUBLISHER_FAULT";
3646
+ };
3647
+ type PublisherFaultCorroborationDetectorConfig = {
3648
+ /**
3649
+ * Issue types raised by the **publishing** client about its own outbound path. **Required.**
3650
+ *
3651
+ * The natural set from client-monitor-js: `encoder-bottleneck`, `capture-bottleneck`,
3652
+ * `dry-outbound-track`. All three mean "I am failing to produce or send this properly", which is
3653
+ * the half of the story the receivers cannot see.
3654
+ */
3655
+ publisherIssueTypes: string[];
3656
+ /**
3657
+ * Issue types raised by the **subscribing** clients about the track they receive. **Required.**
3658
+ *
3659
+ * The natural set: `freezed-video-track`, `dry-inbound-track`, `video-recovery-failed`. These say
3660
+ * "I am not getting this properly", which is the half the publisher cannot see.
3661
+ */
3662
+ receiverIssueTypes: string[];
3663
+ /** Minimum subscribers of the track that must be complaining. Default `2`. */
3664
+ minAffectedReceivers: number;
3665
+ /** Re-arm time (ms) per published track. Default `60_000`. */
3666
+ cooldownMs: number;
3667
+ };
3668
+ /** The two-sided evidence behind one finding. */
3669
+ type CorroboratedPublisherFault = {
3670
+ trackId: string;
3671
+ kind: string;
3672
+ publisherClientId: string;
3673
+ /** The publisher's own open issue types on this track. */
3674
+ publisherIssueTypes: string[];
3675
+ /** The receiver-side open issue types across this track's subscribers. */
3676
+ receiverIssueTypes: string[];
3677
+ receivers: number;
3678
+ affectedReceivers: number;
3679
+ affectedClientIds: string[];
3680
+ publisherBitrate?: number;
3681
+ };
3682
+ /**
3683
+ * Fires only when **both ends of one published track are complaining at the same time**: the
3684
+ * publisher about its own send path, and its subscribers about receiving it.
3685
+ *
3686
+ * ### How this differs from `IssueFanOutDetector`
3687
+ *
3688
+ * Fan-out sees one end. It observes that most of Alice's subscribers are unhappy and *infers* that
3689
+ * the fault is on Alice's side, because the affected clients share a publisher and nothing else. That
3690
+ * inference is sound, and it is still a inference: the same observation is produced by the SFU
3691
+ * mangling Alice's stream on the way out, with Alice herself perfectly healthy.
3692
+ *
3693
+ * This detector removes the inference. When Alice reports `encoder-bottleneck` *and* four of her six
3694
+ * subscribers report `freezed-video-track` in the same window, there is nothing left to deduce — the
3695
+ * source said it was struggling and the receivers confirmed the consequence. That is the strongest
3696
+ * statement this library can make about where a fault sits, and it is only available to something
3697
+ * holding both ends at once. Neither the publisher nor any receiver can reach this conclusion alone.
3698
+ *
3699
+ * Run both: fan-out is broader and catches the SFU-forwarding case where the publisher is fine;
3700
+ * this one is narrower and, when it fires, needs no interpretation.
3701
+ *
3702
+ * ### Silence here is not health
3703
+ *
3704
+ * A quiet detector means only that the two halves have not coincided — most commonly because the
3705
+ * publisher is genuinely fine and the fault is in forwarding, which is exactly the case `fan-out`
3706
+ * exists to report. Do not read "no corroborated fault" as "no publisher-side problem".
3707
+ *
3708
+ * ```ts
3709
+ * observedCall.addDetector('publisher-fault-corroboration-detector', {
3710
+ * publisherIssueTypes: [ 'encoder-bottleneck', 'capture-bottleneck', 'dry-outbound-track' ],
3711
+ * receiverIssueTypes: [ 'freezed-video-track', 'dry-inbound-track' ],
3712
+ * });
3713
+ * ```
3714
+ *
3715
+ * ### Requires a `RemoteTrackResolver`
3716
+ *
3717
+ * Matching a publisher's issue to its subscribers' issues needs the publisher↔subscriber links. With
3718
+ * no resolver the detector does nothing rather than guessing.
3719
+ */
3720
+ declare class PublisherFaultCorroborationDetector implements Detector, ActiveIssueTracker {
3721
+ private readonly _call;
3722
+ static readonly NAME: "publisher-fault-corroboration-detector";
3723
+ readonly name: "publisher-fault-corroboration-detector";
3724
+ private readonly _config;
3725
+ private readonly _lastRaisedAt;
3726
+ /** Open publisher-side issues that name a track. */
3727
+ private readonly _publisherIssues;
3728
+ /** Open receiver-side issues that name a track. */
3729
+ private readonly _receiverIssues;
3730
+ /** The faults corroborated on the most recent `update()`. Exposed for tests/dashboards. */
3731
+ lastFaults: CorroboratedPublisherFault[];
3732
+ constructor(_call: ObservedCall, config?: Partial<PublisherFaultCorroborationDetectorConfig>);
3733
+ get size(): number;
3734
+ has(issue: ActiveClientIssue): boolean;
3735
+ add(issue: ActiveClientIssue): void;
3736
+ delete(issue: ActiveClientIssue): boolean;
3737
+ clear(): void;
3738
+ update(): void;
3739
+ close(): void;
3740
+ /**
3741
+ * Resolve a publisher-side issue's `trackId` to the outbound track it is about.
3742
+ *
3743
+ * Looked up through the reporting client's own peer connections: the issue names its client, so
3744
+ * the search is bounded by that client's transports rather than by the size of the call.
3745
+ */
3746
+ private _outboundTrackOf;
3747
+ }
3748
+
3749
+ declare const ObserverConcurrentIssueTypes: {
3750
+ /**
3751
+ * The same issue is open in **several unrelated calls at once**. Those clients share no meeting,
3752
+ * no publisher and no room — only the infrastructure serving them.
3753
+ */
3754
+ readonly crossCallConcurrentIssues: "CROSS_CALL_CONCURRENT_ISSUES";
3755
+ /** The cross-call version that also started together. The strongest "it's us" signal available. */
3756
+ readonly crossCallIssueOnsetBurst: "CROSS_CALL_ISSUE_ONSET_BURST";
3757
+ };
3758
+ type ObserverConcurrentIssueDetectorConfig = {
3759
+ /**
3760
+ * The issue types to watch. **Required, and must not be empty** — the detector subscribes to
3761
+ * exactly these and sees nothing else.
3762
+ */
3763
+ issueTypes: string[];
3764
+ /** Minimum distinct clients sharing the open issue. Default `3`. */
3765
+ minAffectedClients: number;
3766
+ /**
3767
+ * Minimum number of *distinct calls* the affected clients must span. Default `2`.
3768
+ *
3769
+ * This is what makes an observer-scoped finding mean something a call-scoped one doesn't. Without
3770
+ * it, one thirty-person meeting with congestion satisfies every client-count threshold and raises
3771
+ * a fleet-wide alert for what is really one bad room — which `CallConcurrentIssueDetector` has
3772
+ * already reported. Requiring two or more independent calls is the difference between a
3773
+ * coincidence and a shared cause.
3774
+ */
3775
+ minAffectedCalls: number;
3776
+ /**
3777
+ * Fraction of the calls in flight that must be affected. Default `0` — off, because absolute
3778
+ * counts matter more than ratios here: three broken calls out of a thousand is still worth
3779
+ * knowing about, and a ratio threshold would hide it. Raise it if you only care about fleet-wide
3780
+ * events.
3781
+ *
3782
+ * Note there is deliberately **no participant ratio** at this scope. Six broken calls out of forty
3783
+ * is a handful of clients against the whole fleet, so any meaningful client ratio would suppress
3784
+ * exactly the finding this detector exists to produce.
3785
+ */
3786
+ affectedCallRatioThreshold: number;
3787
+ /**
3788
+ * When the onsets fall within this span (ms), the finding is escalated to
3789
+ * `CROSS_CALL_ISSUE_ONSET_BURST`. Default `2_000`.
3790
+ */
3791
+ onsetBurstWindowInMs: number;
3792
+ /** Re-arm time (ms) per issue type. Default `60_000`. */
3793
+ cooldownMs: number;
3794
+ };
3795
+ /** What the detector currently knows about one issue type across the fleet. */
3796
+ type ObserverConcurrentIssueGroup = {
3797
+ type: string;
3798
+ issues: ActiveClientIssue[];
3799
+ clientIds: string[];
3800
+ totalClients: number;
3801
+ affectedRatio: number;
3802
+ callIds: string[];
3803
+ totalCalls: number;
3804
+ affectedCallRatio: number;
3805
+ /** Per-call breakdown, largest first — the first question anyone asks is "which calls, how badly?". */
3806
+ perCall: {
3807
+ callId: string;
3808
+ affectedClients: number;
3809
+ totalClients: number;
3810
+ }[];
3811
+ /**
3812
+ * Spread of the onsets, in **observer** time (ms). Client clocks are never compared across
3813
+ * machines here — skew between them would masquerade as a synchronized event.
3814
+ */
3815
+ onsetSpreadInMs: number;
3816
+ firstObservedAt: number;
3817
+ };
3818
+ /**
3819
+ * Answers **"is our infrastructure in trouble?"** — the same issue open across several *unrelated*
3820
+ * calls at the same moment.
3821
+ *
3822
+ * This is not the call-scoped question with a bigger denominator, which is why it is a separate
3823
+ * detector with separate gates and its own finding types. Participant count alone is a bad fleet
3824
+ * signal: one thirty-person meeting where everybody is congested clears every client threshold, yet
3825
+ * it has an obvious local explanation. Clients in *different* calls share no room, no publisher and
3826
+ * no host — only the servers and the network. When the same issue opens across several of them at
3827
+ * once, the infrastructure is the only remaining common factor, and that is the finding worth paging
3828
+ * someone about.
3829
+ *
3830
+ * ```ts
3831
+ * observer.addObserverDetector('observer-concurrent-issue-detector', {
3832
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
3833
+ * minAffectedCalls: 3,
3834
+ * });
3835
+ *
3836
+ * observer.on('observer-issue', ({ issue }) => {
3837
+ * if (issue.type === ObserverConcurrentIssueTypes.crossCallIssueOnsetBurst) page(issue);
3838
+ * });
3839
+ * // → CROSS_CALL_ISSUE_ONSET_BURST { issueType: 'congestion', calls: 40, affectedCalls: 6, … }
3840
+ * ```
3841
+ *
3842
+ * Onsets are compared on the **observer clock** (`observedAt`), never on client clocks: participants
3843
+ * degrading together within a couple of seconds is far more likely to be a deploy, a TURN failover or
3844
+ * a link flap than a coincidence — but only if the timestamps being compared came from one clock.
3845
+ */
3846
+ declare class ObserverConcurrentIssueDetector implements Detector, ActiveIssueTracker {
3847
+ private readonly _observer;
3848
+ static readonly NAME: "observer-concurrent-issue-detector";
3849
+ readonly name: "observer-concurrent-issue-detector";
3850
+ private readonly _config;
3851
+ private readonly _lastRaisedAt;
3852
+ /** issue type -> the issues of that type currently open anywhere in the fleet. */
3853
+ private readonly _byType;
3854
+ private _size;
3855
+ /** The groups that qualified on the most recent `update()`. Exposed for tests/dashboards. */
3856
+ lastGroups: ObserverConcurrentIssueGroup[];
3857
+ constructor(_observer: Observer, config?: Partial<ObserverConcurrentIssueDetectorConfig>);
3858
+ get size(): number;
3859
+ has(issue: ActiveClientIssue): boolean;
3860
+ add(issue: ActiveClientIssue): void;
3861
+ delete(issue: ActiveClientIssue): boolean;
3862
+ clear(): void;
3863
+ update(): void;
3864
+ close(): void;
3865
+ private _groupOf;
3866
+ }
3867
+
3868
+ declare const IssueFanOutTypes: {
3869
+ /** Most receivers of one published track have the same issue open → the fault follows the source. */
3870
+ readonly publishedTrackIssueFanOut: "PUBLISHED_TRACK_ISSUE_FAN_OUT";
3871
+ /** Exactly one receiver of a track has it → that receiver's own problem. */
3872
+ readonly singleReceiverIssue: "SINGLE_RECEIVER_ISSUE";
3873
+ };
3874
+ type IssueFanOutDetectorConfig = {
3875
+ /**
3876
+ * The receiver-side issue types to attribute to publishers. **Required, and must not be empty** —
3877
+ * the detector subscribes to exactly these.
3878
+ *
3879
+ * There is no "all types" option. Which of a receiver's complaints are worth blaming a publisher
3880
+ * for is application knowledge: `freezed-video-track` fanning out across a track's subscribers
3881
+ * implicates the source, `cpulimitation` fanning out the same way implicates the receivers'
3882
+ * hardware and would be a false accusation.
3883
+ */
3884
+ issueTypes: string[];
3885
+ /** Minimum receivers of the track before a ratio is meaningful. Default `3`. */
3886
+ minReceivers: number;
3887
+ /** Fraction of a track's receivers that must share the issue. Default `0.6`. */
3888
+ affectedRatioThreshold: number;
3889
+ /** Also report the "only one receiver is affected" case. Default `true`. */
3890
+ reportSingleReceiver: boolean;
3891
+ /** Re-arm time (ms) per (track, issue type). Default `60_000`. */
3892
+ cooldownMs: number;
3893
+ };
3894
+ /**
3895
+ * Attributes **client-reported issues to the published track they are about**, then asks how far
3896
+ * the problem fans out across that track's receivers.
3897
+ *
3898
+ * The join is what makes this possible: a receiver-side issue payload carries `trackId` (the client
3899
+ * detectors report it for every track-scoped issue), the observer resolves that to an inbound track,
3900
+ * and `RemoteTrackResolver` links the inbound track to the `remoteOutboundTrack` that published it.
3901
+ * With the whole subscriber set of one source in hand, the verdict is straightforward and is the
3902
+ * single most useful thing a server can say:
3903
+ *
3904
+ * - **most receivers of Alice's track are affected** → the fault is on Alice's path — her uplink, the
3905
+ * SFU's ingress, or its forwarding of that stream.
3906
+ * - **one receiver of Alice's track is affected** → that receiver's downlink. Nothing to do with
3907
+ * Alice, even though the symptom is reported against her stream.
3908
+ *
3909
+ * Deliberately generic over the issue vocabulary: `freezed-video-track`, `keyframe-storm`,
3910
+ * `audio-concealment`, `video-decoder-overloaded`, `stuck-decoder` and anything a custom client
3911
+ * detector invents all fan out the same way, so one mechanism replaces a family of symptom-specific
3912
+ * detectors.
3913
+ *
3914
+ * ### It walks the affected tracks, never all of them
3915
+ *
3916
+ * The detector is fed open issues by the call's registry and keeps only those carrying a `trackId`.
3917
+ * Each tick it resolves *those* tracks to their publishers — never the published tracks of the call,
3918
+ * of which there are many more and almost all of them fine. A call with nothing wrong costs one
3919
+ * `size === 0` check.
3920
+ *
3921
+ * ### Requires a `RemoteTrackResolver`
3922
+ *
3923
+ * Without publisher↔subscriber links there is no way to know which receivers belong to one source,
3924
+ * so the detector does nothing when the call has no resolver. It does not fall back to guessing:
3925
+ * "one receiver of an unknown set" is not a statement worth raising.
3926
+ */
3927
+ declare class IssueFanOutDetector implements Detector, ActiveIssueTracker {
3928
+ private readonly _call;
3929
+ static readonly NAME = "issue-fan-out-detector";
3930
+ readonly name = "issue-fan-out-detector";
3931
+ private readonly _config;
3932
+ private readonly _lastRaisedAt;
3933
+ /** Open issues that name a track. Issues without a `trackId` cannot be attributed and are dropped. */
3934
+ private readonly _trackIssues;
3935
+ constructor(_call: ObservedCall, config?: Partial<IssueFanOutDetectorConfig>);
3936
+ get size(): number;
3937
+ has(issue: ActiveClientIssue): boolean;
3938
+ add(issue: ActiveClientIssue): void;
3939
+ delete(issue: ActiveClientIssue): boolean;
3940
+ clear(): void;
3941
+ update(): void;
3942
+ close(): void;
3943
+ /**
3944
+ * Resolve the issue's `trackId` to the outbound track that published it.
3945
+ *
3946
+ * Looked up through the reporting client's own peer connections rather than by scanning the call:
3947
+ * the issue names its client, so the search is bounded by that client's transports (typically one
3948
+ * or two) instead of by the size of the meeting.
3949
+ */
3950
+ private _publisherOf;
3951
+ }
3952
+
3953
+ /**
3954
+ * One completed sampling bucket: how many/which clients reported congestion during it.
3955
+ *
3956
+ * `totalClients` (and therefore `congestedClientRatio`) is a snapshot of `observer.numberOfClients`
3957
+ * taken when the bucket closes — an approximation of "how many clients could have been congested",
3958
+ * not a claim that every client sent exactly one sample within the bucket. Good enough for a ratio
3959
+ * that only needs to be comparable bucket-to-bucket.
3960
+ */
3961
+ type SfuCongestionDetectorBucket = {
3962
+ observedAt: number;
3963
+ totalClients: number;
3964
+ congestedClients: number;
3965
+ congestedClientRatio: number;
3966
+ affectedClientIds: string[];
3967
+ affectedCallIds: string[];
3968
+ };
3969
+ type SfuCongestionDetectorReport = {
3970
+ affectedCallIds: string[];
3971
+ affectedClientIds: string[];
3972
+ congestedClientRatio: number;
3973
+ totalNumberOfClients: number;
3974
+ numberOfCongestedClients: number;
3975
+ historySize: number;
3976
+ baselineCongestedClientRatio: number;
3977
+ robustZ: number;
3978
+ absoluteIncrease: number;
3979
+ relativeIncrease: number;
3980
+ };
3981
+ type SfuCongestionDetectorConfig = {
3982
+ consumedClientIssueTypes: string[];
3983
+ emittedObserverIssueType: string;
3984
+ samplesSendingTimeInMs: number;
3985
+ historySize: number;
3986
+ minHistorySize: number;
3987
+ minAffectedClients: number;
3988
+ minAbsoluteRatioIncrease: number;
3989
+ minRelativeRatioIncrease: number;
3990
+ robustZThreshold: number;
3991
+ };
3992
+ /** The statistical/practical-significance verdict for one candidate bucket against its baseline. */
3993
+ type SfuCongestionDetectorEvaluation = {
3994
+ isCongested: boolean;
3995
+ baselineCongestedClientRatio: number;
3996
+ robustZ: number;
3997
+ absoluteIncrease: number;
3998
+ relativeIncrease: number;
3999
+ };
4000
+ /**
4001
+ * Detects a **shared** congestion event: many clients, across different calls, reporting congestion
4002
+ * inside the same slice of time.
4003
+ *
4004
+ * Only add this when the observer's calls all come from the **same SFU** — the finding's whole
4005
+ * meaning is "these clients have nothing in common except that server", and that is only true if the
4006
+ * server really is the common factor.
4007
+ *
4008
+ * ### Why fixed-interval buckets, and not the update tick
4009
+ *
4010
+ * The obvious implementation counts congested clients on each `update()`. It is wrong here, for two
4011
+ * separate reasons:
4012
+ *
4013
+ * - **The tick is not evenly spaced.** `update()` fires when a client is updated, so its rate is a
4014
+ * function of how many clients are connected and how their sampling happens to interleave. Two
4015
+ * counts taken from windows of different length are not comparable, and this detector's entire
4016
+ * job is to compare a count against earlier counts.
4017
+ * - **Clients report on their own schedule.** A client sends a sample roughly every
4018
+ * `samplesSendingTimeInMs`, unsynchronised with every other client. A window shorter than that
4019
+ * systematically undercounts — half the congested clients simply hadn't spoken yet — and the
4020
+ * undercount varies with arrival phase, which is noise indistinguishable from signal.
4021
+ *
4022
+ * So the detector runs on a wall-clock interval and closes a bucket every
4023
+ * `samplesSendingTimeInMs`, giving every client a fair chance to be heard in each one. Buckets are
4024
+ * equal-length and equally lagged, which is what makes bucket-to-bucket comparison mean something.
4025
+ * {@link update} is deliberately empty: nothing here is driven by the update tick.
4026
+ *
4027
+ * ### Occurrences, not intervals
4028
+ *
4029
+ * Unlike `ConcurrentIssueDetector`, this one ignores resolutions — see {@link delete}. It counts how
4030
+ * many *distinct clients reported* congestion in a bucket, not how many are still congested. A
4031
+ * client that hits congestion and immediately drops its bitrate resolves the issue within seconds
4032
+ * and would vanish from an open-interval view, yet it is exactly the evidence wanted here.
4033
+ */
4034
+ declare class SfuCongestionDetector implements Detector, ActiveIssueTracker {
4035
+ private readonly _observer;
4036
+ static readonly NAME = "sfu-congestion-detector";
4037
+ readonly name = "sfu-congestion-detector";
4038
+ private readonly _config;
4039
+ private readonly _history;
4040
+ private readonly _trackedIssues;
4041
+ private timer;
4042
+ private _lastEvaluatedBucket;
4043
+ constructor(_observer: Observer, config?: Partial<SfuCongestionDetectorConfig>);
4044
+ get size(): number;
4045
+ /** Record the issue against the bucket currently open. The timer, not this, closes the bucket. */
4046
+ add(issue: ActiveClientIssue): void;
4047
+ /**
4048
+ * Deliberately a no-op returning `false`.
4049
+ *
4050
+ * Resolutions are not interesting here. A congested client typically fixes its own symptom by
4051
+ * dropping bitrate hard, so the issue closes within seconds — but it still *happened*, and it is
4052
+ * evidence that the server was under pressure during this bucket. What matters is how many
4053
+ * distinct clients reported congestion within the bucket and whether that count suddenly jumps,
4054
+ * not how long any one client's issue stayed open.
4055
+ *
4056
+ * Nothing leaks: the tracked set is emptied wholesale every time a bucket closes.
4057
+ */
4058
+ delete(_issue: ActiveClientIssue): boolean;
4059
+ clear(): void;
4060
+ has(issue: ActiveClientIssue): boolean;
4061
+ close(): void;
4062
+ /** The completed buckets kept so far, oldest first. Read-only — for introspection/tests. */
4063
+ get history(): readonly SfuCongestionDetectorBucket[];
4064
+ /**
4065
+ * Intentionally empty — see the class description.
4066
+ *
4067
+ * Everything here is driven by the bucket timer, because the update tick is neither evenly spaced
4068
+ * nor long enough for every client to have reported. Counting on it would compare windows of
4069
+ * different lengths and call the difference a signal.
4070
+ */
4071
+ update(): void;
4072
+ private _closeBucket;
4073
+ /**
4074
+ * Evaluate only the latest completed bucket (the candidate) against the buckets before it (the
4075
+ * baseline) — never against itself. Reached once per newly-closed bucket via {@link update}; the
4076
+ * identity check below additionally guards against evaluating the same bucket twice, in case
4077
+ * `update()` is ever called again before the next rotation.
4078
+ */
4079
+ private _evaluateLatestBucket;
4080
+ /**
4081
+ * Is `candidate` — the latest completed bucket — abnormally high compared with the `baseline`
4082
+ * buckets before it?
4083
+ *
4084
+ * Requires both **statistical** significance (a robust z-score against a median+MAD baseline —
4085
+ * deliberately not Mann-Kendall, which asks "is this a monotonic trend", not "is the latest point
4086
+ * an outlier"; a single sudden spike on an otherwise flat series is exactly what should trigger
4087
+ * here and exactly what a trend test would miss) and **practical** significance (enough affected
4088
+ * clients, and a big enough absolute/relative jump — a statistically significant move in a tiny
4089
+ * or trivial ratio is not worth an alert).
4090
+ */
4091
+ private _evaluateBucket;
4092
+ }
4093
+
4094
+ declare const TrackDeliveryMismatchTypes: {
4095
+ /**
4096
+ * The source is sending, but **none** of its subscribers are receiving → the media is being lost
4097
+ * between the publisher and the receivers. In an SFU that means the forwarding path.
4098
+ */
4099
+ readonly publishedTrackNotDelivered: "PUBLISHED_TRACK_NOT_DELIVERED";
4100
+ /**
4101
+ * The source is sending and most subscribers are fine, but **some** are dry → those consumers are
4102
+ * broken individually (in mediasoup, the usual mitigation is recreating the consumer).
4103
+ */
4104
+ readonly receiverTrackNotDelivered: "RECEIVER_TRACK_NOT_DELIVERED";
4105
+ /**
4106
+ * The source itself stopped producing, so its subscribers being dry is expected and **not** an
4107
+ * SFU fault. Reported so the other two verdicts can be trusted as *not* being this.
4108
+ */
4109
+ readonly publisherTrackDry: "PUBLISHER_TRACK_DRY";
4110
+ };
4111
+ type TrackDeliveryMismatchDetectorConfig = {
4112
+ /** The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`. */
4113
+ dryInboundIssueType: string;
4114
+ /** The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`. */
4115
+ dryOutboundIssueType: string;
4116
+ /** Minimum subscribers before "all of them" means anything. Default `2`. */
4117
+ minReceivers: number;
4118
+ /** Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`. */
4119
+ allReceiversRatio: number;
4120
+ /** Re-arm time (ms) per (track, verdict). Default `60_000`. */
4121
+ cooldownMs: number;
4122
+ };
4123
+ /**
4124
+ * Answers **"is the media actually getting through?"** by joining the two ends of a published track.
4125
+ *
4126
+ * A dry track is the clearest possible symptom — no bytes are arriving — but on its own it is
4127
+ * ambiguous, and the ambiguity is precisely what a single endpoint cannot resolve. A receiver seeing
4128
+ * silence cannot tell whether the camera was switched off, the SFU stopped forwarding, or its own
4129
+ * consumer wedged. All three look identical from the browser.
4130
+ *
4131
+ * With the publisher↔subscriber links this becomes a three-way decision:
4132
+ *
4133
+ * | publisher | subscribers | verdict |
4134
+ * |---|---|---|
4135
+ * | sending | **all** dry | `PUBLISHED_TRACK_NOT_DELIVERED` — the SFU/forwarding path |
4136
+ * | sending | **some** dry | `RECEIVER_TRACK_NOT_DELIVERED` — those consumers (recreate them) |
4137
+ * | dry | any dry | `PUBLISHER_TRACK_DRY` — the source stopped; not an SFU fault |
4138
+ *
4139
+ * The publisher side is judged from **both** signals available: its own `dry-outbound-track` issue
4140
+ * when the client reports one, and — as the fallback, and the corroboration when it does not — the
4141
+ * observed outbound RTP (`deltaPacketsSent`). That combination is what makes the first row
4142
+ * trustworthy: the server can state that packets demonstrably left the publisher during the same
4143
+ * interval in which every receiver got nothing.
4144
+ *
4145
+ * This is the "SFU forwarding mismatch" check, and notably it needs **no** mediasoup instrumentation
4146
+ * — the client's own dry-track verdicts plus the resolver links are sufficient.
4147
+ */
4148
+ declare class TrackDeliveryMismatchDetector implements Detector, ActiveIssueTracker {
4149
+ private readonly call;
4150
+ static readonly NAME: "track-delivery-mismatch-detector";
4151
+ readonly name: "track-delivery-mismatch-detector";
4152
+ private readonly _config;
4153
+ private readonly _lastRaisedAt;
4154
+ private readonly dryOutboundTracks;
4155
+ private readonly dryInboundTracks;
4156
+ constructor(call: ObservedCall, config?: Partial<TrackDeliveryMismatchDetectorConfig>);
4157
+ close(): void;
4158
+ add(issue: ActiveClientIssue): void;
4159
+ delete(issue: ActiveClientIssue): boolean;
4160
+ get size(): number;
4161
+ clear(): void;
4162
+ has(issue: ActiveClientIssue): boolean;
4163
+ update(): void;
4164
+ }
4165
+
4166
+ declare const TurnServerHealthTypes: {
4167
+ /** One TURN server's clients are in trouble while other servers' clients are fine. */
4168
+ readonly turnServerDegraded: "TURN_SERVER_DEGRADED";
4169
+ };
4170
+ type TurnServerHealthDetectorConfig = {
4171
+ /** Minimum clients on a server before a ratio is meaningful. Default `5`. */
4172
+ minClientsPerServer: number;
4173
+ /** Fraction of a server's clients that must have an open issue. Default `0.5`. */
4174
+ degradedRatioThreshold: number;
4175
+ /**
4176
+ * Which client issue types count as "in trouble". Empty (default) means **any** open issue —
4177
+ * appropriate here, because the question is not *what* is wrong with each client but whether
4178
+ * trouble clusters on one relay.
4179
+ */
4180
+ issueTypes: string[];
4181
+ /** Consecutive ticks the condition must hold before raising. Default `2`. */
4182
+ consecutiveTicks: number;
4183
+ /** Re-arm time (ms) before raising again for the same server. Default `60_000`. */
4184
+ cooldownMs: number;
4185
+ };
4186
+ /** The per-server view this detector builds. */
4187
+ type TurnServerHealth = {
4188
+ serverUrl: string;
4189
+ /** Distinct clients whose media is relayed through this server. */
4190
+ clients: number;
4191
+ /** Of those, how many currently have at least one open issue. */
4192
+ degradedClients: number;
4193
+ degradedRatio: number;
4194
+ affectedClientIds: string[];
4195
+ /** The open issue types seen on this server's clients, most common first. */
4196
+ issueTypes: string[];
4197
+ };
4198
+ /**
4199
+ * An **observer-level** detector that groups relayed clients by the TURN server carrying them and
4200
+ * compares the servers against each other.
4201
+ *
4202
+ * Counting TURN usage is not useful on its own; knowing that `turn-eu-1` has 22 of 30 clients in
4203
+ * trouble while `turn-eu-2` has 1 of 34 is. Because the comparison spans calls it lives on
4204
+ * `observer.detectors` and raises `observer-issue` — one actionable alert instead of fifty
4205
+ * per-client ones. Each finding carries the other servers' ratios as context, since "half the
4206
+ * clients here are unhappy" only means something relative to the rest of the fleet.
4207
+ *
4208
+ * Whether a client is in trouble comes from **its own reported issues**, not from thresholds applied
4209
+ * here. The client already decides that far better than a server-side rule could; the value this
4210
+ * adds is the grouping — the dimension no endpoint can see.
4211
+ *
4212
+ * For a relay that has stopped serving entirely, see `TurnServerOutageDetector`: this detector needs
4213
+ * clients *on* the server to ask how many are unhappy, and an outage takes them away.
4214
+ */
4215
+ declare class TurnServerHealthDetector implements Detector {
4216
+ private readonly _observer;
4217
+ static readonly NAME = "turn-server-health-detector";
4218
+ readonly name = "turn-server-health-detector";
4219
+ private readonly _config;
4220
+ private readonly _streaks;
4221
+ private readonly _lastRaisedAt;
4222
+ /** The per-server rollup computed on the most recent `update()`. */
4223
+ lastServers: TurnServerHealth[];
4224
+ constructor(_observer: Observer, config?: Partial<TurnServerHealthDetectorConfig>);
4225
+ update(): void;
4226
+ close(): void;
4227
+ private _serverHealth;
4228
+ }
4229
+
4230
+ declare const TurnServerOutageTypes: {
4231
+ /** One TURN server's relayed population collapsed while the rest of the fleet is fine. */
4232
+ readonly turnServerOutage: "TURN_SERVER_OUTAGE";
4233
+ };
4234
+ type TurnServerOutageDetectorConfig = {
4235
+ /**
4236
+ * How many clients a server must have been carrying at its peak before its collapse means
4237
+ * anything. Below this, one or two people leaving looks like an outage. Default `5`.
4238
+ */
4239
+ minClientsAtPeak: number;
4240
+ /**
4241
+ * Fraction of the peak population that must be gone or disrupted. Default `0.8` — an outage is
4242
+ * near-total by definition; partial degradation is `TurnServerHealthDetector`'s question.
4243
+ */
4244
+ lossRatioThreshold: number;
4245
+ /**
4246
+ * Window (ms) the peak population is measured over. Long enough to span a real outage's onset,
4247
+ * short enough that yesterday's peak isn't held against today. Default `120_000`.
4248
+ */
4249
+ peakWindowMs: number;
4250
+ /**
4251
+ * Require a healthy **control group** — clients not relayed through this server that are still
4252
+ * connected — before blaming the server. Without this, a call ending, a fleet-wide network
4253
+ * event, or the observer shutting down all look exactly like a TURN outage. Default `true`.
4254
+ */
4255
+ requireControlGroup: boolean;
4256
+ /** Minimum clients elsewhere before the control group is statistically worth anything. Default `5`. */
4257
+ minControlGroupClients: number;
4258
+ /** Fraction of the control group that must still be healthy. Default `0.7`. */
4259
+ controlGroupHealthyRatio: number;
4260
+ /** Consecutive ticks the condition must hold before raising. Default `2`. */
4261
+ consecutiveTicks: number;
4262
+ /**
4263
+ * Re-arm time (ms) per server. Long by default (`300_000`) — an outage is one event, not one
4264
+ * per tick, and a server that stays down would otherwise alert forever.
4265
+ */
4266
+ cooldownMs: number;
4267
+ };
4268
+ /**
4269
+ * Detects a **TURN server outage** — a relay that has stopped serving — by watching its client
4270
+ * population collapse while the rest of the fleet carries on.
4271
+ *
4272
+ * This is the case its sibling `TurnServerHealthDetector` structurally *cannot* see, and the
4273
+ * distinction is worth being precise about. That detector groups clients by the server relaying them
4274
+ * and asks how many are reporting issues. It needs clients on the server to ask the question. When a
4275
+ * TURN server goes down completely, allocation fails: existing sessions drop, and new clients never
4276
+ * obtain a relay candidate through it at all, so they are never attributed to it. The server's
4277
+ * population goes to zero and the health detector falls silent for the worst possible reason — it
4278
+ * has nobody left to ask. Degradation makes clients unhappy; an outage makes them *disappear*.
4279
+ *
4280
+ * So the signal here is absence, measured against the server's own recent peak:
4281
+ *
4282
+ * - clients gone entirely (their relayed peer connections closed, or they re-negotiated onto a
4283
+ * different path), plus
4284
+ * - clients still attributed to the server whose ICE or connection state is `disconnected` /
4285
+ * `failed` / `closed` — the ones mid-collapse, which is what you catch if you look during the
4286
+ * outage rather than after it.
4287
+ *
4288
+ * ### The control group is the whole design
4289
+ *
4290
+ * Absence is a dangerous signal: a call ending, everyone going home at 6pm, a fleet-wide network
4291
+ * event, and the observer itself shutting down all produce exactly the same collapse. The detector
4292
+ * therefore refuses to blame a server unless clients **not** relayed through it are demonstrably
4293
+ * still connected — `requireControlGroup`, on by default. "Everyone on `turn-eu-1` vanished" is
4294
+ * ambiguous; "everyone on `turn-eu-1` vanished while 200 clients elsewhere are fine" is an outage.
4295
+ *
4296
+ * That comparison is only available to something watching every call at once, which is why this is
4297
+ * an observer-level detector raising `observer-issue` — one alert for the fleet, not one per
4298
+ * abandoned call.
4299
+ *
4300
+ * ### Caveats worth knowing before you tune it
4301
+ *
4302
+ * Clients that fail over cleanly to a second TURN server still count as lost here, which is
4303
+ * correct — the server did stop serving them — but it means a well-configured fleet with automatic
4304
+ * failover reports outages that users never felt. That is the intended behaviour: the failover
4305
+ * worked *and* the server is down are both true, and you want to know the second one.
4306
+ *
4307
+ * A genuinely quiet server (last call of the day ends) is suppressed by the control group, not by
4308
+ * the collapse test. If you run a small deployment where the control group is routinely below
4309
+ * `minControlGroupClients`, this detector will stay quiet — prefer alerting on your TURN server's
4310
+ * own health checks there, since a handful of clients cannot distinguish these cases.
4311
+ */
4312
+ declare class TurnServerOutageDetector implements Detector {
4313
+ private readonly _observer;
4314
+ static readonly NAME = "turn-server-outage-detector";
4315
+ readonly name = "turn-server-outage-detector";
4316
+ private readonly _config;
4317
+ /** serverUrl -> recent population observations, used to derive the windowed peak. */
4318
+ private readonly _peaks;
4319
+ private readonly _streaks;
4320
+ private readonly _lastRaisedAt;
4321
+ constructor(_observer: Observer, config?: Partial<TurnServerOutageDetectorConfig>);
4322
+ update(): void;
4323
+ close(): void;
4324
+ /** Distinct clients on a server, split by whether their relayed transport is actually up. */
4325
+ private _populationOf;
4326
+ /** Record this tick's population and return the peak across `peakWindowMs`. */
4327
+ private _recordAndPeak;
4328
+ /**
4329
+ * Everyone *not* relayed through `serverUrl`: clients on other TURN servers plus every client
4330
+ * the observer knows about that isn't relayed at all. The healthy share of that group is what
4331
+ * separates "this server broke" from "everything broke".
4332
+ */
4333
+ private _controlGroup;
4334
+ }
4335
+
4336
+ declare const UnconsumedTrackTypes: {
4337
+ /** A track is being published to the SFU that nobody is subscribed to — pure wasted uplink. */
4338
+ readonly unconsumedPublishedTrack: "UNCONSUMED_PUBLISHED_TRACK";
4339
+ };
4340
+ type UnconsumedTrackDetectorConfig = {
4341
+ /** How long a track must stay unconsumed while sending before reporting (ms). Default `30_000`. */
4342
+ minUnconsumedDurationInMs: number;
4343
+ /** Ignore tracks below this bitrate — a trickle isn't worth an alert (bps). Default `50_000`. */
4344
+ minBitrate: number;
4345
+ /** Re-arm time (ms) per track. Default `300_000`. */
4346
+ cooldownMs: number;
4347
+ };
4348
+ /**
4349
+ * Finds tracks that are **published but consumed by nobody** — uplink and SFU ingress spent on media
4350
+ * that is never forwarded anywhere.
4351
+ *
4352
+ * This is the one detector that reads the resolver's *silence* as the signal: an outbound track with
4353
+ * an empty `remoteInboundTracks` set, still pushing packets. It reads `call.unconsumedOutboundTracks`,
4354
+ * which the resolver maintains as tracks gain and lose subscribers, so a healthy call costs one
4355
+ * `size === 0` check rather than a walk over every published track. The usual causes are a participant
4356
+ * publishing while everyone has them hidden or muted-in-UI, a simulcast layer no viewer's bandwidth
4357
+ * ever selects, or an application that forgot to stop a track after the last subscriber left.
4358
+ *
4359
+ * It is deliberately slow to fire: `minUnconsumedDurationInMs` must elapse with the track still
4360
+ * sending, because a brief gap between publishing and the first subscription is completely normal at
4361
+ * join time.
4362
+ *
4363
+ * ### Careful: this detector is only sound with a resolver
4364
+ *
4365
+ * "No subscribers" and "no resolver configured" produce the identical observation — an empty link
4366
+ * set. Without a `RemoteTrackResolver` this would report *every* published track in the call as
4367
+ * unconsumed, so it checks `call.remoteTrackResolver` at runtime and does nothing without one.
4368
+ */
4369
+ declare class UnconsumedTrackDetector implements Detector {
4370
+ private readonly call;
4371
+ static readonly NAME = "unconsumed-track-detector";
4372
+ readonly name = "unconsumed-track-detector";
4373
+ readonly config: UnconsumedTrackDetectorConfig;
4374
+ /** trackId -> when it was first seen sending with no subscribers. */
4375
+ private readonly _unconsumedSince;
4376
+ private readonly _lastRaisedAt;
4377
+ constructor(call: ObservedCall, config?: Partial<UnconsumedTrackDetectorConfig>);
4378
+ update(): void;
4379
+ close(): void;
4380
+ }
4381
+
4382
+ /**
4383
+ * Detectors that reason **across calls**, created once on the observer.
4384
+ *
4385
+ * Adding one means: give the class a `static readonly NAME`, add its entry here, and add a `case` to
4386
+ * `Observer.addObserverDetector`. This map is what types the call site — the config is checked
4387
+ * against the right detector and an unknown name won't compile.
4388
+ */
4389
+ type AvailableObserverScopeDetectorsConfigs = {
4390
+ [SfuCongestionDetector.NAME]: SfuCongestionDetectorConfig;
4391
+ [ObserverConcurrentIssueDetector.NAME]: ObserverConcurrentIssueDetectorConfig;
4392
+ [ClientPopulationIssueDetector.NAME]: ClientPopulationIssueDetectorConfig;
4393
+ [TurnServerHealthDetector.NAME]: TurnServerHealthDetectorConfig;
4394
+ [TurnServerOutageDetector.NAME]: TurnServerOutageDetectorConfig;
4395
+ };
4396
+ /**
4397
+ * Detectors that reason **within one call**, created for every call the observer opens.
4398
+ *
4399
+ * Note there is no detector in both maps. "Is this meeting in trouble?" and "is our infrastructure in
4400
+ * trouble?" are different questions with different gates and different findings, so they are separate
4401
+ * classes — `CallConcurrentIssueDetector` and `ObserverConcurrentIssueDetector` — rather than one
4402
+ * class branching on what it was handed.
4403
+ */
4404
+ type AvailableCallScopeDetectorsConfigs = {
4405
+ [UnconsumedTrackDetector.NAME]: UnconsumedTrackDetectorConfig;
4406
+ [TrackDeliveryMismatchDetector.NAME]: TrackDeliveryMismatchDetectorConfig;
4407
+ [CallConcurrentIssueDetector.NAME]: CallConcurrentIssueDetectorConfig;
4408
+ [IssueFanOutDetector.NAME]: IssueFanOutDetectorConfig;
4409
+ [PublisherFaultCorroborationDetector.NAME]: PublisherFaultCorroborationDetectorConfig;
4410
+ };
4411
+ type AvailableDetectorsConfigs = AvailableObserverScopeDetectorsConfigs | AvailableCallScopeDetectorsConfigs;
4412
+ declare class Detectors {
4413
+ private _detectors;
4414
+ constructor(...detectors: Detector[]);
4415
+ get listOfNames(): string[];
4416
+ get size(): number;
4417
+ add(detector: Detector): void;
4418
+ get(name: string): Detector | undefined;
4419
+ remove(detector: Detector): void;
4420
+ update(): void;
4421
+ clear(): void;
4422
+ private _close;
4423
+ }
4424
+
4425
+ /**
4426
+ * The set of client issues currently believed to be **open**, plus the fan-out that pushes them to
4427
+ * whoever asked for them.
4428
+ *
4429
+ * ### Push, not poll
4430
+ *
4431
+ * A detector does not scan for the issues it cares about; it registers as an
4432
+ * {@link ActiveIssueTracker} for the types it consumes and is handed them as they open and close.
4433
+ * The cost of a detector is then proportional to the issues it actually receives, not to the number
4434
+ * of participants — a healthy 500-client fleet does no work per tick.
4435
+ *
4436
+ * ```ts
4437
+ * observer.activeIssuesRegistry.addIssueTracker('congestion', detector);
4438
+ * ```
4439
+ *
4440
+ * There is **no wildcard**. A tracker names the types it consumes, and nothing else reaches it. "Feed
4441
+ * me everything and I'll work out what matters" pushes the decision from the application — which
4442
+ * knows its client build and its issue vocabulary — onto a detector that has to guess, and it makes
4443
+ * the cost of a subscription unbounded and invisible. If a detector should watch five issue types,
4444
+ * the caller lists five issue types.
4445
+ *
4446
+ * ### Two levels
4447
+ *
4448
+ * Every call owns a registry constructed with the observer's as its `parent`. An add or delete
4449
+ * touches both, so a call-scoped tracker sees only that call's issues while an observer-scoped one
4450
+ * sees the fleet — without either side iterating the other. The child keeps **its own** storage:
4451
+ * `size` is this scope's count, and {@link clear} (called when the call closes) removes only this
4452
+ * scope's issues from the parent and never touches the parent's tracker registrations.
4453
+ *
4454
+ * ### Only keyed issues arrive here
4455
+ *
4456
+ * An issue without a `key` has no lifecycle — nothing can ever close it — so treating it as "active"
4457
+ * would mean holding a symptom that may have ended long ago. Keyless issues stay one-shot: emitted
4458
+ * as `client-issue`, never registered. `client-monitor-js` >= 4.6.0 sends `key` on everything
4459
+ * stateful.
4460
+ */
4461
+ declare class ActiveIssuesRegistry implements ActiveIssueTracker {
4462
+ private readonly parent?;
4463
+ private readonly issues;
4464
+ private readonly typesToTrackers;
4465
+ constructor(parent?: ActiveIssueTracker | undefined);
4466
+ get size(): number;
4467
+ /**
4468
+ * The open issues in this scope, in insertion order.
4469
+ *
4470
+ * Insertion order is age order (`observedAt` is assigned on insert), which is what lets a consumer
4471
+ * stop at the first entry newer than its cutoff instead of scanning the whole set.
4472
+ */
4473
+ values(): IterableIterator<ActiveClientIssue>;
4474
+ [Symbol.iterator](): IterableIterator<ActiveClientIssue>;
4475
+ has(issue: ActiveClientIssue): boolean;
4476
+ add(issue: ActiveClientIssue): this;
4477
+ delete(issue: ActiveClientIssue): boolean;
4478
+ /** Feed `tracker` every issue of `type` as it opens and closes. One call per type; no wildcard. */
4479
+ addIssueTracker(type: string, tracker: ActiveIssueTracker): this;
4480
+ removeIssueTracker(tracker: ActiveIssueTracker): this;
4481
+ /**
4482
+ * Drop every issue in this scope, e.g. because the call closed.
4483
+ *
4484
+ * Deletes through {@link delete} so the parent sheds exactly this scope's issues. Tracker
4485
+ * *registrations* survive: a detector subscribed to the observer's registry must keep receiving
4486
+ * issues after any one call ends.
4487
+ */
4488
+ clear(): void;
4489
+ private _trackIssue;
4490
+ private _untrackIssue;
4491
+ /**
4492
+ * Apply `apply` to every tracker interested in `type`.
4493
+ *
4494
+ * A tracker throwing must not abort the fan-out: the issue has already been added to (or removed
4495
+ * from) this registry, so a partial dispatch would leave the remaining trackers permanently out of
4496
+ * step with it. One broken detector should not desynchronise the others.
4497
+ */
4498
+ private _trackersOf;
4499
+ private _safely;
4500
+ }
4501
+
4502
+ type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
4503
+ callId: string;
4504
+ appData?: AppData;
4505
+ closeCallIfEmptyForMs?: number;
4506
+ /**
4507
+ * When `true`, the call's `update()` is invoked whenever a client accepts a sample. When `false`, it is not.
4508
+ *
4509
+ * DEFAULT: `true` — the call is updated on every client sample, which is the most common use case.
4510
+ */
4511
+ autoUpdateOnClientUpdate?: boolean;
4512
+ };
4513
+ type ObservedCallEvents = {
4514
+ update: [];
4515
+ newclient: [ObservedClient];
4516
+ empty: [];
4517
+ 'not-empty': [];
4518
+ close: [];
4519
+ };
4520
+ declare interface ObservedCall {
3024
4521
  on<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
3025
4522
  off<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
3026
4523
  once<U extends keyof ObservedCallEvents>(event: U, listener: (...args: ObservedCallEvents[U]) => void): this;
@@ -3028,7 +4525,7 @@ declare interface ObservedCall {
3028
4525
  }
3029
4526
  declare class ObservedCall<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
3030
4527
  readonly observer: Observer;
3031
- updater?: Updater;
4528
+ readonly activeIssuesRegistry: ActiveIssuesRegistry;
3032
4529
  scoreCalculator: ScoreCalculator;
3033
4530
  readonly detectors: Detectors;
3034
4531
  readonly callId: string;
@@ -3036,6 +4533,17 @@ declare class ObservedCall<AppData extends Record<string, unknown> = Record<stri
3036
4533
  readonly clientsUsedTurn: Set<string>;
3037
4534
  readonly calculatedScore: CalculatedScore;
3038
4535
  remoteTrackResolver?: RemoteTrackResolver;
4536
+ /**
4537
+ * Published tracks that currently have **no** subscriber linked to them.
4538
+ *
4539
+ * Maintained by the `RemoteTrackResolver` at the exact moments a track gains or loses its last
4540
+ * subscriber — the only moments the answer can change. `UnconsumedTrackDetector` reads this
4541
+ * instead of walking every published track in the call, so in a healthy call (where the set is
4542
+ * empty) it does no work at all.
4543
+ *
4544
+ * Empty when no resolver is configured: without links, "no subscribers" is unknowable.
4545
+ */
4546
+ readonly unconsumedOutboundTracks: Set<ObservedOutboundTrack>;
3039
4547
  totalAddedClients: number;
3040
4548
  totalRemovedClients: number;
3041
4549
  numberOfIssues: number;
@@ -3050,15 +4558,21 @@ declare class ObservedCall<AppData extends Record<string, unknown> = Record<stri
3050
4558
  startedAt?: number;
3051
4559
  endedAt?: number;
3052
4560
  closedAt?: number;
3053
- readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs'>;
4561
+ readonly settings: Pick<ObservedCallSettings, 'closeCallIfEmptyForMs' | 'autoUpdateOnClientUpdate'>;
3054
4562
  /** Ancestry base shared by all Observer-bus events originating at this call. */
3055
4563
  readonly eventScope: ObservedCallScope;
3056
4564
  private closeTimer?;
3057
- constructor(settings: ObservedCallSettings<AppData>, observer: Observer);
4565
+ constructor(settings: ObservedCallSettings<AppData>, observer: Observer, activeIssuesRegistry: ActiveIssuesRegistry);
3058
4566
  get numberOfClients(): number;
3059
4567
  get score(): number | undefined;
3060
- /** Raise a call-level (server-side) issue; surfaced on the Observer bus as `call-issue`. */
3061
- addIssue(issue: ClientIssue): void;
4568
+ addDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
4569
+ /**
4570
+ * Raise a call-level (server-side) finding; surfaced on the Observer bus as `call-issue`.
4571
+ *
4572
+ * `payload` takes an **object** — it is delivered to in-process handlers, so there is nothing to
4573
+ * serialise for. Pass a string only if you already have one.
4574
+ */
4575
+ addIssue(issue: ObserverIssue): void;
3062
4576
  close(): void;
3063
4577
  getObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(clientId: string): ObservedClient<ClientAppData> | undefined;
3064
4578
  createObservedClient<ClientAppData extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedClientSettings<ClientAppData>): ObservedClient<ClientAppData> | undefined;
@@ -3071,21 +4585,393 @@ declare class ObservedCall<AppData extends Record<string, unknown> = Record<stri
3071
4585
  private _notify;
3072
4586
  }
3073
4587
 
3074
- type Middleware<T> = (input: T, next: (nextInput: T) => void) => void;
3075
- interface Processor<T> {
3076
- finalCallback?: Callback<T>;
3077
- process(value: T): void;
3078
- addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
3079
- removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
3080
- }
3081
- type Callback<T> = (input: T) => void;
3082
- declare class MiddlewareProcessor<T> implements Processor<T> {
3083
- private stack;
3084
- finalCallback?: Callback<T>;
3085
- addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
3086
- removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
3087
- process(value: T): void;
3088
- }
4588
+ type Middleware<T> = (input: T, next: (nextInput: T) => void) => void;
4589
+ interface Processor<T> {
4590
+ finalCallback?: Callback<T>;
4591
+ process(value: T): void;
4592
+ addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
4593
+ removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
4594
+ }
4595
+ type Callback<T> = (input: T) => void;
4596
+ declare class MiddlewareProcessor<T> implements Processor<T> {
4597
+ private stack;
4598
+ finalCallback?: Callback<T>;
4599
+ addMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
4600
+ removeMiddleware(...middlewares: Middleware<T>[]): Processor<T>;
4601
+ process(value: T): void;
4602
+ }
4603
+
4604
+ /** Raised once if one receiver turns out to be dragging a publisher down for everyone. */
4605
+ declare const LOWEST_COMMON_DENOMINATOR_ISSUE = "WORST_RECEIVER_CONTAGION";
4606
+ /** The measurements behind a decided verdict — everything needed to check the call yourself. */
4607
+ type SimulcastReceiverEvidence = {
4608
+ callId: string;
4609
+ trackId: string;
4610
+ publisherClientId: string;
4611
+ worstReceiverClientId: string;
4612
+ publisherBitrate: number;
4613
+ worstReceiverBitrate: number;
4614
+ medianReceiverBitrate: number;
4615
+ /** How closely the publisher's bitrate followed the **worst** receiver's, `0..1`. */
4616
+ trackingWithWorst: number;
4617
+ /** The same against the **median** receiver — the control. */
4618
+ trackingWithMedian: number;
4619
+ };
4620
+ type SimulcastReceiverReportPayload = ({
4621
+ /** The publisher held up while one receiver lagged: layers are being chosen per consumer. */
4622
+ verdict: 'layer-decided-per-receiver';
4623
+ evidence: SimulcastReceiverEvidence;
4624
+ } | {
4625
+ /** The publisher tracked its worst receiver: everyone is getting the lowest common denominator. */
4626
+ verdict: 'layer-decided-lowest-common-denominator';
4627
+ evidence: SimulcastReceiverEvidence;
4628
+ } | {
4629
+ /** Gave up without the conditions needed to judge. **Not a pass.** */
4630
+ verdict: 'inconclusive';
4631
+ reason: string;
4632
+ }) & {
4633
+ startedAt: number;
4634
+ /** How many times the check actually ran — i.e. how much the verdict is worth. */
4635
+ checks: number;
4636
+ };
4637
+ type SimulcastReceiverValidatorConfig = {
4638
+ /** Minimum receivers of a published track before the comparison means anything. */
4639
+ minReceivers: number;
4640
+ /** How long a track's bitrates are correlated over (ms). */
4641
+ windowMs: number;
4642
+ /** Samples needed inside the window before it can be judged. */
4643
+ minSamples: number;
4644
+ /** The worst receiver must be at most this share of the median, or there is nothing to be dragged by. */
4645
+ outlierRatioThreshold: number;
4646
+ /** How closely the publisher must follow the worst receiver to count as dragged. */
4647
+ trackingRatioThreshold: number;
4648
+ /** Clean checks required before concluding per-receiver adaptation. One could be luck. */
4649
+ minChecks: number;
4650
+ };
4651
+ /**
4652
+ * Answers one question: **does this SFU adapt each receiver on its own, or does one bad receiver
4653
+ * drag the publisher down for everyone?**
4654
+ *
4655
+ * That is what simulcast (or SVC) exists to prevent. With several encodings available the server can
4656
+ * hand the struggling participant a lower layer and leave everyone else alone. Without it — or with
4657
+ * a server that relays RTCP end to end instead of terminating it, so the publisher's bandwidth
4658
+ * estimate collapses to the minimum across all receivers — the only way to serve the slowest
4659
+ * participant is to make the source send less, and everybody gets the lowest common denominator.
4660
+ *
4661
+ * The two causes are worth naming because the *observation* cannot separate them: the publisher's
4662
+ * bitrate tracking its worst receiver looks identical either way. What the check establishes is
4663
+ * whether per-receiver adaptation is happening at all. If the verdict is
4664
+ * `layer-decided-lowest-common-denominator`, look at both — is simulcast/SVC actually enabled with
4665
+ * layers selected per consumer, and is the SFU terminating receiver reports rather than forwarding
4666
+ * them?
4667
+ *
4668
+ * ### The control matters more than the correlation
4669
+ *
4670
+ * "Publisher follows worst receiver" alone proves nothing: when the whole call degrades together,
4671
+ * the publisher follows *everyone*, and that is ordinary adaptation working correctly. The verdict
4672
+ * only goes against the deployment when the publisher tracks the worst receiver **more closely than
4673
+ * it tracks the median** — the worst receiver is leading, not merely coinciding.
4674
+ *
4675
+ * Likewise, a window with no outlier in it is not evidence of health, it is an untested SFU: if
4676
+ * nobody is struggling, there is nothing for per-receiver adaptation to do. Those windows are
4677
+ * skipped and never counted in `checks`.
4678
+ *
4679
+ * ### Why a validator, not a detector
4680
+ *
4681
+ * This is a property of the SFU build and configuration, not of this moment: a server doing
4682
+ * per-receiver layer selection at 09:00 still is at 17:00. Re-deriving it every tick cannot produce
4683
+ * new information — it would only keep a sliding window alive per published track for the life of
4684
+ * every call. So it decides once, reports, and releases that state.
4685
+ *
4686
+ * ```ts
4687
+ * observer.on('validation-ready', ({ validator, report }) => {
4688
+ * if (validator !== 'simulcast-receivers' || !report.ready) return;
4689
+ * console.log(report.verdict); // 'layer-decided-per-receiver' | ... | 'inconclusive'
4690
+ * });
4691
+ *
4692
+ * observer.addValidator('simulcast-receivers');
4693
+ * onDeploy(() => observer.addValidator('simulcast-receivers')); // check again
4694
+ * ```
4695
+ *
4696
+ * ### `inconclusive` is not a pass
4697
+ *
4698
+ * The check only runs when a publisher has several receivers and one of them is far behind the
4699
+ * median; plenty of healthy deployments never present that. Concluding from the absence of a failure
4700
+ * would verify nothing, so `checks` counts the times the check genuinely ran, and a validator that
4701
+ * is cancelled (or whose observer closes) finishes `inconclusive` with the reason why.
4702
+ */
4703
+ declare class SimulcastReceiverValidator implements Validator<SimulcastReceiverReportPayload> {
4704
+ private readonly _observer;
4705
+ readonly onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void;
4706
+ static readonly NAME: "simulcast-receivers";
4707
+ readonly name: "simulcast-receivers";
4708
+ readonly startedAt: number;
4709
+ report: ValidationReport<SimulcastReceiverReportPayload>;
4710
+ private readonly _config;
4711
+ private readonly _windows;
4712
+ private _checks;
4713
+ private _done;
4714
+ constructor(_observer: Observer, onDone: (report: ValidationReport<SimulcastReceiverReportPayload>) => void, config?: Partial<SimulcastReceiverValidatorConfig>);
4715
+ /** How many times the comparison actually ran. `0` means nothing was established. */
4716
+ get checks(): number;
4717
+ /** Give up without a verdict, freeing anything waiting on this validator. */
4718
+ cancel(reason?: string): void;
4719
+ update(): void;
4720
+ /** Returns `true` when a verdict was reached and the caller should stop iterating. */
4721
+ private _inspect;
4722
+ /**
4723
+ * Settle on a verdict, exactly once.
4724
+ *
4725
+ * The guard is not paranoia: `onDone` removes this validator from the observer, and a second call
4726
+ * would emit a second `validation-ready` for a validator that is no longer registered — e.g. when
4727
+ * `observer.close()` cancels a validator that decided earlier in the same tick.
4728
+ */
4729
+ private _finish;
4730
+ private _windowOf;
4731
+ }
4732
+
4733
+ /** Raised once if the resolver turns out never to link anything. */
4734
+ declare const UNRESOLVED_TRACK_LINKS_ISSUE = "REMOTE_TRACK_LINKS_UNRESOLVED";
4735
+ /** What the check actually saw, whichever way it went. */
4736
+ type RemoteTrackLinkEvidence = {
4737
+ /** Calls that presented the conditions for linking: a resolver, ≥2 clients, and inbound tracks. */
4738
+ eligibleCalls: number;
4739
+ /** Inbound tracks seen across those calls. */
4740
+ inboundTracks: number;
4741
+ /** Of those, how many were linked to the outbound track that published them. */
4742
+ linkedInboundTracks: number;
4743
+ /** `linkedInboundTracks / inboundTracks`. */
4744
+ linkedRatio: number;
4745
+ /** A call that presented the conditions, for the reader to go and look at. */
4746
+ exampleCallId?: string;
4747
+ };
4748
+ type RemoteTrackResolverReportPayload = ({
4749
+ /** The resolver is linking subscribers to publishers. The detectors that need links will work. */
4750
+ verdict: 'links-resolved';
4751
+ evidence: RemoteTrackLinkEvidence;
4752
+ } | {
4753
+ /** Every condition for linking was met, repeatedly, and nothing was ever linked. */
4754
+ verdict: 'no-links-resolved';
4755
+ evidence: RemoteTrackLinkEvidence;
4756
+ } | {
4757
+ /** Never saw a call that could have been linked. **Not a pass.** */
4758
+ verdict: 'inconclusive';
4759
+ reason: string;
4760
+ }) & {
4761
+ startedAt: number;
4762
+ /** How many times the check genuinely ran — i.e. how much the verdict is worth. */
4763
+ checks: number;
4764
+ };
4765
+ type RemoteTrackResolverValidatorConfig = {
4766
+ /** Participants a call needs before it can plausibly have publisher↔subscriber links. Default `2`. */
4767
+ minClients: number;
4768
+ /** Inbound tracks that must be present in a call before it counts as a check. Default `2`. */
4769
+ minInboundTracks: number;
4770
+ /** Share of inbound tracks that must be linked to conclude the resolver works. Default `0.5`. */
4771
+ linkedRatioThreshold: number;
4772
+ /** Eligible calls to observe before concluding either way. One could be a race. Default `3`. */
4773
+ minChecks: number;
4774
+ };
4775
+ /**
4776
+ * Answers one question: **is the `RemoteTrackResolver` actually linking anything?**
4777
+ *
4778
+ * ### Why this is worth a validator
4779
+ *
4780
+ * Four things in this library are built on publisher↔subscriber links —
4781
+ * `IssueFanOutDetector`, `TrackDeliveryMismatchDetector`, `UnconsumedTrackDetector` and
4782
+ * `SimulcastReceiverValidator`. Every one of them checks `call.remoteTrackResolver` and, finding no
4783
+ * links, correctly does nothing rather than guessing.
4784
+ *
4785
+ * That is the right behaviour and it produces a nasty failure mode: a resolver wired to the wrong id
4786
+ * field, or a mediasoup `producerId` the application never attaches, leaves all four permanently
4787
+ * silent — and **silence is what a healthy deployment looks like too**. You would conclude your
4788
+ * calls were clean when in fact nothing was ever examined. This check exists to make that specific
4789
+ * mistake loud.
4790
+ *
4791
+ * ### `inconclusive` is not a pass
4792
+ *
4793
+ * A verdict is only reached from calls that *could* have been linked: a resolver configured, at
4794
+ * least `minClients` participants, and at least `minInboundTracks` inbound tracks present. A
4795
+ * one-to-one deployment, a lobby full of audio-only listeners, or a quiet period never presents
4796
+ * those conditions — and concluding "resolver works" from calls that had nothing to resolve would be
4797
+ * the very mistake this validator is here to catch. `checks` counts the eligible calls actually
4798
+ * seen; a validator cancelled before reaching `minChecks` finishes `inconclusive` and says so.
4799
+ *
4800
+ * ```ts
4801
+ * observer.on('validation-ready', ({ validator, report }) => {
4802
+ * if (validator !== 'remote-track-resolver' || !report.ready) return;
4803
+ * if (report.verdict === 'no-links-resolved') alert('resolver misconfigured — 4 detectors are inert');
4804
+ * });
4805
+ *
4806
+ * observer.addValidator('remote-track-resolver');
4807
+ * ```
4808
+ *
4809
+ * Run it once at start-up, and again after changing the resolver or the SFU's id scheme. Like every
4810
+ * validator it is one-shot: the answer is a property of the wiring, not of this moment.
4811
+ */
4812
+ declare class RemoteTrackResolverValidator implements Validator<RemoteTrackResolverReportPayload> {
4813
+ private readonly _observer;
4814
+ readonly onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void;
4815
+ static readonly NAME: "remote-track-resolver";
4816
+ readonly name: "remote-track-resolver";
4817
+ readonly startedAt: number;
4818
+ report: ValidationReport<RemoteTrackResolverReportPayload>;
4819
+ private readonly _config;
4820
+ /** Accumulated across every eligible call seen, so one small call cannot decide alone. */
4821
+ private _eligibleCalls;
4822
+ private _inboundTracks;
4823
+ private _linkedInboundTracks;
4824
+ private _exampleCallId?;
4825
+ private _checks;
4826
+ private _done;
4827
+ constructor(_observer: Observer, onDone: (report: ValidationReport<RemoteTrackResolverReportPayload>) => void, config?: Partial<RemoteTrackResolverValidatorConfig>);
4828
+ /** How many eligible calls were actually examined. `0` means nothing was established. */
4829
+ get checks(): number;
4830
+ cancel(reason?: string): void;
4831
+ update(): void;
4832
+ private _evidence;
4833
+ private _finish;
4834
+ }
4835
+
4836
+ /** Raised once if the deployment is not actually delivering the codec it thinks it is. */
4837
+ declare const CODEC_MISMATCH_ISSUE = "CODEC_INCONSISTENCY";
4838
+ /** What the check saw across a call's participants. */
4839
+ type CodecEvidence = {
4840
+ callId: string;
4841
+ kind: 'audio' | 'video';
4842
+ /** Every mime type in use in that call, most common first — e.g. `[ 'video/VP8', 'video/H264' ]`. */
4843
+ mimeTypes: string[];
4844
+ /** How many clients used each, in the same order as {@link mimeTypes}. */
4845
+ clientsPerMimeType: number[];
4846
+ /** Clients considered — those that reported at least one codec of this kind. */
4847
+ clients: number;
4848
+ /** The codec the check was told to expect, when it was given one. */
4849
+ expected?: string;
4850
+ };
4851
+ type CodecConsistencyReportPayload = ({
4852
+ /** One codec per media kind, and it is the expected one if an expectation was given. */
4853
+ verdict: 'codec-consistent';
4854
+ evidence: CodecEvidence[];
4855
+ } | {
4856
+ /** Participants of one call are split across different codecs. */
4857
+ verdict: 'codec-split';
4858
+ evidence: CodecEvidence[];
4859
+ } | {
4860
+ /** Consistent, but not what the deployment believes it negotiated. */
4861
+ verdict: 'unexpected-codec';
4862
+ evidence: CodecEvidence[];
4863
+ } | {
4864
+ /** Never saw a call with enough participants reporting codecs. **Not a pass.** */
4865
+ verdict: 'inconclusive';
4866
+ reason: string;
4867
+ }) & {
4868
+ startedAt: number;
4869
+ checks: number;
4870
+ };
4871
+ type CodecConsistencyValidatorConfig = {
4872
+ /**
4873
+ * The mime type you believe you are delivering, per kind — e.g.
4874
+ * `{ video: 'video/VP8', audio: 'audio/opus' }`.
4875
+ *
4876
+ * Optional. Without it the check still reports a *split* (participants disagreeing with each
4877
+ * other), which needs no expectation to be a fact. With it, the check can additionally catch the
4878
+ * case where everyone agrees on the wrong thing — a silent fallback that nothing else notices.
4879
+ */
4880
+ expected?: Partial<Record<'audio' | 'video', string>>;
4881
+ /** Which kinds to inspect. Default both. */
4882
+ kinds: ('audio' | 'video')[];
4883
+ /** Participants a call needs before disagreement is meaningful. Default `3`. */
4884
+ minClients: number;
4885
+ /** Calls to inspect before concluding. Default `3`. */
4886
+ minChecks: number;
4887
+ };
4888
+ /**
4889
+ * Answers: **is every participant of a call actually using the same codec — and is it the one you
4890
+ * think you negotiated?**
4891
+ *
4892
+ * ### Why the server has to answer this
4893
+ *
4894
+ * A client knows only its own codec. It cannot tell whether it is the odd one out, and an SFU that
4895
+ * forwards without transcoding cannot serve a call where participants disagree — so a split is a
4896
+ * real fault with a very confusing symptom: some pairs of participants see each other and some do
4897
+ * not, with no error anywhere. Only something holding every participant of a call at once can see
4898
+ * the split at all.
4899
+ *
4900
+ * The second half is the quieter failure. A deployment configured for VP9 or AV1 will fall back to
4901
+ * VP8 whenever one endpoint cannot negotiate the preferred codec, and nothing reports that — the
4902
+ * call works, the bitrate is higher than it should be, and the team believes it shipped AV1 months
4903
+ * ago. Give the check an `expected` mime type and it will say so.
4904
+ *
4905
+ * ### Why a validator and not a detector
4906
+ *
4907
+ * The answer is a property of the deployment — SDP munging, codec preferences, the SFU build — not
4908
+ * of this moment. A deployment that negotiates VP8 at 09:00 negotiates VP8 at 17:00. Re-deriving it
4909
+ * every tick would walk every codec of every peer connection of every call, forever, to re-learn a
4910
+ * constant. So it decides once and stops.
4911
+ *
4912
+ * Start it again after a deploy, or after changing codec preferences:
4913
+ *
4914
+ * ```ts
4915
+ * observer.addValidator('codec-consistency', {
4916
+ * expected: { video: 'video/VP9', audio: 'audio/opus' },
4917
+ * });
4918
+ *
4919
+ * observer.on('validation-ready', ({ validator, report }) => {
4920
+ * if (validator !== 'codec-consistency' || !report.ready) return;
4921
+ * // 'codec-consistent' | 'codec-split' | 'unexpected-codec' | 'inconclusive'
4922
+ * });
4923
+ * ```
4924
+ *
4925
+ * ### `inconclusive` is not a pass
4926
+ *
4927
+ * Only calls with at least `minClients` participants *reporting codecs of that kind* count as a
4928
+ * check. An audio-only deployment will never say anything about video, and concluding "video codecs
4929
+ * are consistent" from calls that carried no video would verify nothing.
4930
+ *
4931
+ * ### Comparison is by mime type only
4932
+ *
4933
+ * `sdpFmtpLine` carries profile and level — `profile-level-id` for H.264, `profile-id` for VP9 — and
4934
+ * two clients on the same mime type with different profiles are not truly interchangeable. That is
4935
+ * deliberately out of scope: fmtp differences are common, usually benign, and would make this check
4936
+ * noisy enough to ignore. It answers the coarse question, which is the one that is actually wrong in
4937
+ * practice.
4938
+ */
4939
+ declare class CodecConsistencyValidator implements Validator<CodecConsistencyReportPayload> {
4940
+ private readonly _observer;
4941
+ readonly onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void;
4942
+ static readonly NAME: "codec-consistency";
4943
+ readonly name: "codec-consistency";
4944
+ readonly startedAt: number;
4945
+ report: ValidationReport<CodecConsistencyReportPayload>;
4946
+ private readonly _config;
4947
+ private readonly _inspectedCallIds;
4948
+ private _evidence;
4949
+ private _checks;
4950
+ private _done;
4951
+ constructor(_observer: Observer, onDone: (report: ValidationReport<CodecConsistencyReportPayload>) => void, config?: Partial<CodecConsistencyValidatorConfig>);
4952
+ get checks(): number;
4953
+ cancel(reason?: string): void;
4954
+ update(): void;
4955
+ /** One evidence entry per configured kind that the call actually carried. */
4956
+ private _tallyOf;
4957
+ private _finish;
4958
+ }
4959
+
4960
+ /**
4961
+ * The validators `observer.addValidator(name, config)` knows how to build, and the config each takes.
4962
+ *
4963
+ * Adding one means: write the class with a `static readonly NAME`, add its entry here, and add a
4964
+ * `case` to `addValidator`. The map is what gives the call site its types —
4965
+ * `addValidator('simulcast-receivers', { … })` type-checks the config against the right validator,
4966
+ * and an unknown name won't compile.
4967
+ */
4968
+ type AvailableValidatorConfigs = {
4969
+ [SimulcastReceiverValidator.NAME]: SimulcastReceiverValidatorConfig;
4970
+ [RemoteTrackResolverValidator.NAME]: RemoteTrackResolverValidatorConfig;
4971
+ [CodecConsistencyValidator.NAME]: CodecConsistencyValidatorConfig;
4972
+ };
4973
+ /** A validator name that can be started. */
4974
+ type ValidatorName = keyof AvailableValidatorConfigs;
3089
4975
 
3090
4976
  type SampleRejectedReason = 'observer-closed' | 'missing-callId' | 'missing-clientId';
3091
4977
 
@@ -3107,9 +4993,6 @@ type AcceptMiddlewarePayload = {
3107
4993
  * `next(payload)` to continue the chain. Not calling `next` **drops** the sample.
3108
4994
  */
3109
4995
  type AcceptMiddleware = Middleware<AcceptMiddlewarePayload>;
3110
- type ObserverUpdateConfig = {
3111
- updatePolicy?: 'update-on-any-call-updated' | 'update-when-all-call-updated' | 'none';
3112
- };
3113
4996
  /** Produces the initial `appData` for a call created without an explicit `appData`. */
3114
4997
  type CallAppDataFactory = (params: {
3115
4998
  callId: string;
@@ -3120,289 +5003,191 @@ type ClientAppDataFactory = (params: {
3120
5003
  clientId: string;
3121
5004
  observedCall: ObservedCall;
3122
5005
  }) => Record<string, unknown>;
3123
- type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = ObserverUpdateConfig & {
3124
- defaultCallUpdatePolicy?: ObservedCallSettings['updatePolicy'];
5006
+ type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = {
3125
5007
  appData?: AppData;
3126
5008
  closeClientIfIdleForMs?: number;
3127
5009
  closeCallIfEmptyForMs?: number;
3128
5010
  /**
3129
- * Optional factory invoked when a call is created without an explicit `appData`
3130
- * (e.g. lazily by `accept()`), so apps can enrich appData without pre-creating the
3131
- * entity. `appData` is application-owned; it is never modified by the `accept()` context.
3132
- */
3133
- createCallAppData?: CallAppDataFactory;
3134
- /** Same as `createCallAppData`, for clients. Receives the (already-created) parent call. */
3135
- createClientAppData?: ClientAppDataFactory;
3136
- /**
3137
- * Optional factory invoked when a client is created, producing a per-client sink that
3138
- * receives every sample the client accepts (or `undefined` for no sink). The destination
3139
- * can be derived from `callId` / `clientId`.
3140
- */
3141
- createClientSink?: ClientSampleSinkFactory;
3142
- /**
3143
- * Optional factory invoked when a call is created, producing the call's `RemoteTrackResolver`
3144
- * (or `undefined` for none). Use the built-ins
3145
- * (`createDefaultMediasoupRemoteTrackResolverFactory()` / `createP2pRemoteTrackResolverFactory()`)
3146
- * or build a `RemoteTrackResolver` with custom publisher/subscriber id resolvers.
3147
- */
3148
- createRemoteTrackResolver?: RemoteTrackResolverFactory;
3149
- };
3150
- declare interface Observer {
3151
- on<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
3152
- off<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
3153
- once<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
3154
- emit<U extends keyof ObserverEvents>(event: U, ...args: ObserverEvents[U]): boolean;
3155
- }
3156
- declare class Observer<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
3157
- readonly config: ObserverConfig<AppData>;
3158
- readonly observedTURN: ObservedTURN;
3159
- readonly observedCalls: Map<string, ObservedCall<Record<string, unknown>>>;
3160
- readonly observedMediasoupRouters: Map<string, ObservedMediasoupRouter<Record<string, unknown>>>;
3161
- updater?: Updater;
3162
- /** Ancestry base shared by all Observer-bus events originating at the observer. */
3163
- readonly eventScope: ObserverEventBase;
3164
- closed: boolean;
3165
- totalAddedCall: number;
3166
- totalRemovedCall: number;
3167
- numberOfClientsUsingTurn: number;
3168
- numberOfClients: number;
3169
- numberOfInboundRtpStreams: number;
3170
- numberOfOutboundRtpStreams: number;
3171
- numberOfDataChannels: number;
3172
- numberOfPeerConnections: number;
3173
- /** Global, pre-dispatch middleware chain run on every accepted sample. */
3174
- readonly acceptMiddlewares: MiddlewareProcessor<AcceptMiddlewarePayload>;
3175
- /**
3176
- * Observer-scoped detector registry (ships empty), run on every `observer.update()` — the place
3177
- * for findings that span **calls**, e.g. "many calls on the same SFU degraded at once". Detectors
3178
- * raise findings with `observer.addIssue(...)`, surfaced on the bus as `observer-issue`.
3179
- * (For findings within a single call use `observedCall.detectors`.)
3180
- */
3181
- readonly detectors: Detectors;
3182
- constructor(config?: ObserverConfig<AppData>);
3183
- get numberOfCalls(): number;
3184
- get appData(): AppData | undefined;
3185
- getObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(callId: string): ObservedCall<T> | undefined;
3186
- createObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
3187
- getOrCreateObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
3188
- createObservedMediasoupRouter<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedMediasoupRouterSettings<T> & {
3189
- matchPeerConnectionByWebRtcTransportId?: boolean;
3190
- }): ObservedMediasoupRouter<Record<string, unknown>> | undefined;
3191
- close(): void;
3192
- accept(sample: ClientSample, context?: AcceptContext): void;
3193
- update(): void;
3194
- /**
3195
- * Raise an observer-scoped (cross-call / SFU-wide) finding. Emitted on the bus as
3196
- * `observer-issue`. Intended for `observer.detectors`, but the application may call it too.
5011
+ * When `true` (the default), every call update triggers an observer-wide `update()` pass.
5012
+ *
5013
+ * There is deliberately no timer and no separate policy object: a call is updated when any of its
5014
+ * clients is, and the observer is updated when any of its calls is — so the observer is updated
5015
+ * exactly when any client anywhere is. Set to `false` only if you drive `observer.update()`
5016
+ * yourself, and note that observer-scoped detectors and validators run *nowhere else*.
3197
5017
  */
3198
- addIssue(issue: ClientIssue): void;
3199
- /** Emit an Observer-bus event. */
3200
- private _notify;
3201
- }
3202
-
3203
- declare enum ClientEventTypes {
3204
- CLIENT_JOINED = "CLIENT_JOINED",
3205
- CLIENT_LEFT = "CLIENT_LEFT",
3206
- PEER_CONNECTION_OPENED = "PEER_CONNECTION_OPENED",
3207
- PEER_CONNECTION_CLOSED = "PEER_CONNECTION_CLOSED",
3208
- MEDIA_TRACK_ADDED = "MEDIA_TRACK_ADDED",
3209
- MEDIA_TRACK_REMOVED = "MEDIA_TRACK_REMOVED",
3210
- MEDIA_TRACK_RESUMED = "MEDIA_TRACK_RESUMED",
3211
- MEDIA_TRACK_MUTED = "MEDIA_TRACK_MUTED",
3212
- MEDIA_TRACK_UNMUTED = "MEDIA_TRACK_UNMUTED",
3213
- ICE_GATHERING_STATE_CHANGED = "ICE_GATHERING_STATE_CHANGED",
3214
- PEER_CONNECTION_STATE_CHANGED = "PEER_CONNECTION_STATE_CHANGED",
3215
- ICE_CONNECTION_STATE_CHANGED = "ICE_CONNECTION_STATE_CHANGED",
3216
- DATA_CHANNEL_OPEN = "DATA_CHANNEL_OPEN",
3217
- DATA_CHANNEL_CLOSED = "DATA_CHANNEL_CLOSED",
3218
- DATA_CHANNEL_ERROR = "DATA_CHANNEL_ERROR",
3219
- NEGOTIATION_NEEDED = "NEGOTIATION_NEEDED",
3220
- SIGNALING_STATE_CHANGE = "SIGNALING_STATE_CHANGE",
3221
- ICE_CANDIDATE = "ICE_CANDIDATE",
3222
- ICE_CANDIDATE_ERROR = "ICE_CANDIDATE_ERROR",
3223
- PRODUCER_ADDED = "PRODUCER_ADDED",
3224
- PRODUCER_REMOVED = "PRODUCER_REMOVED",
3225
- PRODUCER_PAUSED = "PRODUCER_PAUSED",
3226
- PRODUCER_RESUMED = "PRODUCER_RESUMED",
3227
- CONSUMER_ADDED = "CONSUMER_ADDED",
3228
- CONSUMER_REMOVED = "CONSUMER_REMOVED",
3229
- CONSUMER_PAUSED = "CONSUMER_PAUSED",
3230
- CONSUMER_RESUMED = "CONSUMER_RESUMED",
3231
- DATA_PRODUCER_CREATED = "DATA_PRODUCER_CREATED",
3232
- DATA_PRODUCER_CLOSED = "DATA_PRODUCER_CLOSED",
3233
- DATA_CONSUMER_CREATED = "DATA_CONSUMER_CREATED",
3234
- DATA_CONSUMER_CLOSED = "DATA_CONSUMER_CLOSED"
3235
- }
3236
-
3237
- /**
3238
- * Small statistics helpers used by the aggregators and detectors.
3239
- *
3240
- * Rationale: a mean is a poor summary for call telemetry — one participant with a 1500 ms RTT
3241
- * skews the average for nine healthy ones. Detectors should reason with medians, high percentiles
3242
- * and "affected ratios" instead.
3243
- */
3244
- /** A distribution summary of a numeric sample set. */
3245
- type StatsSummary = {
3246
- count: number;
3247
- min: number;
3248
- max: number;
3249
- mean: number;
3250
- median: number;
3251
- p75: number;
3252
- p95: number;
3253
- };
3254
- /**
3255
- * The p-th percentile (0..1) using linear interpolation between closest ranks.
3256
- * Returns `undefined` for an empty input.
3257
- */
3258
- declare function percentile(values: number[], p: number): number | undefined;
3259
- /** The median (50th percentile). `undefined` for an empty input. */
3260
- declare function median(values: number[]): number | undefined;
3261
- /** Summarize a numeric sample set. Returns `undefined` for an empty input. */
3262
- declare function summarize(values: number[]): StatsSummary | undefined;
3263
- /**
3264
- * A counter-reset-safe delta: the increase of a cumulative counter between two observations.
3265
- *
3266
- * Returns `0` when the counter went backwards (reset / SSRC reuse) or when either side is missing.
3267
- * NOTE the guard is `>=` on **defined** values — a previous value of `0` is a perfectly valid
3268
- * baseline, so `0 -> 5` correctly yields `5` (using a truthiness check here silently drops the
3269
- * first interval of every counter, which is exactly when the first loss/freeze event happens).
3270
- */
3271
- declare function counterDelta(previous: number | undefined, current: number | undefined): number;
3272
- /** An entry retained by {@link SlidingWindow}. */
3273
- type SlidingWindowEntry<T> = {
3274
- timestamp: number;
3275
- value: T;
3276
- };
3277
- /**
3278
- * A time-bounded ring buffer used by detectors that reason over a window ("N of M clients degraded
3279
- * within 10 s"). Entries older than `windowMs` are evicted on write and on read.
3280
- */
3281
- declare class SlidingWindow<T> {
3282
- readonly windowMs: number;
3283
- /** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
3284
- readonly maxEntries: number;
3285
- private _entries;
3286
- constructor(windowMs: number,
3287
- /** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
3288
- maxEntries?: number);
3289
- /** Add an entry (defaults to `Date.now()`), then evict anything outside the window. */
3290
- add(value: T, timestamp?: number): void;
3291
- /** The entries still inside the window, oldest first. */
3292
- entries(now?: number): SlidingWindowEntry<T>[];
3293
- /** The values still inside the window, oldest first. */
3294
- values(now?: number): T[];
3295
- get size(): number;
3296
- clear(): void;
3297
- private _evict;
3298
- }
3299
-
3300
- /**
3301
- * Thresholds deciding when a single receiver counts as "degraded". A receiver is degraded when it
3302
- * trips **any** of these in the current tick.
3303
- */
3304
- type ReceiverHealthThresholds = {
3305
- /** Fraction of packets lost in the tick (0..1). */
3306
- fractionLost: number;
3307
- /** Freezes observed in the tick. */
3308
- freezeCount: number;
3309
- /** Fraction of received frames dropped before rendering (0..1). */
3310
- framesDroppedRatio: number;
3311
- /** Fraction of received audio samples concealed (0..1). */
3312
- concealmentRatio: number;
3313
- /** Mean jitter-buffer delay of the tick (ms). */
3314
- jitterBufferDelayInMs: number;
3315
- /** Round-trip time reported for the receiving peer connection (ms). */
3316
- rttInMs: number;
3317
- };
3318
- declare const defaultReceiverHealthThresholds: ReceiverHealthThresholds;
3319
- /** The per-receiver view the aggregator builds for one subscribed track. */
3320
- type ReceiverDistributionEntry = {
3321
- observedInboundTrack: ObservedInboundTrack;
3322
- clientId: string;
3323
- peerConnectionId: string;
3324
- degraded: boolean;
3325
- /** Why it was marked degraded (empty when healthy). */
3326
- reasons: string[];
3327
- bitrate: number;
3328
- fractionLost?: number;
3329
- jitter?: number;
3330
- rttInMs?: number;
3331
- jitterBufferDelayInMs?: number;
3332
- concealmentRatio?: number;
3333
- framesDroppedRatio?: number;
3334
- deltaFreezeCount: number;
3335
- deltaPliCount: number;
3336
- deltaNackCount: number;
3337
- deltaPacketsReceived: number;
3338
- };
3339
- /** The publisher side of the distribution. */
3340
- type PublisherDistributionEntry = {
3341
- observedOutboundTrack: ObservedOutboundTrack;
3342
- clientId: string;
3343
- peerConnectionId: string;
3344
- /** `true` when the publisher's own egress looks fine (so degradation is downstream). */
3345
- healthy: boolean;
3346
- reasons: string[];
3347
- bitrate: number;
3348
- /** Loss reported back by the SFU/remote via RTCP (0..1). */
3349
- remoteFractionLost?: number;
3350
- remoteRttInMs?: number;
3351
- qualityLimitationReason?: string;
3352
- deltaPacketsSent: number;
3353
- };
3354
- /**
3355
- * One published track and everything observed about how it was delivered to its subscribers.
3356
- * This is the primitive most cross-client detectors are built on.
3357
- */
3358
- type ObservedTrackDistribution = {
3359
- trackId: string;
3360
- kind: string;
3361
- publisher: PublisherDistributionEntry;
3362
- receivers: ReceiverDistributionEntry[];
3363
- numberOfReceivers: number;
3364
- numberOfHealthyReceivers: number;
3365
- numberOfDegradedReceivers: number;
3366
- /** degradedReceivers / receivers (0..1); `0` when there are no receivers. */
3367
- degradedRatio: number;
3368
- /** Distribution summaries across receivers (undefined when no receiver reported the metric). */
3369
- bitrate?: StatsSummary;
3370
- fractionLost?: StatsSummary;
3371
- jitter?: StatsSummary;
3372
- rttInMs?: StatsSummary;
3373
- jitterBufferDelayInMs?: StatsSummary;
3374
- concealmentRatio?: StatsSummary;
3375
- /** Fan-out counters: how many receivers saw the symptom, and the total across them. */
3376
- freezes: {
3377
- affectedReceivers: number;
3378
- total: number;
3379
- };
3380
- plis: {
3381
- affectedReceivers: number;
3382
- total: number;
5018
+ autoUpdateOnCallUpdate?: boolean;
5019
+ inboundTrackDegradationThresholds?: {
5020
+ deltaFreezeCount: number;
5021
+ framesDroppedRatio: number;
5022
+ jitterBufferDelayInMs: number;
5023
+ concealmentRatio: number;
5024
+ rttInMs: number;
3383
5025
  };
3384
- concealment: {
3385
- affectedReceivers: number;
5026
+ outboundTrackDegradationThresholds?: {
5027
+ fractionLost: number;
5028
+ rttInMs: number;
3386
5029
  };
5030
+ /**
5031
+ * Optional factory invoked when a call is created without an explicit `appData`
5032
+ * (e.g. lazily by `accept()`), so apps can enrich appData without pre-creating the
5033
+ * entity. `appData` is application-owned; it is never modified by the `accept()` context.
5034
+ */
5035
+ createCallAppData?: CallAppDataFactory;
5036
+ /** Same as `createCallAppData`, for clients. Receives the (already-created) parent call. */
5037
+ createClientAppData?: ClientAppDataFactory;
5038
+ /**
5039
+ * Optional factory invoked when a client is created, producing a per-client sink that
5040
+ * receives every sample the client accepts (or `undefined` for no sink). The destination
5041
+ * can be derived from `callId` / `clientId`.
5042
+ */
5043
+ createClientSink?: ClientSampleSinkFactory;
5044
+ /**
5045
+ * Optional factory invoked when a call is created, producing the call's `RemoteTrackResolver`
5046
+ * (or `undefined` for none). Use the built-ins
5047
+ * (`createDefaultMediasoupRemoteTrackResolverFactory()` / `createP2pRemoteTrackResolverFactory()`)
5048
+ * or build a `RemoteTrackResolver` with custom publisher/subscriber id resolvers.
5049
+ */
5050
+ createRemoteTrackResolver?: RemoteTrackResolverFactory;
3387
5051
  };
3388
- /**
3389
- * Builds {@link ObservedTrackDistribution}s by walking
3390
- * `ObservedOutboundTrack.remoteInboundTracks` — the publisher → subscribers links maintained by a
3391
- * `RemoteTrackResolver`. Without a configured resolver there are no links and the aggregator
3392
- * yields nothing.
3393
- *
3394
- * It is stateless per call: build it once and call `aggregate()` on each `call.update()`.
3395
- */
3396
- declare class TrackDistributionAggregator {
3397
- private readonly _call;
3398
- readonly thresholds: ReceiverHealthThresholds;
3399
- constructor(_call: ObservedCall, thresholds?: ReceiverHealthThresholds);
3400
- /** Aggregate every published track in the call that currently has linked subscribers. */
3401
- aggregate(): ObservedTrackDistribution[];
3402
- /** Aggregate a single published track, or `undefined` when it has no linked subscribers. */
3403
- aggregateTrack(outboundTrack: ObservedOutboundTrack): ObservedTrackDistribution | undefined;
3404
- private _publisherEntry;
3405
- private _receiverEntry;
5052
+ declare interface Observer {
5053
+ on<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
5054
+ off<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
5055
+ once<U extends keyof ObserverEvents>(event: U, listener: (...args: ObserverEvents[U]) => void): this;
5056
+ emit<U extends keyof ObserverEvents>(event: U, ...args: ObserverEvents[U]): boolean;
5057
+ }
5058
+ declare class Observer<AppData extends Record<string, unknown> = Record<string, unknown>> extends EventEmitter {
5059
+ readonly observedTURN: ObservedTURN;
5060
+ readonly observedCalls: Map<string, ObservedCall<Record<string, unknown>>>;
5061
+ readonly observedMediasoupRouters: Map<string, ObservedMediasoupRouter<Record<string, unknown>>>;
5062
+ /** Ancestry base shared by all Observer-bus events originating at the observer. */
5063
+ readonly eventScope: ObserverEventBase;
5064
+ readonly config: ObserverConfig<AppData>;
5065
+ closed: boolean;
5066
+ totalAddedCall: number;
5067
+ totalRemovedCall: number;
5068
+ numberOfClientsUsingTurn: number;
5069
+ numberOfClients: number;
5070
+ numberOfInboundRtpStreams: number;
5071
+ numberOfOutboundRtpStreams: number;
5072
+ numberOfDataChannels: number;
5073
+ numberOfPeerConnections: number;
5074
+ /** Global, pre-dispatch middleware chain run on every accepted sample. */
5075
+ readonly acceptMiddlewares: MiddlewareProcessor<AcceptMiddlewarePayload>;
5076
+ /**
5077
+ * Fleet-wide index of every open client issue, maintained incrementally as issues open and close.
5078
+ * Each call's index propagates into this one, so cross-call queries cost O(matching issues) rather
5079
+ * than a walk over every call and client. This is what observer-scoped detectors read.
5080
+ */
5081
+ readonly activeIssuesRegistry: ActiveIssuesRegistry;
5082
+ /**
5083
+ * Validators currently running. Each removes itself when it finishes, so this is normally empty —
5084
+ * a validator is a one-shot check, not a permanent fixture. Start one with {@link addValidator}.
5085
+ */
5086
+ readonly validators: Set<RunningValidator>;
5087
+ /**
5088
+ * Observer-scoped detector registry, run on every `observer.update()` — the place for findings
5089
+ * that span **calls**, e.g. "many calls on the same SFU degraded at once". Detectors raise
5090
+ * findings with `observer.addIssue(...)`, surfaced on the bus as `observer-issue`.
5091
+ * (For findings within a single call use `observedCall.detectors`.)
5092
+ *
5093
+ * Starts **empty**. Populate it with {@link addObserverDetector}, or `detectors.add(...)` for an
5094
+ * instance you built yourself.
5095
+ */
5096
+ readonly detectors: Detectors;
5097
+ /**
5098
+ * The call-scoped detectors to build on **every** call this observer creates, in registration
5099
+ * order. Written by {@link addCallDetector}; read by `createObservedCall`.
5100
+ *
5101
+ * Nothing is created implicitly. There is no detector configuration in `ObserverConfig` and no
5102
+ * default set, because a detector that nobody asked for is a detector nobody will act on: it costs
5103
+ * time on every tick and raises findings into a handler that was not written to expect them. An
5104
+ * application says what it wants to watch, or it watches nothing.
5105
+ *
5106
+ * ```ts
5107
+ * observer.addCallDetector('call-concurrent-issue-detector', {
5108
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
5109
+ * });
5110
+ * ```
5111
+ */
5112
+ readonly callDetectorConfigs: Map<keyof AvailableCallScopeDetectorsConfigs, Partial<UnconsumedTrackDetectorConfig | TrackDeliveryMismatchDetectorConfig | CallConcurrentIssueDetectorConfig | IssueFanOutDetectorConfig | PublisherFaultCorroborationDetectorConfig>>;
5113
+ constructor(config?: Partial<ObserverConfig<AppData>>);
5114
+ get numberOfCalls(): number;
5115
+ get appData(): AppData | undefined;
5116
+ addObserverDetector<K extends keyof AvailableObserverScopeDetectorsConfigs>(name: K, config?: Partial<AvailableObserverScopeDetectorsConfigs[K]>): this;
5117
+ /**
5118
+ * Enable a call-scoped detector for calls created **from now on**.
5119
+ *
5120
+ * This edits the config, not the live calls: calls already open keep the detector set they were
5121
+ * built with. To add one to an existing call, use `observedCall.addDetector(...)` directly.
5122
+ */
5123
+ addCallDetector<K extends keyof AvailableCallScopeDetectorsConfigs>(name: K, config?: Partial<AvailableCallScopeDetectorsConfigs[K]>): this;
5124
+ /** Stop building `name` on calls created from now on. Calls already open are untouched. */
5125
+ removeCallDetector(name: keyof AvailableCallScopeDetectorsConfigs): this;
5126
+ /**
5127
+ * Start a structural check. It runs on each `observer.update()` until it can decide, reports once
5128
+ * on `validation-ready`, and removes itself.
5129
+ *
5130
+ * ```ts
5131
+ * observer.validate('simulcast-receiver-validator', { minChecks: 5 });
5132
+ * ```
5133
+ *
5134
+ * Config keys are optional and merged over that validator's defaults. Call it again — after a
5135
+ * deploy, say — to check again; there is no revalidation timer, because a deploy rather than
5136
+ * elapsed time is what makes a structural verdict stale.
5137
+ */
5138
+ addValidator<K extends keyof AvailableValidatorConfigs>(name: K, config?: Partial<AvailableValidatorConfigs[K]>): this;
5139
+ getObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(callId: string): ObservedCall<T> | undefined;
5140
+ createObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
5141
+ getOrCreateObservedCall<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedCallSettings<T>): ObservedCall<T> | undefined;
5142
+ createObservedMediasoupRouter<T extends Record<string, unknown> = Record<string, unknown>>(settings: ObservedMediasoupRouterSettings<T> & {
5143
+ matchPeerConnectionByWebRtcTransportId?: boolean;
5144
+ }): ObservedMediasoupRouter<Record<string, unknown>> | undefined;
5145
+ close(): void;
5146
+ accept(sample: ClientSample, context?: AcceptContext): void;
5147
+ update(): void;
5148
+ /**
5149
+ * Raise an observer-scoped (cross-call / SFU-wide) finding. Emitted on the bus as
5150
+ * `observer-issue`. Intended for `observer.detectors`, but the application may call it too.
5151
+ *
5152
+ * `payload` takes an **object**; see `ObserverIssue`.
5153
+ */
5154
+ addIssue(issue: ObserverIssue): void;
5155
+ /** Emit an Observer-bus event. */
5156
+ private _notify;
5157
+ }
5158
+
5159
+ declare enum ClientEventTypes {
5160
+ CLIENT_JOINED = "CLIENT_JOINED",
5161
+ CLIENT_LEFT = "CLIENT_LEFT",
5162
+ PEER_CONNECTION_OPENED = "PEER_CONNECTION_OPENED",
5163
+ PEER_CONNECTION_CLOSED = "PEER_CONNECTION_CLOSED",
5164
+ MEDIA_TRACK_ADDED = "MEDIA_TRACK_ADDED",
5165
+ MEDIA_TRACK_REMOVED = "MEDIA_TRACK_REMOVED",
5166
+ MEDIA_TRACK_RESUMED = "MEDIA_TRACK_RESUMED",
5167
+ MEDIA_TRACK_MUTED = "MEDIA_TRACK_MUTED",
5168
+ MEDIA_TRACK_UNMUTED = "MEDIA_TRACK_UNMUTED",
5169
+ ICE_GATHERING_STATE_CHANGED = "ICE_GATHERING_STATE_CHANGED",
5170
+ PEER_CONNECTION_STATE_CHANGED = "PEER_CONNECTION_STATE_CHANGED",
5171
+ ICE_CONNECTION_STATE_CHANGED = "ICE_CONNECTION_STATE_CHANGED",
5172
+ DATA_CHANNEL_OPEN = "DATA_CHANNEL_OPEN",
5173
+ DATA_CHANNEL_CLOSED = "DATA_CHANNEL_CLOSED",
5174
+ DATA_CHANNEL_ERROR = "DATA_CHANNEL_ERROR",
5175
+ NEGOTIATION_NEEDED = "NEGOTIATION_NEEDED",
5176
+ SIGNALING_STATE_CHANGE = "SIGNALING_STATE_CHANGE",
5177
+ ICE_CANDIDATE = "ICE_CANDIDATE",
5178
+ ICE_CANDIDATE_ERROR = "ICE_CANDIDATE_ERROR",
5179
+ PRODUCER_ADDED = "PRODUCER_ADDED",
5180
+ PRODUCER_REMOVED = "PRODUCER_REMOVED",
5181
+ PRODUCER_PAUSED = "PRODUCER_PAUSED",
5182
+ PRODUCER_RESUMED = "PRODUCER_RESUMED",
5183
+ CONSUMER_ADDED = "CONSUMER_ADDED",
5184
+ CONSUMER_REMOVED = "CONSUMER_REMOVED",
5185
+ CONSUMER_PAUSED = "CONSUMER_PAUSED",
5186
+ CONSUMER_RESUMED = "CONSUMER_RESUMED",
5187
+ DATA_PRODUCER_CREATED = "DATA_PRODUCER_CREATED",
5188
+ DATA_PRODUCER_CLOSED = "DATA_PRODUCER_CLOSED",
5189
+ DATA_CONSUMER_CREATED = "DATA_CONSUMER_CREATED",
5190
+ DATA_CONSUMER_CLOSED = "DATA_CONSUMER_CLOSED"
3406
5191
  }
3407
5192
 
3408
5193
  /** Thresholds deciding when a client counts as degraded on the receiving / sending side. */
@@ -3456,687 +5241,314 @@ type CallHealth = {
3456
5241
  inboundFractionLost?: StatsSummary;
3457
5242
  concealmentRatio?: StatsSummary;
3458
5243
  /** How many clients reported each quality-limitation reason on their outbound video. */
3459
- qualityLimitation: {
3460
- cpu: number;
3461
- bandwidth: number;
3462
- other: number;
3463
- };
3464
- freezes: {
3465
- affectedClients: number;
3466
- total: number;
3467
- };
3468
- };
3469
- /**
3470
- * Aggregates a call along the **client** axis (as `TrackDistributionAggregator` does along the
3471
- * publisher→subscriber axis): per-client health split into sending vs receiving, plus percentile
3472
- * rollups and "affected ratio" counts for the whole call.
3473
- *
3474
- * Build once per call and call `aggregate()` on each `call.update()`.
3475
- */
3476
- declare class CallHealthAggregator {
3477
- private readonly _call;
3478
- readonly thresholds: ClientHealthThresholds;
3479
- constructor(_call: ObservedCall, thresholds?: ClientHealthThresholds);
3480
- aggregate(): CallHealth;
3481
- private _clientHealth;
3482
- }
3483
-
3484
- /** The finding types this detector raises (as `call-issue.type`). */
3485
- declare const CommonSourceDegradationTypes: {
3486
- /** The publisher's egress looks fine, yet most subscribers are degraded → downstream/SFU suspected. */
3487
- readonly publisherHealthySubscribersDegraded: "PUBLISHER_HEALTHY_SUBSCRIBERS_DEGRADED";
3488
- /** The publisher itself is impaired and every subscriber sees it → source-side problem. */
3489
- readonly publisherDegradedForAllSubscribers: "PUBLISHER_DEGRADED_FOR_ALL_SUBSCRIBERS";
3490
- /** Exactly one subscriber is degraded while the rest are fine → that receiver's own problem. */
3491
- readonly singleSubscriberDegraded: "SINGLE_SUBSCRIBER_DEGRADED";
3492
- /** Several (but not most) subscribers degraded on the same source. */
3493
- readonly multipleSubscribersDegraded: "MULTIPLE_SUBSCRIBERS_DEGRADED";
3494
- };
3495
- type CommonSourceDegradationDetectorConfig = {
3496
- /** Minimum subscribers before a ratio is meaningful. Default `3`. */
3497
- minReceivers: number;
3498
- /** degradedRatio at/above which the problem is treated as common to the source. Default `0.6`. */
3499
- degradedRatioThreshold: number;
3500
- /** Per-receiver health thresholds (forwarded to the aggregator). */
3501
- thresholds?: Partial<ReceiverHealthThresholds>;
3502
- /**
3503
- * Consecutive ticks a condition must hold before an issue is raised, to avoid flapping on a
3504
- * single bad sample. Default `2`.
3505
- */
3506
- consecutiveTicks: number;
3507
- };
3508
- /**
3509
- * Compares a published track against **all** of its subscribers and decides where the fault lies.
3510
- *
3511
- * This is the detector that only a server-side observer can run: a single browser cannot know
3512
- * whether the other participants receiving the same source see the same thing. Requires a
3513
- * `RemoteTrackResolver` (`ObserverConfig.createTrackResolver`) — without publisher↔subscriber links
3514
- * there is nothing to compare and the detector stays silent.
3515
- */
3516
- declare class CommonSourceDegradationDetector implements Detector {
3517
- private readonly _call;
3518
- readonly name = "common-source-degradation-detector";
3519
- private readonly _config;
3520
- private readonly _aggregator;
3521
- /** trackId -> consecutive ticks the same finding held. */
3522
- private readonly _streaks;
3523
- /** The distributions computed on the most recent `update()` (handy for dashboards). */
3524
- lastDistributions: ObservedTrackDistribution[];
3525
- constructor(_call: ObservedCall, config?: Partial<CommonSourceDegradationDetectorConfig>);
3526
- update(): void;
3527
- private _classify;
3528
- private _payload;
3529
- }
3530
-
3531
- declare const CallWideDegradationTypes: {
3532
- /** Most participants degraded at once → shared cause (call/SFU/network), not individual endpoints. */
3533
- readonly callWideQualityDegradation: "CALL_WIDE_QUALITY_DEGRADATION";
3534
- /** Most participants' **receiving** side degraded → downstream/egress suspected. */
3535
- readonly callWideInboundDegradation: "CALL_WIDE_INBOUND_DEGRADATION";
3536
- /** Most participants' **sending** side degraded → ingress suspected. */
3537
- readonly callWideOutboundDegradation: "CALL_WIDE_OUTBOUND_DEGRADATION";
3538
- };
3539
- type CallWideDegradationDetectorConfig = {
3540
- /** Minimum participants before a ratio means anything. Default `3`. */
3541
- minClients: number;
3542
- /** Affected-client ratio at/above which the call is considered call-wide degraded. Default `0.5`. */
3543
- degradedRatioThreshold: number;
3544
- /** Per-client health thresholds. */
3545
- thresholds?: Partial<ClientHealthThresholds>;
3546
- /** Consecutive ticks the condition must hold before raising. Default `2`. */
3547
- consecutiveTicks: number;
3548
- };
3549
- /**
3550
- * Raises a call-level finding when a **majority of participants** are degraded in the same window —
3551
- * the signal that something shared is wrong rather than one person's Wi-Fi.
3552
- *
3553
- * It also splits by direction, because that is what makes the finding actionable: if most clients'
3554
- * receiving side is bad the suspicion is egress/downstream; if most clients' sending side is bad it
3555
- * points at ingress. Reported with medians/percentiles, never means, so a single 1500 ms outlier
3556
- * can't manufacture (or mask) a call-wide alert.
3557
- */
3558
- declare class CallWideDegradationDetector implements Detector {
3559
- private readonly _call;
3560
- readonly name = "call-wide-degradation-detector";
3561
- private readonly _config;
3562
- private readonly _aggregator;
3563
- private _streak?;
3564
- /** The health rollup computed on the most recent `update()`. */
3565
- lastHealth?: CallHealth;
3566
- constructor(_call: ObservedCall, config?: Partial<CallWideDegradationDetectorConfig>);
3567
- update(): void;
3568
- private _classify;
3569
- private _payload;
3570
- }
3571
-
3572
- declare const PliAndFreezeFanOutTypes: {
3573
- /** Many receivers of the same publisher requested keyframes at once. */
3574
- readonly publisherPliStorm: "PUBLISHER_PLI_STORM";
3575
- /** Many receivers of the same publisher froze at once. */
3576
- readonly publishedVideoFrozenForMultipleReceivers: "PUBLISHED_VIDEO_FROZEN_FOR_MULTIPLE_RECEIVERS";
3577
- };
3578
- type PliAndFreezeFanOutDetectorConfig = {
3579
- /** Minimum receivers of the track before a fan-out ratio is meaningful. Default `3`. */
3580
- minReceivers: number;
3581
- /** Fraction of receivers that must be affected within the window. Default `0.5`. */
3582
- affectedRatioThreshold: number;
3583
- /** The correlation window (ms) symptoms are counted over. Default `10_000`. */
3584
- windowMs: number;
3585
- /** Minimum PLIs across receivers inside the window before a storm is declared. Default `5`. */
3586
- minPliCount: number;
3587
- /** Re-arm time (ms) before the same finding can be raised again for a track. Default `30_000`. */
3588
- cooldownMs: number;
3589
- thresholds?: Partial<ReceiverHealthThresholds>;
3590
- };
3591
- /**
3592
- * Detects **fan-out** symptoms: one publisher's stream causing many receivers to request keyframes
3593
- * (PLI) or to freeze within the same window.
3594
- *
3595
- * A single receiver sending PLIs is unremarkable — it lost some packets. Nineteen of twenty
3596
- * receivers of the *same* source doing it inside ten seconds is not twenty coincidental network
3597
- * faults; it points at the source's output, the SFU's forwarding, or a keyframe/burst problem
3598
- * upstream. That distinction requires seeing every subscriber of a track at once, which is why this
3599
- * belongs server-side.
3600
- *
3601
- * Symptoms are accumulated in a sliding window rather than judged per tick, because a burst is
3602
- * spread over a few samples.
3603
- */
3604
- declare class PliAndFreezeFanOutDetector implements Detector {
3605
- private readonly _call;
3606
- readonly name = "pli-and-freeze-fan-out-detector";
3607
- private readonly _config;
3608
- private readonly _aggregator;
3609
- private readonly _windows;
3610
- constructor(_call: ObservedCall, config?: Partial<PliAndFreezeFanOutDetectorConfig>);
3611
- update(): void;
3612
- close(): void;
3613
- private _evaluate;
3614
- private _windowOf;
3615
- }
3616
-
3617
- declare const AudioImpairmentFanOutTypes: {
3618
- /** Most receivers of one microphone are concealing audio → the problem follows that source. */
3619
- readonly publishedAudioDegradedForMajority: "PUBLISHED_AUDIO_DEGRADED_FOR_MAJORITY";
3620
- /** Receivers across the call are under jitter-buffer pressure → shared delivery problem. */
3621
- readonly callWideAudioJitterBufferStress: "CALL_WIDE_AUDIO_JITTER_BUFFER_STRESS";
3622
- };
3623
- type AudioImpairmentFanOutDetectorConfig = {
3624
- /** Minimum receivers of a track before a ratio is meaningful. Default `3`. */
3625
- minReceivers: number;
3626
- /** Fraction of receivers that must be impaired. Default `0.6`. */
3627
- affectedRatioThreshold: number;
3628
- /** Jitter-buffer delay (ms) above which a receiver counts as stressed. Default `500`. */
3629
- jitterBufferDelayInMs: number;
3630
- /** Minimum audio receivers in the call before the call-wide check runs. Default `5`. */
3631
- minCallReceivers: number;
3632
- /** Consecutive ticks the condition must hold before raising. Default `2`. */
3633
- consecutiveTicks: number;
3634
- thresholds?: Partial<ReceiverHealthThresholds>;
3635
- };
3636
- /**
3637
- * Audio-side fan-out analysis, using the concealment / jitter-buffer metrics WebRTC exposes on the
3638
- * receiver (NetEQ's own account of how hard it is working to keep audio smooth).
3639
- *
3640
- * Two questions a browser can't answer:
3641
- *
3642
- * 1. *Does the impairment follow the source?* If every receiver of Alice's microphone is concealing
3643
- * ~20% of samples while every receiver of Bob's is at ~0.1%, the fault is on Alice's path, not on
3644
- * the receivers.
3645
- * 2. *Is the whole call's audio delivery under pressure?* One receiver with a huge jitter buffer is
3646
- * their own network; fifteen of twenty at once is shared.
3647
- */
3648
- declare class AudioImpairmentFanOutDetector implements Detector {
3649
- private readonly _call;
3650
- readonly name = "audio-impairment-fan-out-detector";
3651
- private readonly _config;
3652
- private readonly _aggregator;
3653
- private readonly _trackStreaks;
3654
- private _callStreak;
3655
- constructor(_call: ObservedCall, config?: Partial<AudioImpairmentFanOutDetectorConfig>);
3656
- update(): void;
3657
- close(): void;
3658
- private _evaluateCallWide;
3659
- }
3660
-
3661
- declare const IceDisruptionTypes: {
3662
- /** Many participants lost/failed their ICE connection inside the same window. */
3663
- readonly callIceDisruption: "CALL_ICE_DISRUPTION";
3664
- };
3665
- type IceDisruptionDetectorConfig = {
3666
- /** Minimum participants before a ratio is meaningful. Default `3`. */
3667
- minClients: number;
3668
- /** Fraction of participants that must be disrupted within the window. Default `0.5`. */
3669
- affectedRatioThreshold: number;
3670
- /** Correlation window (ms). Default `10_000`. */
3671
- windowMs: number;
3672
- /** Re-arm time (ms) before raising again. Default `60_000`. */
3673
- cooldownMs: number;
3674
- };
3675
- /**
3676
- * Detects an **ICE disruption storm**: many participants of a call losing connectivity inside the
3677
- * same short window.
3678
- *
3679
- * One client losing ICE is routine (they walked out of Wi-Fi range). Twenty-eight of thirty-five
3680
- * doing it within five seconds is an infrastructure event, and that conclusion is only reachable by
3681
- * correlating clients — which is the whole point of doing it here rather than in the browser.
3682
- *
3683
- * ### Prefer the issue-driven path
3684
- *
3685
- * This detector works from **raw ICE state transitions**, which is the fallback for clients that do
3686
- * not report issues. If your clients run client-monitor-js >= 4.6.0, prefer
3687
- * `ConcurrentIssueDetector` configured with the ICE issue types instead:
3688
- *
3689
- * ```ts
3690
- * new ConcurrentIssueDetector(observedCall, {
3691
- * issueTypes: [ 'ice-disconnected', 'ice-connection-failed', 'ice-transport-stalled', 'unstable-ice-path' ],
3692
- * });
3693
- * ```
3694
- *
3695
- * The client's own detector is better at deciding *whether* a transport is really disrupted: it
3696
- * raises `ice-disconnected` only once `disconnected` has persisted past a threshold, so the transient
3697
- * blips ICE routinely heals on its own never produce an issue at all, and it distinguishes a terminal
3698
- * `failed` from a stalled transport from an unstable path. This detector cannot make those
3699
- * distinctions — a raw `disconnected` that recovers in 200 ms looks identical to one that never does.
3700
- * It also gains resolution intervals, so "they all recovered together" becomes observable.
3701
- *
3702
- * ### Why this one subscribes to the bus
3703
- *
3704
- * ICE transitions are discrete events that can occur and revert between two `update()` ticks;
3705
- * polling `iceConnectionState` per tick would miss short flaps. It therefore implements `close()`
3706
- * to drop its listeners (called automatically when the call closes or the detector is removed).
3707
- */
3708
- declare class IceDisruptionDetector implements Detector {
3709
- private readonly _call;
3710
- readonly name = "ice-disruption-detector";
3711
- private readonly _config;
3712
- /** clientId -> the last time that client was seen disrupted, inside the window. */
3713
- private readonly _disruptions;
3714
- private _lastRaisedAt;
3715
- private readonly _onIceStateChanged;
3716
- private readonly _onConnectionStateChanged;
3717
- constructor(_call: ObservedCall, config?: Partial<IceDisruptionDetectorConfig>);
3718
- update(): void;
3719
- close(): void;
3720
- }
3721
-
3722
- declare const TurnServerHealthTypes: {
3723
- /** One TURN server's clients are degraded while other servers' clients are fine. */
3724
- readonly turnServerDegraded: "TURN_SERVER_DEGRADED";
3725
- };
3726
- type TurnServerHealthDetectorConfig = {
3727
- /** Minimum clients on a server before a ratio is meaningful. Default `5`. */
3728
- minClientsPerServer: number;
3729
- /** Fraction of a server's clients that must be degraded. Default `0.5`. */
3730
- degradedRatioThreshold: number;
3731
- /** RTT (ms) above which a relayed peer connection counts as degraded. Default `400`. */
3732
- rttInMs: number;
3733
- /** Inbound loss fraction above which a relayed peer connection counts as degraded. Default `0.03`. */
3734
- fractionLost: number;
3735
- /** Consecutive ticks the condition must hold before raising. Default `2`. */
3736
- consecutiveTicks: number;
3737
- /** Re-arm time (ms) before raising again for the same server. Default `60_000`. */
3738
- cooldownMs: number;
3739
- };
3740
- /** The per-server view this detector builds. */
3741
- type TurnServerHealth = {
3742
- serverUrl: string;
3743
- peerConnections: number;
3744
- degradedPeerConnections: number;
3745
- degradedRatio: number;
3746
- rttInMs?: StatsSummary;
3747
- fractionLost?: StatsSummary;
3748
- affectedClientIds: string[];
5244
+ qualityLimitation: {
5245
+ cpu: number;
5246
+ bandwidth: number;
5247
+ other: number;
5248
+ };
5249
+ freezes: {
5250
+ affectedClients: number;
5251
+ total: number;
5252
+ };
3749
5253
  };
3750
5254
  /**
3751
- * An **observer-level** detector that groups relayed peer connections by the TURN server carrying
3752
- * them and compares the servers against each other.
5255
+ * Aggregates a call along the **client** axis (as `TrackDistributionAggregator` does along the
5256
+ * publisher→subscriber axis): per-client health split into sending vs receiving, plus percentile
5257
+ * rollups and "affected ratio" counts for the whole call.
3753
5258
  *
3754
- * Counting TURN usage is not useful on its own; knowing that `turn-eu-1` has 22 of 30 clients in
3755
- * trouble while `turn-eu-2` has 1 of 34 is. Because the comparison spans calls, it lives on
3756
- * `observer.detectors` and raises `observer-issue` — a single actionable alert instead of fifty
3757
- * per-client ones.
5259
+ * Build once per call and call `aggregate()` on each `call.update()`.
3758
5260
  */
3759
- declare class TurnServerHealthDetector implements Detector {
3760
- private readonly _observer;
3761
- readonly name = "turn-server-health-detector";
3762
- private readonly _config;
3763
- private readonly _streaks;
3764
- private readonly _lastRaisedAt;
3765
- /** The per-server rollup computed on the most recent `update()`. */
3766
- lastServers: TurnServerHealth[];
3767
- constructor(_observer: Observer, config?: Partial<TurnServerHealthDetectorConfig>);
3768
- update(): void;
3769
- close(): void;
3770
- private _serverHealth;
5261
+ declare class CallHealthAggregator {
5262
+ private readonly _call;
5263
+ readonly thresholds: ClientHealthThresholds;
5264
+ constructor(_call: ObservedCall, thresholds?: ClientHealthThresholds);
5265
+ aggregate(): CallHealth;
5266
+ private _clientHealth;
3771
5267
  }
3772
5268
 
3773
- /** A snapshot of everyone currently reporting one issue type. */
3774
- type IssueCohort = {
3775
- /** The issue type, without the `-resolved` suffix. */
3776
- type: string;
3777
- /** The open intervals of that type, one per (client, key). */
3778
- issues: ActiveClientIssue[];
3779
- /** Distinct clients with at least one open interval of this type. */
3780
- clientIds: string[];
3781
- /** `clientIds.length / totalClients` (0..1). */
3782
- affectedRatio: number;
3783
- /** Clients considered for the ratio. */
3784
- totalClients: number;
3785
- /**
3786
- * Spread of the onsets, in **observer** time (ms) — `max(observedAt) - min(observedAt)`.
3787
- *
3788
- * Measured on the observer clock on purpose: `raisedAt` comes from each client's own clock, and
3789
- * comparing those across machines makes clock skew look like an infrastructure event. A small
3790
- * spread means the issues began together, which is the signature of a shared cause.
3791
- */
3792
- onsetSpreadInMs: number;
3793
- /** The earliest onset (observer clock). */
3794
- firstObservedAt: number;
3795
- };
3796
- /** Options controlling how long an unresolved issue is trusted. */
3797
- type IssueRegistryConfig = {
3798
- /**
3799
- * Drop an active issue this long after it was opened, even without a resolution (ms).
3800
- *
3801
- * A safety net for the case the lifecycle can't cover: a client that dies without its monitor
3802
- * running `close()` never sends the `-resolved` companion, and a stuck "active" issue would make
3803
- * every concurrency detector fire forever. Default `120_000`.
3804
- */
3805
- maxIssueAgeInMs: number;
3806
- };
3807
5269
  /**
3808
- * Read-side index over the **active** client issues in a call (or across an observer's calls).
5270
+ * Turning a correlation into a **conclusion**.
5271
+ *
5272
+ * Every detector in this library ultimately reports the same shape of observation: *N clients have
5273
+ * issue X open at once, and here is what they have in common*. That is useful but not yet
5274
+ * actionable — someone still has to know that congestion spread across unrelated calls means the
5275
+ * server, while CPU limitation spread across unrelated calls means a bad client release. This module
5276
+ * holds that interpretation step so it is stated once, consistently, instead of being re-derived by
5277
+ * whoever reads the alert at 3am.
3809
5278
  *
3810
- * `ObservedClient.activeIssues` holds the raw per-client state; this puts the questions detectors
3811
- * actually ask on top of it:
5279
+ * ### Two functions, because there are two questions
3812
5280
  *
3813
- * - *who currently has issue X?* → {@link cohortOf}
3814
- * - *which issue types are shared by several clients right now?* → {@link cohorts}
3815
- * - *which receivers of this published track are broken?* → {@link byTrackIds}
5281
+ * A detector already knows its scope — it was constructed with an `ObservedCall` or with the
5282
+ * `Observer`. Handing that scope back to a single generic function meant every caller supplied
5283
+ * fields the other scope needed and its own scope ignored: a call-scoped detector passing
5284
+ * `affectedCalls: 1, totalCalls: 1` forever, an observer-scoped one passing a participant ratio that
5285
+ * was deliberately never read. Placeholders like that are a standing invitation to read them as if
5286
+ * they meant something.
3816
5287
  *
3817
- * The distinction that matters is **concurrency**. A window-based count ("N clients reported
3818
- * congestion in the last 10 s") is a heuristic that has to guess whether the symptoms are still
3819
- * happening; an active-issue set is ground truth, because the client tells the server when the
3820
- * episode ends. Overlapping intervals are much stronger evidence of a common cause than
3821
- * near-in-time reports.
5288
+ * So there are two entry points, each taking only the facts its scope actually has:
3822
5289
  *
3823
- * It is a *view*, holding no state of its own beyond configuration, so it can be constructed per
3824
- * detector and queried on every `update()`.
5290
+ * - {@link concludeCallIssue} — within one call. The axis is *how much of the meeting*, and whether
5291
+ * the affected clients all subscribe to one published track.
5292
+ * - {@link concludeObserverIssue} — across calls. The axis is *how many independent calls*, which is
5293
+ * the only thing that separates "one bad room" from "our infrastructure".
5294
+ *
5295
+ * Neither the issue family nor the spread concludes anything alone: congestion in one call is a
5296
+ * meeting problem, congestion in six calls is an infrastructure problem, and the issue type is
5297
+ * identical in both.
3825
5298
  */
3826
- declare class IssueRegistry {
3827
- private readonly _source;
3828
- private readonly _config;
3829
- constructor(_source: ObservedCall | Observer, config?: Partial<IssueRegistryConfig>);
3830
- /** Every open issue in scope, with stale entries filtered out. */
3831
- activeIssues(now?: number): ActiveClientIssue[];
3832
- /** The number of clients in scope — the denominator for every ratio. */
3833
- get totalClients(): number;
3834
- /** The cohort for one issue type (empty `issues` when nobody has it open). */
3835
- cohortOf(type: string, now?: number): IssueCohort;
3836
- /** One cohort per issue type currently open somewhere in scope, largest first. */
3837
- cohorts(now?: number): IssueCohort[];
3838
- /**
3839
- * Open issues attributed to any of the given track ids — the join that turns a receiver-side
3840
- * issue into a statement about a **published** track (`trackId` → inbound track →
3841
- * `remoteOutboundTrack` → publisher).
5299
+ /** Where the fault most likely sits, given who is affected. */
5300
+ type IssueFaultDomain =
5301
+ /** Independent calls affected at once — they share only the servers and the network. */
5302
+ 'infrastructure'
5303
+ /** One call, broadly affected — something that call shares (its SFU worker, room, or host). */
5304
+ | 'call'
5305
+ /** The subscribers of one published track — the publisher's path or the forwarding of it. */
5306
+ | 'published-track'
5307
+ /** A single endpoint — its own device or last mile. */
5308
+ | 'endpoint'
5309
+ /** Independent calls affected, but by something endpoints own — a client build, not a server. */
5310
+ | 'client-population'
5311
+ /** Not enough signal to attribute. */
5312
+ | 'unknown';
5313
+ /** A stated verdict, attached to the raised issue payload. */
5314
+ type IssueConclusion = {
5315
+ /** Where to look. */
5316
+ faultDomain: IssueFaultDomain;
5317
+ /** One line, written to be readable in an alert without opening a dashboard. */
5318
+ summary: string;
5319
+ /** What to check first. Omitted when the issue family is unknown to this module. */
5320
+ recommendation?: string;
5321
+ /**
5322
+ * How much the spread alone justifies the verdict, `0..1`. Not a probability — a coarse ranking
5323
+ * so alerting can threshold on it. More independent calls, or a tighter onset, means higher.
3842
5324
  */
3843
- byTrackIds(trackIds: Iterable<string>, now?: number): ActiveClientIssue[];
3844
- /** Open issues reported by one client. */
3845
- byClientId(clientId: string, now?: number): ActiveClientIssue[];
3846
- private _toCohort;
3847
- /** Works for both scopes: a call yields its clients, an observer yields every call's clients. */
3848
- private _clients;
3849
- }
3850
-
3851
- declare const ConcurrentIssueTypes: {
3852
- /** Several participants have the same issue open **at the same time**. */
3853
- readonly concurrentClientIssues: "CONCURRENT_CLIENT_ISSUES";
3854
- /** Those issues also *began* together — the signature of one infrastructure event. */
3855
- readonly issueOnsetBurst: "ISSUE_ONSET_BURST";
5325
+ confidence: number;
3856
5326
  };
3857
- type ConcurrentIssueDetectorConfig = {
3858
- /**
3859
- * Only consider these issue types. Empty (default) means every type the clients report — which
3860
- * is usually what you want, since the detector is generic over the client's vocabulary.
3861
- */
3862
- issueTypes: string[];
3863
- /** Minimum participants in scope before a ratio is meaningful. Default `3`. */
3864
- minClients: number;
3865
- /** Minimum distinct clients sharing the open issue. Default `3`. */
3866
- minAffectedClients: number;
3867
- /** Fraction of participants that must share it. Default `0.5`. */
3868
- affectedRatioThreshold: number;
5327
+ /** The facts a **call-scoped** conclusion is drawn from. */
5328
+ type CallIssueSpread = {
5329
+ issueType: string;
5330
+ /** Distinct clients of this call with the issue open. */
5331
+ affectedClients: number;
5332
+ /** Participants in the call — the denominator. */
5333
+ totalClients: number;
5334
+ /** True when the onsets clustered — a shared trigger rather than drift. */
5335
+ onsetBurst: boolean;
3869
5336
  /**
3870
- * When the onsets of a qualifying cohort fall within this span (ms), the finding is escalated to
3871
- * `ISSUE_ONSET_BURST` — they didn't just overlap, they started together. Default `2_000`.
5337
+ * Set when the affected clients are the subscriber set of **one published track**.
5338
+ *
5339
+ * The strongest call-scoped statement available: those clients share a publisher and nothing
5340
+ * else, so the receivers are exonerated and the source's path is implicated.
3872
5341
  */
3873
- onsetBurstWindowInMs: number;
3874
- /** Re-arm time (ms) per issue type. Default `60_000`. */
3875
- cooldownMs: number;
3876
- /** Forwarded to the {@link IssueRegistry} (stale-issue expiry). */
3877
- registry?: Partial<IssueRegistryConfig>;
5342
+ publishedTrackId?: string;
5343
+ };
5344
+ /** The facts an **observer-scoped** conclusion is drawn from. */
5345
+ type ObserverIssueSpread = {
5346
+ issueType: string;
5347
+ /** Distinct clients across the fleet with the issue open. */
5348
+ affectedClients: number;
5349
+ /** Clients in the fleet. Reported for context; it does not gate anything at this scope. */
5350
+ totalClients: number;
5351
+ /** Distinct calls containing at least one affected client. The dimension that matters here. */
5352
+ affectedCalls: number;
5353
+ /** Calls in flight. */
5354
+ totalCalls: number;
5355
+ /** True when the onsets clustered. */
5356
+ onsetBurst: boolean;
3878
5357
  };
3879
5358
  /**
3880
- * Raises a finding when **several participants have the same issue open simultaneously**.
3881
- *
3882
- * This is the generic replacement for a family of symptom-specific detectors. The client already
3883
- * decides *what* is wrong for itself — `congestion`, `ice-disconnected`, `audio-concealment`,
3884
- * `video-decoder-overloaded`, and so on — with detectors that have hysteresis and multi-signal
3885
- * confirmation behind them. Re-deriving those verdicts from raw counters server-side would be
3886
- * strictly worse. What the server uniquely knows is *how many other participants are in the same
3887
- * state right now*, which is exactly the difference between "one person's Wi-Fi" and "our problem".
3888
- *
3889
- * Concurrency is judged from the **active issue set** (open interval per key), not from a sliding
3890
- * window of recent reports. That distinction matters: a window has to guess whether a symptom is
3891
- * still happening, whereas an interval is closed by the client when the episode actually ends
3892
- * (client-monitor-js >= 4.6.0 ships the `<type>-resolved` companion for this purpose).
3893
- *
3894
- * When the onsets also cluster inside `onsetBurstWindowInMs`, the finding is escalated to
3895
- * `ISSUE_ONSET_BURST`: participants degrading *together within a couple of seconds* is far more
3896
- * likely to be a deploy, a TURN failover or a link flap than a coincidence. Onsets are compared on
3897
- * the **observer clock** (`observedAt`), never on client clocks, because clock skew between
3898
- * machines would otherwise masquerade as a synchronized event.
3899
- *
3900
- * Works at both scopes — pass an `ObservedCall` for "this meeting", or the `Observer` for
3901
- * "everything this SFU is serving".
5359
+ * Draw the conclusion for a group of clients **within one call**.
5360
+ *
5361
+ * Ordered most-to-least specific: a track-scoped group is a stronger statement than a call-wide one,
5362
+ * and a single affected endpoint is not a statement about the call at all.
3902
5363
  */
3903
- declare class ConcurrentIssueDetector implements Detector {
3904
- private readonly _scope;
3905
- readonly name = "concurrent-issue-detector";
3906
- private readonly _config;
3907
- private readonly _registry;
3908
- private readonly _lastRaisedAt;
3909
- private readonly _isObserverScope;
3910
- /** The cohorts that qualified on the most recent `update()`. */
3911
- lastCohorts: IssueCohort[];
3912
- constructor(_scope: ObservedCall | Observer, config?: Partial<ConcurrentIssueDetectorConfig>);
3913
- update(): void;
3914
- close(): void;
3915
- private _raise;
3916
- }
5364
+ declare function concludeCallIssue(spread: CallIssueSpread): IssueConclusion;
5365
+ /**
5366
+ * Draw the conclusion for a group of clients spanning **several calls**.
5367
+ *
5368
+ * One affected call is not an observer-scoped finding — it has an obvious local explanation and the
5369
+ * call-scoped detector has already reported it — so that case returns `call` and says so rather than
5370
+ * dressing it up as a fleet event.
5371
+ *
5372
+ * Which domain breadth implicates depends on the family, and this is the whole reason the module
5373
+ * exists: `congestion` across unrelated calls points at the servers, `cpulimitation` across unrelated
5374
+ * calls points at what those *endpoints* share — a client release, a browser version, shared
5375
+ * virtualised hardware — and pointing an SFU team at the second one wastes a night.
5376
+ */
5377
+ declare function concludeObserverIssue(spread: ObserverIssueSpread): IssueConclusion;
3917
5378
 
3918
- declare const IssueFanOutTypes: {
3919
- /** Most receivers of one published track have the same issue open → the fault follows the source. */
3920
- readonly publishedTrackIssueFanOut: "PUBLISHED_TRACK_ISSUE_FAN_OUT";
3921
- /** Exactly one receiver of a track has it → that receiver's own problem. */
3922
- readonly singleReceiverIssue: "SINGLE_RECEIVER_ISSUE";
3923
- };
3924
- type IssueFanOutDetectorConfig = {
3925
- /** Only consider these issue types. Empty (default) = every type the receivers report. */
3926
- issueTypes: string[];
3927
- /** Minimum receivers of the track before a ratio is meaningful. Default `3`. */
3928
- minReceivers: number;
3929
- /** Fraction of a track's receivers that must share the issue. Default `0.6`. */
3930
- affectedRatioThreshold: number;
3931
- /** Also report the "only one receiver is affected" case. Default `true`. */
3932
- reportSingleReceiver: boolean;
3933
- /** Re-arm time (ms) per (track, issue type). Default `60_000`. */
3934
- cooldownMs: number;
3935
- registry?: Partial<IssueRegistryConfig>;
5379
+ /** An entry retained by {@link SlidingWindow}. */
5380
+ type SlidingWindowEntry<T> = {
5381
+ timestamp: number;
5382
+ value: T;
3936
5383
  };
3937
5384
  /**
3938
- * Attributes **client-reported issues to the published track they are about**, then asks how far
3939
- * the problem fans out across that track's receivers.
5385
+ * A time-bounded buffer used by detectors that reason over a window ("N of M clients degraded within
5386
+ * 10 s"). Entries older than `windowMs` are evicted on write and on read.
3940
5387
  *
3941
- * The join is what makes this possible: a receiver-side issue payload carries `trackId` (the client
3942
- * detectors report it for every track-scoped issue), the observer resolves that to an inbound track,
3943
- * and `RemoteTrackResolver` links the inbound track to the `remoteOutboundTrack` that published it.
3944
- * With the whole subscriber set of one source in hand, the verdict is straightforward and is the
3945
- * single most useful thing a server can say:
5388
+ * ### Ordering is enforced, not assumed
3946
5389
  *
3947
- * - **most receivers of Alice's track are affected** → the fault is on Alice's path — her uplink, the
3948
- * SFU's ingress, or its forwarding of that stream. Corroborated by whether the publisher's own
3949
- * egress looks healthy.
3950
- * - **one receiver of Alice's track is affected** → that receiver's downlink. Nothing to do with
3951
- * Alice, even though the symptom is reported against her stream.
5390
+ * Eviction walks from the front and stops at the first entry still inside the window, which is only
5391
+ * correct if entries are ordered by timestamp. Callers mostly pass `Date.now()` and are ordered by
5392
+ * construction — but not always: a timestamp taken from a client sample, or two calls inside the
5393
+ * same millisecond, can arrive out of order, and one such entry would park itself at the head and
5394
+ * stop eviction *permanently*, so the window would grow without bound and keep reporting symptoms
5395
+ * from hours ago.
5396
+ *
5397
+ * Rather than trust the caller, {@link add} inserts in timestamp order. Appending (the overwhelmingly
5398
+ * common case) stays O(1); an out-of-order insert costs a short backward scan, because such entries
5399
+ * are near the tail in practice.
5400
+ *
5401
+ * ### The window advances on the newest observation
3952
5402
  *
3953
- * Note this is deliberately generic over the issue vocabulary: `freezed-video-track`,
3954
- * `keyframe-storm`, `audio-concealment`, `video-decoder-overloaded`, `stuck-decoder` and anything a
3955
- * custom client detector invents all fan out the same way, so one mechanism replaces a family of
3956
- * symptom-specific detectors.
5403
+ * Eviction is relative to the largest timestamp seen, not the one just passed. A caller that reads
5404
+ * with a `now` behind the newest entry (a replayed sample, a clock that stepped back) would
5405
+ * otherwise un-evict nothing and, worse, a caller passing an old `now` to {@link add} would evict
5406
+ * everything newer.
3957
5407
  */
3958
- declare class IssueFanOutDetector implements Detector {
3959
- private readonly _call;
3960
- readonly name = "issue-fan-out-detector";
3961
- private readonly _config;
3962
- private readonly _registry;
3963
- private readonly _aggregator;
3964
- private readonly _lastRaisedAt;
3965
- constructor(_call: ObservedCall, config?: Partial<IssueFanOutDetectorConfig>);
3966
- update(): void;
3967
- close(): void;
3968
- /** Exposed for tests/dashboards: the distributions this detector reasons over. */
3969
- distributionsOf(): ObservedOutboundTrack[];
3970
- }
3971
-
3972
- declare const WorstReceiverContagionTypes: {
3973
- /**
3974
- * A publisher's sending bitrate is tracking its **worst** receiver — one bad downlink is
3975
- * dragging the quality everyone else gets.
3976
- */
3977
- readonly worstReceiverContagion: "WORST_RECEIVER_CONTAGION";
3978
- };
3979
- type WorstReceiverContagionDetectorConfig = {
3980
- /** Minimum receivers before the comparison means anything. Default `3`. */
3981
- minReceivers: number;
3982
- /** Observation window (ms) the correlation is computed over. Default `30_000`. */
3983
- windowMs: number;
3984
- /** Minimum samples inside the window before judging. Default `4`. */
3985
- minSamples: number;
5408
+ declare class SlidingWindow<T> {
5409
+ readonly windowMs: number;
5410
+ /** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
5411
+ readonly maxEntries: number;
5412
+ private _entries;
5413
+ private _latest;
5414
+ constructor(windowMs: number,
5415
+ /** Optional hard cap on retained entries, to bound memory on very chatty inputs. */
5416
+ maxEntries?: number);
5417
+ /** Add an entry (defaults to `Date.now()`), then evict anything outside the window. */
5418
+ add(value: T, timestamp?: number): void;
3986
5419
  /**
3987
- * How closely the publisher's bitrate must track the worst receiver's, relative to the spread
3988
- * between the worst and the median receiver. Default `0.75`.
5420
+ * The entries still inside the window, oldest first.
5421
+ *
5422
+ * A **copy** — callers routinely map/sort what they get back, and handing out the live array let
5423
+ * them mutate the window from the outside.
3989
5424
  */
3990
- trackingRatioThreshold: number;
5425
+ entries(now?: number): SlidingWindowEntry<T>[];
5426
+ /** The values still inside the window, oldest first. */
5427
+ values(now?: number): T[];
3991
5428
  /**
3992
- * The worst receiver must be this much worse than the median receiver before the situation even
3993
- * counts as "one bad apple" (0..1 of the median). Default `0.5` — i.e. at most half.
5429
+ * How many entries are inside the window as of `now`.
5430
+ *
5431
+ * Prefer this to `values(now).length`: counting through {@link values} allocates an array of every
5432
+ * entry only to read its length, which on a hot path is the whole cost of the call.
3994
5433
  */
3995
- outlierRatioThreshold: number;
3996
- /** Re-arm time (ms) per track. Default `120_000`. */
3997
- cooldownMs: number;
3998
- };
3999
- /**
4000
- * Detects the classic SFU misconfiguration: **the sender adapting to the worst receiver**.
4001
- *
4002
- * In a correctly built SFU the RTCP feedback loop is *terminated* at the server — each receiver's
4003
- * reports drive what that receiver is sent, and the publisher encodes for the server, not for the
4004
- * unluckiest participant. When the loop is instead relayed end to end, the publisher's bandwidth
4005
- * estimate collapses to the minimum across all receivers, so a single participant on a bad 3G link
4006
- * silently downgrades the stream *everyone* sees. Simulcast exists precisely to prevent this
4007
- * "lowest common denominator" outcome, and its absence (or an SFU that forwards RR/REMB verbatim)
4008
- * reproduces it.
4009
- *
4010
- * The signature is a correlation, not a threshold, so it is judged over a window: the publisher's
4011
- * outbound bitrate moving in lockstep with the *worst* receiver's inbound bitrate, while the median
4012
- * receiver has ample headroom. A publisher that drops because of its own uplink or CPU shows no such
4013
- * relationship — every receiver falls together and there is no outlier to track.
4014
- *
4015
- * This is arguably the most valuable thing an observer can detect, because the damage is invisible
4016
- * from every individual endpoint: the publisher sees "my bitrate went down", each healthy receiver
4017
- * sees "my video got worse", and nobody can see the causal link except the server.
4018
- */
4019
- declare class WorstReceiverContagionDetector implements Detector {
4020
- private readonly _call;
4021
- readonly name = "worst-receiver-contagion-detector";
4022
- private readonly _config;
4023
- private readonly _aggregator;
4024
- private readonly _windows;
4025
- private readonly _lastRaisedAt;
4026
- constructor(_call: ObservedCall, config?: Partial<WorstReceiverContagionDetectorConfig>);
4027
- update(): void;
4028
- close(): void;
4029
- private _windowOf;
5434
+ count(now?: number): number;
5435
+ /** Retained entries, without evicting first. See {@link count} for the windowed answer. */
5436
+ get size(): number;
5437
+ clear(): void;
5438
+ private _evict;
4030
5439
  }
4031
5440
 
4032
- declare const TrackDeliveryMismatchTypes: {
4033
- /**
4034
- * The source is sending, but **none** of its subscribers are receiving → the media is being lost
4035
- * between the publisher and the receivers. In an SFU that means the forwarding path.
4036
- */
4037
- readonly publishedTrackNotDelivered: "PUBLISHED_TRACK_NOT_DELIVERED";
4038
- /**
4039
- * The source is sending and most subscribers are fine, but **some** are dry → those consumers are
4040
- * broken individually (in mediasoup, the usual mitigation is recreating the consumer).
4041
- */
4042
- readonly receiverTrackNotDelivered: "RECEIVER_TRACK_NOT_DELIVERED";
5441
+ type TrendTesterConfig = {
4043
5442
  /**
4044
- * The source itself stopped producing, so its subscribers being dry is expected and **not** an
4045
- * SFU fault. Reported so the other two verdicts can be trusted as *not* being this.
5443
+ * How many of the most recent values to keep. The one knob that controls how far back "trend"
5444
+ * looks, for both tests — there is deliberately no separate window per test.
4046
5445
  */
4047
- readonly publisherTrackDry: "PUBLISHER_TRACK_DRY";
4048
- };
4049
- type TrackDeliveryMismatchDetectorConfig = {
4050
- /** The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`. */
4051
- dryInboundIssueType: string;
4052
- /** The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`. */
4053
- dryOutboundIssueType: string;
4054
- /** Minimum subscribers before "all of them" means anything. Default `2`. */
4055
- minReceivers: number;
4056
- /** Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`. */
4057
- allReceiversRatio: number;
4058
- /** Re-arm time (ms) per (track, verdict). Default `60_000`. */
4059
- cooldownMs: number;
4060
- registry?: Partial<IssueRegistryConfig>;
5446
+ size?: number;
5447
+ /** Page-Hinkley's drift tolerance and detection threshold — see {@link pageHinkley}. */
5448
+ pageHinkleyDelta?: number;
5449
+ pageHinkleyLambda?: number;
5450
+ /** Mann-Kendall's significance level — see {@link mannKendallVerdict}. */
5451
+ mannKendallAlpha?: number;
4061
5452
  };
4062
5453
  /**
4063
- * Answers **"is the media actually getting through?"** by joining the two ends of a published track.
5454
+ * Streaming home for `stats.ts`'s two trend tests: feed it one value at a time via {@link add}
5455
+ * instead of re-running the batch functions over an array you manage yourself.
4064
5456
  *
4065
- * A dry track is the clearest possible symptom — no bytes are arriving — but on its own it is
4066
- * ambiguous, and the ambiguity is precisely what a single endpoint cannot resolve. A receiver seeing
4067
- * silence cannot tell whether the camera was switched off, the SFU stopped forwarding, or its own
4068
- * consumer wedged. All three look identical from the browser.
5457
+ * ### The two tests answer different questions
4069
5458
  *
4070
- * With the publisher↔subscriber links this becomes a three-way decision:
5459
+ * Take a client's RTT, sampled every couple of seconds. Two things can go wrong with it, and only
5460
+ * one of them looks like a spike:
4071
5461
  *
4072
- * | publisher | subscribers | verdict |
4073
- * |---|---|---|
4074
- * | sending | **all** dry | `PUBLISHED_TRACK_NOT_DELIVERED` — the SFU/forwarding path |
4075
- * | sending | **some** dry | `RECEIVER_TRACK_NOT_DELIVERED` — those consumers (recreate them) |
4076
- * | dry | any dry | `PUBLISHER_TRACK_DRY` — the source stopped; not an SFU fault |
5462
+ * - **Mann-Kendall** asks *"is this drifting?"* — a monotonic trend, regardless of shape or scale.
5463
+ * `40, 45, 52, 61, 70, 84 ms` is a rising path with no single dramatic step; every jump is small
5464
+ * and plausible on its own. Mann-Kendall counts how many later samples exceed earlier ones and
5465
+ * reports whether that lopsidedness could plausibly be chance. It is rank-based, so one absurd
5466
+ * reading (a 4000 ms outlier from a stalled event loop) moves it by exactly one pair, not by the
5467
+ * 4000.
5468
+ * - **Page-Hinkley** asks *"did it change, and when?"* — a step. `40, 42, 39, 41, 180, 176, 182 ms`
5469
+ * is not a trend at all; it is one level followed by a different level, which is what a route
5470
+ * change or a TURN failover looks like. It accumulates the deviation from the running mean and
5471
+ * fires when the cumulative excess passes `lambda`.
4077
5472
  *
4078
- * The publisher side is judged from **both** signals available: its own `dry-outbound-track` issue
4079
- * when the client reports one, and — as the fallback, and the corroboration when it does not — the
4080
- * observed outbound RTP (`deltaPacketsSent`). That combination is what makes the first row
4081
- * trustworthy: the server can state that packets demonstrably left the publisher during the same
4082
- * interval in which every receiver got nothing.
5473
+ * Neither subsumes the other, which is why both live here on one window. A slow climb toward
5474
+ * unusability shows up in Mann-Kendall and never trips Page-Hinkley; a hard failover trips
5475
+ * Page-Hinkley immediately while Mann-Kendall may read `no-trend`, because after the step the series
5476
+ * is flat again. `tests/trendTester.spec.ts` builds both RTT series and shows exactly this.
4083
5477
  *
4084
- * This is the "SFU forwarding mismatch" check, and notably it needs **no** mediasoup instrumentation
4085
- * — the client's own dry-track verdicts plus the resolver links are sufficient.
4086
- */
4087
- declare class TrackDeliveryMismatchDetector implements Detector {
4088
- private readonly _call;
4089
- readonly name = "track-delivery-mismatch-detector";
4090
- private readonly _config;
4091
- private readonly _registry;
4092
- private readonly _aggregator;
4093
- private readonly _lastRaisedAt;
4094
- constructor(_call: ObservedCall, config?: Partial<TrackDeliveryMismatchDetectorConfig>);
4095
- update(): void;
4096
- close(): void;
4097
- }
4098
-
4099
- declare const UnconsumedTrackTypes: {
4100
- /** A track is being published to the SFU that nobody is subscribed to — pure wasted uplink. */
4101
- readonly unconsumedPublishedTrack: "UNCONSUMED_PUBLISHED_TRACK";
4102
- };
4103
- type UnconsumedTrackDetectorConfig = {
4104
- /** How long a track must stay unconsumed while sending before reporting (ms). Default `30_000`. */
4105
- minUnconsumedDurationInMs: number;
4106
- /** Ignore tracks below this bitrate — a trickle isn't worth an alert (bps). Default `50_000`. */
4107
- minBitrate: number;
4108
- /** Re-arm time (ms) per track. Default `300_000`. */
4109
- cooldownMs: number;
4110
- };
4111
- /**
4112
- * Finds tracks that are **published but consumed by nobody** — uplink and SFU ingress spent on media
4113
- * that is never forwarded anywhere.
5478
+ * ```ts
5479
+ * const rtt = new TrendTester({ size: 30, mannKendallAlpha: 0.05, pageHinkleyLambda: 50 });
4114
5480
  *
4115
- * This is the one detector that reads the resolver's *silence* as the signal: an outbound track with
4116
- * an empty `remoteInboundTracks` set, still pushing packets. The usual causes are a participant
4117
- * publishing while everyone has them hidden or muted-in-UI, a simulcast layer no viewer's bandwidth
4118
- * ever selects, or an application that forgot to stop a track after the last subscriber left.
5481
+ * peerConnection.on('update', () => {
5482
+ * if (peerConnection.currentRttInMs === undefined) return; // no measurement is not a measurement
5483
+ * rtt.add(peerConnection.currentRttInMs);
4119
5484
  *
4120
- * It is deliberately slow to fire: `minUnconsumedDurationInMs` must elapse with the track still
4121
- * sending, because a brief gap between publishing and the first subscription is completely normal at
4122
- * join time.
5485
+ * if (rtt.mannKendall().trend === 'increasing') warn('RTT is drifting up');
5486
+ * if (rtt.pageHinkley()?.changeDetected) {
5487
+ * warn('RTT stepped');
5488
+ * rtt.clear(); // the old level is no longer the baseline — judge the new one on its own
5489
+ * }
5490
+ * });
5491
+ * ```
4123
5492
  *
4124
- * ### Careful: this detector is only sound with a resolver
5493
+ * ### Both read the same window
4125
5494
  *
4126
- * "No subscribers" and "no resolver configured" produce the identical observation — an empty link
4127
- * set. Without a `RemoteTrackResolver` this would report *every* published track in the call as
4128
- * unconsumed, so it checks `call.remoteTrackResolver` at runtime and does nothing without one.
5495
+ * `size` is the one knob controlling how far back either test looks. They are kept incremental
5496
+ * differently, because they don't tolerate an evicted point the same way:
5497
+ *
5498
+ * - **Mann-Kendall**'s statistic is a sum over *pairs*, so evicting the oldest value only touches
5499
+ * the pairs it was part of — one pass over the (bounded) window corrects it in O(size) instead of
5500
+ * the O(size²) a full recompute costs.
5501
+ * - **Page-Hinkley**'s statistic is a running minimum of a cumulative sum, which has no cheap
5502
+ * correction for "forget this one old point" — the minimum may have depended on it. It is
5503
+ * recomputed from the window on every {@link add} rather than hand-rolling an incremental version
5504
+ * that would be easy to get subtly wrong. That recompute is O(size), the same order as above.
5505
+ *
5506
+ * ### Non-finite input is rejected, not absorbed
5507
+ *
5508
+ * See {@link add}. A single `NaN` would otherwise destroy the instance permanently.
4129
5509
  */
4130
- declare class UnconsumedTrackDetector implements Detector {
4131
- private readonly _call;
4132
- readonly name = "unconsumed-track-detector";
4133
- private readonly _config;
4134
- /** trackId -> when it was first seen sending with no subscribers. */
4135
- private readonly _unconsumedSince;
4136
- private readonly _lastRaisedAt;
4137
- constructor(_call: ObservedCall, config?: Partial<UnconsumedTrackDetectorConfig>);
4138
- update(): void;
4139
- close(): void;
5510
+ declare class TrendTester {
5511
+ private readonly _size;
5512
+ private readonly _values;
5513
+ private readonly _tieCounts;
5514
+ private readonly _pageHinkleyDelta;
5515
+ private readonly _pageHinkleyLambda;
5516
+ private readonly _mannKendallAlpha;
5517
+ private _s;
5518
+ private _rejected;
5519
+ private _pageHinkleyResult?;
5520
+ constructor(config?: TrendTesterConfig);
5521
+ /** Values rejected by {@link add} for being non-finite. Non-zero means the caller has a bug. */
5522
+ get rejected(): number;
5523
+ /** How many values are currently in the window (`<= size`). */
5524
+ get length(): number;
5525
+ /** The configured window length. */
5526
+ get size(): number;
5527
+ /**
5528
+ * Add the next value in the stream, evicting the oldest once the window is full.
5529
+ *
5530
+ * **Non-finite values are rejected** rather than stored, and the rejection is counted in
5531
+ * {@link rejected}. This is not defensive noise — it is the difference between a bad reading and
5532
+ * a bad instance. `Math.sign(NaN)` is `NaN`, so a single `NaN` would poison the incremental
5533
+ * Mann-Kendall sum `_s` **permanently**: every later `add` and `_evictOldest` adds or subtracts
5534
+ * `NaN`, the z-score is `NaN`, every comparison against it is `false`, and the tester silently
5535
+ * reports `no-trend` forever after. It would also take a `NaN` key in `_tieCounts` that can never
5536
+ * be matched on eviction, since `NaN !== NaN`.
5537
+ *
5538
+ * `undefined` RTT (no measurement this tick) must not be coerced to `0` and passed in either —
5539
+ * "we didn't measure" is not "the trip took no time", and feeding zeros manufactures a downward
5540
+ * trend. Skip the sample instead.
5541
+ */
5542
+ add(value: number): void;
5543
+ /** The current Page-Hinkley read-out over the window. `undefined` before the first value. */
5544
+ pageHinkley(): PageHinkleyResult | undefined;
5545
+ /** The current Mann-Kendall read-out over the window. */
5546
+ mannKendall(): MannKendallResult;
5547
+ /** Drop everything, e.g. after a detected change point, to start judging the trend fresh. */
5548
+ clear(): void;
5549
+ /** Remove the oldest value from the window and correct `_s` for the pairs it was part of. */
5550
+ private _evictOldest;
5551
+ private _bumpTie;
4140
5552
  }
4141
5553
 
4142
5554
  interface Logger {
@@ -4210,4 +5622,4 @@ declare function createInMemorySink(samples?: ClientSample[]): InMemorySink;
4210
5622
  declare function createDefaultMediasoupRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
4211
5623
  declare function createP2pRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
4212
5624
 
4213
- export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, AudioImpairmentFanOutDetector, type AudioImpairmentFanOutDetectorConfig, AudioImpairmentFanOutTypes, type CallAppDataFactory, type CallHealth, CallHealthAggregator, CallWideDegradationDetector, type CallWideDegradationDetectorConfig, CallWideDegradationTypes, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, CommonSourceDegradationDetector, type CommonSourceDegradationDetectorConfig, CommonSourceDegradationTypes, ConcurrentIssueDetector, type ConcurrentIssueDetectorConfig, ConcurrentIssueTypes, type Detector, Detectors, IceDisruptionDetector, type IceDisruptionDetectorConfig, IceDisruptionTypes, InMemorySink, type IssueCohort, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, IssueRegistry, type IssueRegistryConfig, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, type Logger, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, type ObservedTrackDistribution, Observer, type ObserverEventBase, type ObserverEvents, type ObserverLogger, PliAndFreezeFanOutDetector, type PliAndFreezeFanOutDetectorConfig, PliAndFreezeFanOutTypes, type PublisherDistributionEntry, RESOLVED_ISSUE_SUFFIX, type ReceiverDistributionEntry, type ReceiverHealthThresholds, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolvers, type ResolvedClientIssue, type SampleRejectedReason, type ScoreCalculator, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrackDistributionAggregator, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, WorstReceiverContagionDetector, type WorstReceiverContagionDetectorConfig, WorstReceiverContagionTypes, baseIssueType, counterDelta, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultClientHealthThresholds, defaultReceiverHealthThresholds, isResolutionEntry, median, parseIssuePayload, percentile, setObserverLogger, summarize };
5625
+ export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, type ActiveIssueTracker, ActiveIssuesRegistry, type AvailableCallScopeDetectorsConfigs, type AvailableDetectorsConfigs, type AvailableObserverScopeDetectorsConfigs, type AvailableValidatorConfigs, CODEC_MISMATCH_ISSUE, type CallAppDataFactory, CallConcurrentIssueDetector, type CallConcurrentIssueDetectorConfig, type CallConcurrentIssueGroup, CallConcurrentIssueTypes, type CallHealth, CallHealthAggregator, type CallIssueSpread, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientPopulation, type ClientPopulationAxis, ClientPopulationIssueDetector, type ClientPopulationIssueDetectorConfig, ClientPopulationIssueTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type CodecConsistencyReportPayload, CodecConsistencyValidator, type CodecConsistencyValidatorConfig, type CodecEvidence, type CorroboratedPublisherFault, type Detector, Detectors, InMemorySink, type IssueConclusion, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, type IssueFaultDomain, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, LOWEST_COMMON_DENOMINATOR_ISSUE, type Logger, type MannKendallResult, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, ObservedClientIssueRegistry, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, ObserverConcurrentIssueDetector, type ObserverConcurrentIssueDetectorConfig, type ObserverConcurrentIssueGroup, ObserverConcurrentIssueTypes, type ObserverEventBase, type ObserverEvents, type ObserverIssue, type ObserverIssueSpread, type ObserverLogger, type PageHinkleyResult, PublisherFaultCorroborationDetector, type PublisherFaultCorroborationDetectorConfig, PublisherFaultTypes, RESOLVED_ISSUE_SUFFIX, type RemoteTrackLinkEvidence, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolverReportPayload, RemoteTrackResolverValidator, type RemoteTrackResolverValidatorConfig, type RemoteTrackResolvers, type ResolvedActiveClientIssue, type RunningValidator, type SampleRejectedReason, type ScoreCalculator, SfuCongestionDetector, type SfuCongestionDetectorBucket, type SfuCongestionDetectorConfig, type SfuCongestionDetectorEvaluation, type SfuCongestionDetectorReport, type SimulcastReceiverEvidence, type SimulcastReceiverReportPayload, SimulcastReceiverValidator, type SimulcastReceiverValidatorConfig, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrendTester, type TrendTesterConfig, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, TurnServerOutageDetector, type TurnServerOutageDetectorConfig, TurnServerOutageTypes, UNRESOLVED_TRACK_LINKS_ISSUE, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, type ValidationReport, type Validator, type ValidatorName, baseIssueType, concludeCallIssue, concludeObserverIssue, correlation, counterDelta, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultClientHealthThresholds, isClientIssueResolutionEntry, issuePayloadAsString, issuePayloadOf, mannKendall, mannKendallVerdict, median, medianAbsoluteDeviation, pageHinkley, percentile, percentileOfSorted, robustZScore, setObserverLogger, summarize };