@observertc/observer-js 1.0.0-beta.18 → 1.0.0-beta.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.mts CHANGED
@@ -2578,11 +2578,33 @@ type CallSummaryConfig = {
2578
2578
  * with only `enrich` is a perfectly good summary.
2579
2579
  */
2580
2580
  include: CallSummarySection[];
2581
- /** Fold arbitrary state in from any call-scoped event. See {@link CallSummaryEnrichers}. */
2581
+ /**
2582
+ * Fold arbitrary state in from any call-scoped event, into `summary.attachments`. See
2583
+ * {@link CallSummaryEnrichers}.
2584
+ *
2585
+ * Each enricher runs on **every** occurrence of its event, for its own call, so prefer the
2586
+ * low-frequency lifecycle events (`client-joined`, `call-issue`) over `client-updated`, which fires
2587
+ * once per sample per client. Keep them cheap and side-effect free: an enricher that throws is logged
2588
+ * and skipped rather than allowed to disturb the call, but one that is slow is on the ingestion path.
2589
+ */
2582
2590
  enrich?: CallSummaryEnrichers;
2583
- /** Cap on the retained issue log. Default `500`. See the note on truncation. */
2591
+ /**
2592
+ * Cap on retained issues. Default `500`.
2593
+ *
2594
+ * A bound on memory per call, so scale it by how long your calls run and how noisy they are, not by
2595
+ * taste — a two-hour call with a struggling participant can raise hundreds. Past the cap issues are
2596
+ * dropped and `summary.truncated.issues` counts them, so the real total stays recoverable as
2597
+ * `issues.length + (truncated?.issues ?? 0)`. `0` collects the count only, keeping no issue objects.
2598
+ */
2584
2599
  maxIssues: number;
2585
- /** Cap on retained client ids. Default `10_000`. */
2600
+ /**
2601
+ * Cap on retained client ids. Default `10_000`.
2602
+ *
2603
+ * High because the elements are short strings and the usual reason to read a summary is *who was in
2604
+ * this call*. It exists so a webinar-scale room cannot grow the summary without limit. Overflow is
2605
+ * counted in `summary.truncated.clientIds`; `joined`, `left` and `peak` are unaffected by the cap,
2606
+ * since they are counters rather than a list.
2607
+ */
2586
2608
  maxClientIds: number;
2587
2609
  };
2588
2610
  /**
@@ -3667,7 +3689,29 @@ declare class RemoteTrackResolver {
3667
3689
  private readonly resolvers;
3668
3690
  private readonly _publisherIdToOutboundTrack;
3669
3691
  private readonly _subscriberIdToInboundTrack;
3692
+ /**
3693
+ * Tracks whose publisher id the strategy could not resolve **yet**.
3694
+ *
3695
+ * A track announces itself once, but its `attachments` are replaced on every sample, so a key that
3696
+ * is missing from the first sample can appear on the second — and a strategy backed by an
3697
+ * application's own mapping (a server-side `ssrc -> producerId` table, say) is inherently racy
3698
+ * against sample arrival. Resolving only at `*-track-added` meant losing those tracks for their
3699
+ * entire life, silently: an unresolvable outbound track never even reaches
3700
+ * `unconsumedOutboundTracks`, so it is invisible to `UnconsumedTrackDetector` too.
3701
+ *
3702
+ * So they wait here and are retried on their own `*-track-updated`, i.e. exactly when new stats
3703
+ * arrived for them. A track leaves on its first successful resolution or on removal, which makes
3704
+ * a linked track cost one `Set.has` per update and bounds these sets by the unresolved tracks
3705
+ * alive right now.
3706
+ */
3707
+ private readonly _pendingInboundTracks;
3708
+ private readonly _pendingOutboundTracks;
3670
3709
  constructor(observedCall: ObservedCall, resolvers: RemoteTrackResolvers);
3710
+ /** Tracks still waiting for a resolvable publisher id. Diagnostics; normally both are empty. */
3711
+ get pendingTrackCounts(): {
3712
+ inbound: number;
3713
+ outbound: number;
3714
+ };
3671
3715
  /** The published (outbound) track for a publisher id, if any. */
3672
3716
  getOutboundTrackByPublisherId(publisherId: string): ObservedOutboundTrack | undefined;
3673
3717
  /** The subscribed (inbound) track for a subscriber id, if the strategy resolves subscriber ids. */
@@ -3704,18 +3748,47 @@ type CallConcurrentIssueDetectorConfig = {
3704
3748
  * exactly these and sees nothing else.
3705
3749
  */
3706
3750
  issueTypes: string[];
3707
- /** Minimum participants in the call before a ratio is meaningful. Default `3`. */
3751
+ /**
3752
+ * Participants the call needs before a ratio means anything. Default `3`.
3753
+ *
3754
+ * In a 1:1 call "half the participants" is one person, which is a client problem and not a call
3755
+ * problem — `2` effectively disables the ratio gate. Raise it for large-meeting products where you
3756
+ * only care once a handful are affected.
3757
+ */
3708
3758
  minClients: number;
3709
- /** Minimum distinct clients sharing the open issue. Default `3`. */
3759
+ /**
3760
+ * Distinct clients that must share the issue. Default `3`.
3761
+ *
3762
+ * The absolute floor under `affectedRatioThreshold`, so a small call cannot clear a ratio with two
3763
+ * unlucky people. Sensible range `2`–`5`; `2` is the lowest that can still mean "more than one
3764
+ * participant", which is the whole premise.
3765
+ */
3710
3766
  minAffectedClients: number;
3711
- /** Fraction of the call's participants that must share it. Default `0.5`. */
3767
+ /**
3768
+ * Fraction of the call's participants that must share it, `0`–`1`. Default `0.5`.
3769
+ *
3770
+ * Typical `0.3`–`0.7`. Lower catches partial events — a subset on one SFU worker — at the cost of
3771
+ * firing on a few coincidentally unhappy participants; `1` demands literally everyone, which real
3772
+ * incidents rarely produce because someone always reconnects first.
3773
+ */
3712
3774
  affectedRatioThreshold: number;
3713
3775
  /**
3714
- * When the onsets fall within this span (ms), the finding is escalated to `ISSUE_ONSET_BURST` —
3715
- * they didn't just overlap, they started together. Default `2_000`.
3776
+ * Onsets falling within this span (ms) escalate the finding to `ISSUE_ONSET_BURST` — they did not
3777
+ * just overlap, they started together. Default `2_000`.
3778
+ *
3779
+ * Bound this by your sampling period, not below it: onsets are only known as accurately as clients
3780
+ * report them, so a window shorter than one sampling period can only fire by luck. Typical
3781
+ * `1_000`–`5_000`. Wider makes the escalation meaningless, since unrelated issues drift into the
3782
+ * same window.
3716
3783
  */
3717
3784
  onsetBurstWindowInMs: number;
3718
- /** Re-arm time (ms) per issue type. Default `60_000`. */
3785
+ /**
3786
+ * Re-arm time per issue type (ms). Default `60_000`.
3787
+ *
3788
+ * A shared event is one incident, not one per tick. Too low and a persistent problem raises an
3789
+ * issue every tick for as long as it lasts; too high and a genuinely new occurrence is swallowed
3790
+ * by the previous one's cooldown. Typical `30_000`–`300_000`.
3791
+ */
3719
3792
  cooldownMs: number;
3720
3793
  };
3721
3794
  /** What the detector currently knows about one issue type in this call. */
@@ -3779,25 +3852,88 @@ declare class CallConcurrentIssueDetector implements Detector, ActiveIssueTracke
3779
3852
  private _groupOf;
3780
3853
  }
3781
3854
 
3855
+ /** A point on the earth, as an application reports it for one client. */
3856
+ type ClientLocation = {
3857
+ latitude: number;
3858
+ longitude: number;
3859
+ };
3860
+ /**
3861
+ * Approximate cell width at each geohash length, for choosing a precision.
3862
+ *
3863
+ * Index is the character count; the value is the rough cell size at the equator. Cells are taller
3864
+ * than they are wide at high latitudes, so treat these as an order of magnitude, not a radius.
3865
+ */
3866
+ declare const GEOHASH_CELL_SIZES: readonly ["", "~5000 km", "~1250 km", "~156 km", "~39 km", "~5 km", "~1.2 km", "~150 m"];
3867
+ /**
3868
+ * Encode a point as a geohash of `precision` characters — a **grid cell key**, not a cluster.
3869
+ *
3870
+ * ### Why cells rather than "within N kilometres"
3871
+ *
3872
+ * Grouping clients "within a radius" sounds like the natural thing and is a much worse fit. It is a
3873
+ * clustering problem, not a keying one: the groups depend on which client you start from, two
3874
+ * clients can each be within the radius of a third but not of each other, group identity is not
3875
+ * stable as participants join and leave, and maintaining it costs pairwise distance work. None of
3876
+ * that survives contact with a detector that has to produce the *same* group name on every tick so
3877
+ * a cooldown and a control group mean anything.
3878
+ *
3879
+ * A geohash prefix is a plain function of the coordinates: O(1), stable for the life of the client,
3880
+ * and usable directly as a population label. The honest cost is that a cell boundary can separate
3881
+ * two clients who are physically adjacent, which splits a real group into two smaller ones. That
3882
+ * biases towards **missing** a finding rather than inventing one, which is the right direction for
3883
+ * something that raises issues.
3884
+ *
3885
+ * Returns `undefined` for coordinates that are not finite or not on the earth, rather than encoding
3886
+ * nonsense into a plausible-looking cell key.
3887
+ */
3888
+ declare function geohash(location: ClientLocation, precision?: number): string | undefined;
3889
+
3782
3890
  declare const ClientPopulationIssueTypes: {
3783
3891
  /** One issue type is concentrated on one client population while the rest of the fleet is fine. */
3784
3892
  readonly clientPopulationIssue: "CLIENT_POPULATION_ISSUE";
3785
3893
  };
3786
3894
  /** The client attribute to group by. One axis per detector — see the class description. */
3787
- type ClientPopulationAxis = 'browser' | 'engine' | 'platform' | 'operationSystem';
3895
+ type ClientPopulationAxis = 'browser' | 'engine' | 'platform' | 'operationSystem' | 'location';
3896
+ /**
3897
+ * Reads a client's coordinates, for `groupBy: 'location'`. **Required for that axis.**
3898
+ *
3899
+ * There is no coordinate field in `ClientSample`, so the shape is yours: read it off
3900
+ * `client.attachments`, off a custom meta item, or from an `appData` field your accept middleware
3901
+ * filled in. Return `undefined` for clients whose location you do not know — they are then excluded
3902
+ * from both the population and the control group, exactly like a client that never reported its
3903
+ * browser.
3904
+ */
3905
+ type ClientLocationResolver = (client: ObservedClient) => ClientLocation | undefined;
3788
3906
  type ClientPopulationIssueDetectorConfig = {
3789
3907
  /**
3790
3908
  * The issue types to watch. **Required, and must not be empty.**
3791
3909
  *
3792
- * The types worth grouping this way are the ones an endpoint owns: `cpulimitation`,
3910
+ * **Match the issue family to the axis.** On the endpoint axes (`browser`, `engine`, `platform`,
3911
+ * `operationSystem`) the types worth grouping are the ones an endpoint owns: `cpulimitation`,
3793
3912
  * `encoder-bottleneck`, `capture-bottleneck`, `stuck-decoder`, `video-decoder-overloaded`.
3794
3913
  * Grouping a *network* symptom by browser is a category error — `congestion` clusters by ISP and
3795
- * geography, neither of which this detector can see, and it would happily report a browser
3796
- * correlation that is really a "most of our users are on Chrome" artefact.
3914
+ * geography, not by build, and the detector would happily report a browser correlation that is
3915
+ * really a "most of our users are on Chrome" artefact.
3916
+ *
3917
+ * On the `location` axis it is the other way round: group the network symptoms — `congestion`,
3918
+ * `ice-disconnected`, `unstable-ice-path` — and not the endpoint ones, since there is no reason a
3919
+ * decoder should stall by geography.
3797
3920
  */
3798
3921
  issueTypes: string[];
3799
3922
  /** Which client attribute to group by. Default `'browser'`. */
3800
3923
  groupBy: ClientPopulationAxis;
3924
+ /**
3925
+ * Where to read a client's coordinates. **Required when `groupBy` is `'location'`**, ignored
3926
+ * otherwise. See {@link ClientLocationResolver}.
3927
+ */
3928
+ resolveClientLocation?: ClientLocationResolver;
3929
+ /**
3930
+ * Geohash characters to group locations by, i.e. how coarse a "place" is. Default `3` (~156 km).
3931
+ *
3932
+ * `2` ~1250 km, `3` ~156 km, `4` ~39 km, `5` ~5 km. Coarser cells hold more clients, which is what
3933
+ * makes a rate mean anything, so start coarse: a city-sized cell rarely has `minPopulationSize`
3934
+ * participants in it. See `utils/geohash` for why this is a grid cell and not a radius.
3935
+ */
3936
+ locationPrecision: number;
3801
3937
  /**
3802
3938
  * Group by `name` only, or by `name + version`. Default `true` (include version).
3803
3939
  *
@@ -3805,26 +3941,66 @@ type ClientPopulationIssueDetectorConfig = {
3805
3941
  * thing that changed on a date. Set to `false` when comparing whole engines.
3806
3942
  */
3807
3943
  includeVersion: boolean;
3808
- /** Minimum clients in a population before its rate means anything. Default `20`. */
3944
+ /**
3945
+ * Clients in a population before its rate means anything. Default `20`.
3946
+ *
3947
+ * Higher than the other detectors' minimums on purpose: this one compares *rates*, and a rate over
3948
+ * five clients is not a rate. Sensible range `20`–`100`. On the `location` axis this is the field
3949
+ * most likely to silence the detector — a city-sized cell rarely holds twenty concurrent
3950
+ * participants, so reach for a coarser `locationPrecision` before lowering this.
3951
+ */
3809
3952
  minPopulationSize: number;
3810
- /** Minimum affected clients in the population. Default `5`. */
3953
+ /**
3954
+ * Affected clients required within the population. Default `5`.
3955
+ *
3956
+ * Checked independently of `affectedRatioThreshold`, so one unlucky user on a rare browser cannot
3957
+ * page anyone however striking the ratio looks. Sensible range `5`–`20`.
3958
+ */
3811
3959
  minAffectedClients: number;
3812
- /** Share of the population that must be affected. Default `0.3`. */
3960
+ /**
3961
+ * Share of the population that must be affected, `0`–`1`. Default `0.3`.
3962
+ *
3963
+ * Lower than the per-call thresholds deliberately: an issue hitting 30% of one browser version while
3964
+ * the rest of the fleet is clean is already a strong signal, and endpoint faults rarely affect
3965
+ * *everyone* on a build. Typical `0.2`–`0.5`. This is the weakest of the gates —
3966
+ * `minRelativeRisk` is what makes the finding mean anything.
3967
+ */
3813
3968
  affectedRatioThreshold: number;
3814
3969
  /**
3815
3970
  * How many times worse the suspect population must be than the rest of the fleet. Default `3`.
3816
3971
  *
3817
- * This is the control, and it is what makes the finding mean anything. See the class description.
3972
+ * **This is the gate that makes the finding mean anything** — see the class description. Typical
3973
+ * `2`–`10`. At `2` you will see populations that are merely somewhat worse, which is often just a
3974
+ * different usage pattern; at `10` only stark, unambiguous concentrations survive. A spotless
3975
+ * control group yields `Infinity`, which clears any threshold, so the minimum-count gates above are
3976
+ * what stop that from being trivial.
3818
3977
  */
3819
3978
  minRelativeRisk: number;
3820
- /** Minimum clients **outside** the suspect population before a comparison is possible. Default `20`. */
3979
+ /**
3980
+ * Clients **outside** the suspect population before a comparison is possible. Default `20`.
3981
+ *
3982
+ * "Worse than everyone else" needs an everyone else. Sensible range `20`–`100`. Note the practical
3983
+ * consequence: a fleet that is overwhelmingly one browser can never have that browser reported,
3984
+ * because there is no control group left — which is honest, since at 95% Chrome you cannot separate a
3985
+ * Chrome fault from a fleet-wide one.
3986
+ */
3821
3987
  minControlSize: number;
3822
- /** Re-arm time (ms) per (population, issue type). Default `300_000`. */
3988
+ /**
3989
+ * Re-arm time per (population, issue type) in ms. Default `300_000`.
3990
+ *
3991
+ * Long by design: a bad client build is a condition lasting days, not an event, and the action it
3992
+ * prompts — ship a fix, roll back a version — is not one you take twice an hour. Typical
3993
+ * `300_000`–`3_600_000`.
3994
+ */
3823
3995
  cooldownMs: number;
3824
3996
  };
3825
3997
  /** The rollup for one population on one issue type. */
3826
3998
  type ClientPopulation = {
3827
- /** e.g. `'Chrome 141'`, or `'Chrome'` when `includeVersion` is off. */
3999
+ /**
4000
+ * e.g. `'Chrome 141'`, or `'Chrome'` when `includeVersion` is off. For the `'location'` axis this
4001
+ * is the geohash cell — never the coordinates themselves, so an archived payload carries a place
4002
+ * at the configured resolution and not a person's position.
4003
+ */
3828
4004
  population: string;
3829
4005
  axis: ClientPopulationAxis;
3830
4006
  issueType: string;
@@ -3896,12 +4072,48 @@ type ClientPopulation = {
3896
4072
  * Safari is usually one fact reported twice, and a detector that emits both leaves the reader to
3897
4073
  * work out which one is causal. Pick the axis you want to reason about.
3898
4074
  *
4075
+ * ### The `location` axis
4076
+ *
4077
+ * With `groupBy: 'location'` the population is a **geohash cell** rather than a client attribute, so
4078
+ * the same machinery answers a different question: *is this symptom concentrated in one place?* That
4079
+ * is the grouping the note above says browsers cannot give you, and it is the one that matters for
4080
+ * network symptoms.
4081
+ *
4082
+ * The observer does not derive "this client's RTT jumped" — `client-monitor`'s `CongestionDetector`
4083
+ * already owns that verdict, comparing each peer connection's RTT against its own EWMA baseline and
4084
+ * requiring a bandwidth-limitation corroboration before it raises `congestion`. Absolute RTT is not
4085
+ * comparable between clients anyway: someone 200 ms away is *always* 200 ms away, so the only signal
4086
+ * is deviation from that client's own baseline, which is exactly what the client already measures.
4087
+ * This detector's contribution is the part no endpoint can see — that many of the affected clients
4088
+ * are in the same place at the same time.
4089
+ *
4090
+ * ```ts
4091
+ * observer.addObserverDetector('client-population-issue-detector', {
4092
+ * issueTypes: [ 'congestion', 'ice-disconnected' ],
4093
+ * groupBy: 'location',
4094
+ * locationPrecision: 3, // ~156 km cells
4095
+ * resolveClientLocation: (client) => client.attachments?.geo as { latitude: number, longitude: number },
4096
+ * });
4097
+ * ```
4098
+ *
4099
+ * Coordinates are not in `ClientSample`, so `resolveClientLocation` is required — see
4100
+ * {@link ClientLocationResolver}. Only the cell key reaches the issue payload, never the
4101
+ * coordinates, which matters because these payloads are archived into call summaries.
4102
+ *
4103
+ * **The limitation to state plainly: geography is confounded with your topology.** The control group
4104
+ * is "everyone outside this cell", which cannot separate *"the path into this region degraded"* from
4105
+ * *"the SFU that happens to serve this region degraded"*. If a region maps largely onto one
4106
+ * deployment, both hypotheses fit the same evidence. The discriminator is whether clients elsewhere
4107
+ * on the same server also degraded, which is what `SfuCongestionDetector` and
4108
+ * `TurnServerHealthDetector` answer — so the conclusion here points at them rather than claiming an
4109
+ * attribution it cannot support.
4110
+ *
3899
4111
  * ### Clients that never reported their metadata
3900
4112
  *
3901
4113
  * `browser` / `engine` / `platform` / `operationSystem` arrive as client metadata and may be absent —
3902
4114
  * a client that closed before sending them, or an application that does not collect them. Those
3903
4115
  * clients are excluded from **both** the population and the control group rather than bucketed as
3904
- * `'unknown'`. A synthetic `'unknown'` population would be a mixture of every real one, so any rate
4116
+ * `'unknown'`. The same applies to a client whose location `resolveClientLocation` cannot supply. A synthetic `'unknown'` population would be a mixture of every real one, so any rate
3905
4117
  * computed for it means nothing, and leaving those clients in the control group would dilute the
3906
4118
  * comparison with clients whose kind we cannot verify.
3907
4119
  */
@@ -3953,9 +4165,21 @@ type PublisherFaultCorroborationDetectorConfig = {
3953
4165
  * "I am not getting this properly", which is the half the publisher cannot see.
3954
4166
  */
3955
4167
  receiverIssueTypes: string[];
3956
- /** Minimum subscribers of the track that must be complaining. Default `2`. */
4168
+ /**
4169
+ * Subscribers of the track that must be complaining at the same time. Default `2`.
4170
+ *
4171
+ * `1` still yields a genuine two-sided corroboration — publisher and one receiver agreeing is
4172
+ * already more than either says alone — but `2` rules out the case where a single receiver's own
4173
+ * downlink is at fault and merely coincides with the publisher's complaint. Sensible range `1`–`3`;
4174
+ * higher mostly costs you findings in small calls, where a track may only have two subscribers.
4175
+ */
3957
4176
  minAffectedReceivers: number;
3958
- /** Re-arm time (ms) per published track. Default `60_000`. */
4177
+ /**
4178
+ * Re-arm time per published track (ms). Default `60_000`.
4179
+ *
4180
+ * Typical `30_000`–`300_000`. This detector raises the highest-confidence finding in the library,
4181
+ * so it is the one you least want repeating every tick.
4182
+ */
3959
4183
  cooldownMs: number;
3960
4184
  };
3961
4185
  /** The two-sided evidence behind one finding. */
@@ -4054,7 +4278,13 @@ type ObserverConcurrentIssueDetectorConfig = {
4054
4278
  * exactly these and sees nothing else.
4055
4279
  */
4056
4280
  issueTypes: string[];
4057
- /** Minimum distinct clients sharing the open issue. Default `3`. */
4281
+ /**
4282
+ * Distinct clients that must share the open issue. Default `3`.
4283
+ *
4284
+ * Absolute, deliberately — see `affectedCallRatioThreshold` for why no client *ratio* exists at
4285
+ * this scope. Sensible range `3`–`10`; scale it with fleet size, since three clients is a real
4286
+ * signal across five calls and background noise across five hundred.
4287
+ */
4058
4288
  minAffectedClients: number;
4059
4289
  /**
4060
4290
  * Minimum number of *distinct calls* the affected clients must span. Default `2`.
@@ -4078,11 +4308,21 @@ type ObserverConcurrentIssueDetectorConfig = {
4078
4308
  */
4079
4309
  affectedCallRatioThreshold: number;
4080
4310
  /**
4081
- * When the onsets fall within this span (ms), the finding is escalated to
4082
- * `CROSS_CALL_ISSUE_ONSET_BURST`. Default `2_000`.
4311
+ * Onsets falling within this span (ms) escalate the finding to `CROSS_CALL_ISSUE_ONSET_BURST`.
4312
+ * Default `2_000`.
4313
+ *
4314
+ * This is the strongest evidence the detector produces: independent calls starting to fail *at the
4315
+ * same instant* has no explanation other than something they share. Keep it at or above your
4316
+ * sampling period — onsets are only as precise as clients report them — and no wider than a few
4317
+ * seconds, or unrelated failures drift into the same window. Typical `1_000`–`5_000`.
4083
4318
  */
4084
4319
  onsetBurstWindowInMs: number;
4085
- /** Re-arm time (ms) per issue type. Default `60_000`. */
4320
+ /**
4321
+ * Re-arm time per issue type (ms). Default `60_000`.
4322
+ *
4323
+ * Typical `60_000`–`600_000`. A fleet-wide event is one incident: without a cooldown a sustained
4324
+ * outage would raise an issue on every tick for its whole duration.
4325
+ */
4086
4326
  cooldownMs: number;
4087
4327
  };
4088
4328
  /** What the detector currently knows about one issue type across the fleet. */
@@ -4175,13 +4415,38 @@ type IssueFanOutDetectorConfig = {
4175
4415
  * hardware and would be a false accusation.
4176
4416
  */
4177
4417
  issueTypes: string[];
4178
- /** Minimum receivers of the track before a ratio is meaningful. Default `3`. */
4418
+ /**
4419
+ * Receivers a track needs before a ratio means anything. Default `3`.
4420
+ *
4421
+ * With two receivers, "60% affected" is one of them — which is the single-receiver case below, not
4422
+ * a fan-out. Sensible range `2`–`5`; in small calls a published track rarely has more than a couple
4423
+ * of subscribers, so raising this can silence the detector entirely.
4424
+ */
4179
4425
  minReceivers: number;
4180
- /** Fraction of a track's receivers that must share the issue. Default `0.6`. */
4426
+ /**
4427
+ * Fraction of a track's receivers that must share the issue, `0`–`1`. Default `0.6`.
4428
+ *
4429
+ * The higher this is, the more the finding points at the publisher rather than at the network
4430
+ * between: *everyone* receiving this track badly is hard to explain any other way. Typical
4431
+ * `0.5`–`0.8`. Below `0.5` you are reporting "some receivers", which usually means their own
4432
+ * last miles.
4433
+ */
4181
4434
  affectedRatioThreshold: number;
4182
- /** Also report the "only one receiver is affected" case. Default `true`. */
4435
+ /**
4436
+ * Also report when exactly one receiver is affected. Default `true`.
4437
+ *
4438
+ * Kept on because the finding is *useful and correctly weaker*: it is raised with a lower
4439
+ * confidence and the opposite conclusion — one unhappy receiver out of eight points at that
4440
+ * receiver, not at the publisher. Turn it off if you only want publisher-blaming findings and
4441
+ * treat single-receiver trouble as the client's own business.
4442
+ */
4183
4443
  reportSingleReceiver: boolean;
4184
- /** Re-arm time (ms) per (track, issue type). Default `60_000`. */
4444
+ /**
4445
+ * Re-arm time per (track, issue type) in ms. Default `60_000`.
4446
+ *
4447
+ * Per track, so a call with many bad publishers still reports each of them. Typical
4448
+ * `30_000`–`300_000`.
4449
+ */
4185
4450
  cooldownMs: number;
4186
4451
  };
4187
4452
  /**
@@ -4272,14 +4537,76 @@ type SfuCongestionDetectorReport = {
4272
4537
  relativeIncrease: number;
4273
4538
  };
4274
4539
  type SfuCongestionDetectorConfig = {
4540
+ /**
4541
+ * Client issue types counted as "congestion" for this indicator. Default `[ 'congestion' ]`.
4542
+ *
4543
+ * Keep this narrow. Every type you add widens what counts as a congested client, and the whole
4544
+ * method rests on comparing *like with like* across time buckets — mixing in a type that appears
4545
+ * for unrelated reasons raises the baseline and buries the spike you are looking for.
4546
+ */
4275
4547
  consumedClientIssueTypes: string[];
4548
+ /** The `observer-issue` type raised when a bucket is judged congested. Default `'sfu-congestion'`. */
4276
4549
  emittedObserverIssueType: string;
4550
+ /**
4551
+ * How long your clients take to send a sample (ms). Default `10_000`.
4552
+ *
4553
+ * **Set this to your collector's actual sampling period** — it is a description of your clients,
4554
+ * not a tuning knob. The bucket is `samplesSendingTimeInMs * 2`, so every client gets a fair
4555
+ * chance to report at least once inside each bucket. Set it too short and clients that simply had
4556
+ * not reported yet look absent, so the ratio jumps around on sampling noise; too long and the
4557
+ * detector reacts slowly and averages a spike away.
4558
+ */
4277
4559
  samplesSendingTimeInMs: number;
4560
+ /**
4561
+ * Completed buckets kept as history — the candidate plus its baseline. Default `30`.
4562
+ *
4563
+ * At the default bucket size this is ~10 minutes of baseline. Sensible range `10`–`60`. Longer is
4564
+ * more robust to a single odd bucket but slower to accept a genuinely changed normal (a growth
4565
+ * spurt, a new region coming online); shorter adapts quickly but lets a sustained problem become
4566
+ * the new baseline and stop being reported.
4567
+ */
4278
4568
  historySize: number;
4569
+ /**
4570
+ * Buckets required before any judgement is made. Default `5`.
4571
+ *
4572
+ * Below this the detector is silent, which is the point: a median and MAD over two buckets is not
4573
+ * a baseline. Costs `minHistorySize * samplesSendingTimeInMs * 2` of warm-up after start — about
4574
+ * 100 s at the defaults. Do not lower it to make a test fire faster; shorten the bucket instead.
4575
+ */
4279
4576
  minHistorySize: number;
4577
+ /**
4578
+ * Distinct congested clients required in the candidate bucket. Default `3`.
4579
+ *
4580
+ * The absolute floor beneath every ratio below, so that a tiny fleet cannot produce a finding: two
4581
+ * unhappy clients out of four is 50% and means nothing. Raise it on a large fleet where three
4582
+ * clients is always noise.
4583
+ */
4280
4584
  minAffectedClients: number;
4585
+ /**
4586
+ * How far the candidate's congested-client **ratio** must exceed the baseline median, in absolute
4587
+ * terms (`0`–`1`). Default `0.05`, i.e. five percentage points.
4588
+ *
4589
+ * This is the practical-significance gate: it stops a statistically striking move from 0.5% to 2%
4590
+ * being reported as an event. Typical `0.03`–`0.15`.
4591
+ */
4281
4592
  minAbsoluteRatioIncrease: number;
4593
+ /**
4594
+ * How many times the baseline median the candidate ratio must reach. Default `2`.
4595
+ *
4596
+ * Multiplicative counterpart to the absolute gate — both must pass. Typical `1.5`–`3`. Below
4597
+ * `1.5` ordinary fluctuation qualifies; above ~`4` only near-total events do.
4598
+ */
4282
4599
  minRelativeRatioIncrease: number;
4600
+ /**
4601
+ * Robust z-score the candidate must reach against a median+MAD baseline. Default `3`.
4602
+ *
4603
+ * The statistical-significance gate. `3` is the conventional "clearly outside normal variation";
4604
+ * `2` is noticeably chattier, `4`–`5` only for very stable fleets. Median and MAD rather than mean
4605
+ * and standard deviation on purpose — a couple of past incidents in the history would inflate a
4606
+ * standard deviation enough to hide the next one. Note that a perfectly flat baseline gives
4607
+ * `MAD = 0`, where any increase scores `Infinity`; the two ratio gates above are what keep that
4608
+ * honest.
4609
+ */
4283
4610
  robustZThreshold: number;
4284
4611
  };
4285
4612
  /** The statistical/practical-significance verdict for one candidate bucket against its baseline. */
@@ -4402,15 +4729,35 @@ declare const TrackDeliveryMismatchTypes: {
4402
4729
  readonly publisherTrackDry: "PUBLISHER_TRACK_DRY";
4403
4730
  };
4404
4731
  type TrackDeliveryMismatchDetectorConfig = {
4405
- /** The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`. */
4732
+ /**
4733
+ * The receiver-side issue type meaning "no media arriving". Default `'dry-inbound-track'`, which is
4734
+ * what client-monitor-js raises. Only change it if you raise your own equivalent.
4735
+ */
4406
4736
  dryInboundIssueType: string;
4407
- /** The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`. */
4737
+ /**
4738
+ * The publisher-side issue type meaning "not producing". Default `'dry-outbound-track'`, as raised
4739
+ * by client-monitor-js. The pairing of these two types is the whole detector: their **disagreement**
4740
+ * is the finding.
4741
+ */
4408
4742
  dryOutboundIssueType: string;
4409
- /** Minimum subscribers before "all of them" means anything. Default `2`. */
4743
+ /**
4744
+ * Subscribers required before "all of them" means anything. Default `2`.
4745
+ *
4746
+ * With one subscriber, "every receiver is dry" is a single client's report and carries no more
4747
+ * weight than the client issue already does. Sensible range `2`–`4`.
4748
+ */
4410
4749
  minReceivers: number;
4411
- /** Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`. */
4750
+ /**
4751
+ * Fraction of subscribers that must be dry to call it a whole-track delivery failure. Default `1`.
4752
+ *
4753
+ * `1` — literally all of them — on purpose. The inference here is sharp: the publisher says it is
4754
+ * sending and *every* receiver says nothing arrives, so the fault is between them, in the SFU's
4755
+ * forwarding. Lowering it to `0.8` admits mixed evidence, where some receivers do get the media, and
4756
+ * the conclusion no longer follows: that is a per-receiver problem and `IssueFanOutDetector`'s
4757
+ * question. Do not lower it without deciding what the finding then means.
4758
+ */
4412
4759
  allReceiversRatio: number;
4413
- /** Re-arm time (ms) per (track, verdict). Default `60_000`. */
4760
+ /** Re-arm time per (track, verdict) in ms. Default `60_000`. Typical `30_000`–`300_000`. */
4414
4761
  cooldownMs: number;
4415
4762
  };
4416
4763
  /**
@@ -4461,19 +4808,47 @@ declare const TurnServerHealthTypes: {
4461
4808
  readonly turnServerDegraded: "TURN_SERVER_DEGRADED";
4462
4809
  };
4463
4810
  type TurnServerHealthDetectorConfig = {
4464
- /** Minimum clients on a server before a ratio is meaningful. Default `5`. */
4811
+ /**
4812
+ * Clients a server must be carrying before its ratio means anything. Default `5`.
4813
+ *
4814
+ * With two relayed clients, "half are degraded" is one person having a bad time. Sensible range
4815
+ * `5`–`20`; raise it if you run many small TURN deployments, since each needs enough traffic to be
4816
+ * measurable on its own.
4817
+ */
4465
4818
  minClientsPerServer: number;
4466
- /** Fraction of a server's clients that must have an open issue. Default `0.5`. */
4819
+ /**
4820
+ * Fraction of a server's clients that must have an open issue, `0`–`1`. Default `0.5`.
4821
+ *
4822
+ * Typical `0.4`–`0.7`. Remember each finding also carries the *other* servers' ratios, so the
4823
+ * threshold is not doing the comparison on its own — a server at 50% next to peers at 45% reads very
4824
+ * differently from one next to peers at 3%. Below `0.3` you will report servers that are merely
4825
+ * carrying unlucky clients.
4826
+ */
4467
4827
  degradedRatioThreshold: number;
4468
4828
  /**
4469
- * Which client issue types count as "in trouble". Empty (default) means **any** open issue —
4470
- * appropriate here, because the question is not *what* is wrong with each client but whether
4471
- * trouble clusters on one relay.
4829
+ * Which client issue types count as "in trouble". Empty (the default) means **any** open issue.
4830
+ *
4831
+ * The permissive default is deliberate and unusual for this library: the question is not *what* is
4832
+ * wrong with each client but whether trouble clusters on one relay, and a relay problem shows up as
4833
+ * whatever symptom each client happens to notice first. Narrow it to network types
4834
+ * (`congestion`, `ice-disconnected`) if endpoint issues like `cpulimitation` are common enough in
4835
+ * your fleet to blur the comparison between servers.
4472
4836
  */
4473
4837
  issueTypes: string[];
4474
- /** Consecutive ticks the condition must hold before raising. Default `2`. */
4838
+ /**
4839
+ * Consecutive `observer.update()` ticks the condition must hold before raising. Default `2`.
4840
+ *
4841
+ * The de-bounce. `1` reacts immediately and will fire on a single tick where several clients
4842
+ * happened to be mid-reconnect; `2`–`3` costs a tick or two of delay and removes most of that.
4843
+ * Note this counts *ticks*, not time, so how long it actually waits depends on your sample rate.
4844
+ */
4475
4845
  consecutiveTicks: number;
4476
- /** Re-arm time (ms) before raising again for the same server. Default `60_000`. */
4846
+ /**
4847
+ * Re-arm time per server (ms). Default `60_000`.
4848
+ *
4849
+ * Shorter than the outage detector's, because degradation is a condition you may want re-reported as
4850
+ * it persists or worsens, not a single event. Typical `60_000`–`300_000`.
4851
+ */
4477
4852
  cooldownMs: number;
4478
4853
  };
4479
4854
  /** The per-server view this detector builds. */
@@ -4526,8 +4901,11 @@ declare const TurnServerOutageTypes: {
4526
4901
  };
4527
4902
  type TurnServerOutageDetectorConfig = {
4528
4903
  /**
4529
- * How many clients a server must have been carrying at its peak before its collapse means
4530
- * anything. Below this, one or two people leaving looks like an outage. Default `5`.
4904
+ * Clients a server must have been carrying at its peak before its collapse means anything. Default
4905
+ * `5`.
4906
+ *
4907
+ * Below this, one or two people leaving looks like an outage. Sensible range `5`–`50`; the higher it
4908
+ * is the more confident the finding, and the more small deployments go unwatched.
4531
4909
  */
4532
4910
  minClientsAtPeak: number;
4533
4911
  /**
@@ -4536,8 +4914,11 @@ type TurnServerOutageDetectorConfig = {
4536
4914
  */
4537
4915
  lossRatioThreshold: number;
4538
4916
  /**
4539
- * Window (ms) the peak population is measured over. Long enough to span a real outage's onset,
4540
- * short enough that yesterday's peak isn't held against today. Default `120_000`.
4917
+ * Window the peak population is measured over (ms). Default `120_000`.
4918
+ *
4919
+ * Long enough to span a real outage's onset, short enough that yesterday's peak is not held against
4920
+ * today. Typical `60_000`–`600_000`. Too long and the natural end of a busy period reads as a
4921
+ * collapse; too short and a gradual failure never shows a peak to fall from.
4541
4922
  */
4542
4923
  peakWindowMs: number;
4543
4924
  /**
@@ -4546,11 +4927,28 @@ type TurnServerOutageDetectorConfig = {
4546
4927
  * event, or the observer shutting down all look exactly like a TURN outage. Default `true`.
4547
4928
  */
4548
4929
  requireControlGroup: boolean;
4549
- /** Minimum clients elsewhere before the control group is statistically worth anything. Default `5`. */
4930
+ /**
4931
+ * Clients elsewhere before the control group is worth anything. Default `5`.
4932
+ *
4933
+ * If you run a single TURN server there is never a control group, so with `requireControlGroup: true`
4934
+ * this detector can never fire — which is correct rather than unfortunate: with one relay you cannot
4935
+ * distinguish "the relay died" from "everyone went home". Sensible range `5`–`20`.
4936
+ */
4550
4937
  minControlGroupClients: number;
4551
- /** Fraction of the control group that must still be healthy. Default `0.7`. */
4938
+ /**
4939
+ * Fraction of the control group that must still be healthy, `0`–`1`. Default `0.7`.
4940
+ *
4941
+ * The evidence that the rest of the world is fine. Typical `0.6`–`0.9`. Set it too high and a
4942
+ * concurrent unrelated problem elsewhere masks a real outage; too low and a fleet-wide network event
4943
+ * gets blamed on whichever server lost clients first.
4944
+ */
4552
4945
  controlGroupHealthyRatio: number;
4553
- /** Consecutive ticks the condition must hold before raising. Default `2`. */
4946
+ /**
4947
+ * Consecutive `observer.update()` ticks the condition must hold before raising. Default `2`.
4948
+ *
4949
+ * Counts ticks, not time. `1` will fire on a single tick where a batch of clients happened to be
4950
+ * between samples; `2`–`4` is the useful range for something this consequential to declare.
4951
+ */
4554
4952
  consecutiveTicks: number;
4555
4953
  /**
4556
4954
  * Re-arm time (ms) per server. Long by default (`300_000`) — an outage is one event, not one
@@ -4631,11 +5029,31 @@ declare const UnconsumedTrackTypes: {
4631
5029
  readonly unconsumedPublishedTrack: "UNCONSUMED_PUBLISHED_TRACK";
4632
5030
  };
4633
5031
  type UnconsumedTrackDetectorConfig = {
4634
- /** How long a track must stay unconsumed while sending before reporting (ms). Default `30_000`. */
5032
+ /**
5033
+ * How long a track must stay unconsumed **while still sending** before it is reported (ms).
5034
+ * Default `30_000`.
5035
+ *
5036
+ * This is the main guard against a false alarm, because a gap between publishing and the first
5037
+ * subscription is completely normal at join time — and again after every renegotiation. Sensible
5038
+ * range `15_000`–`120_000`. Too low and you report every join; too high and you tolerate wasted
5039
+ * uplink for longer than you need to. Waste is not an outage, so err high.
5040
+ */
4635
5041
  minUnconsumedDurationInMs: number;
4636
- /** Ignore tracks below this bitrate — a trickle isn't worth an alert (bps). Default `50_000`. */
5042
+ /**
5043
+ * Ignore tracks sending below this bitrate (**bits per second**). Default `50_000` (50 kbps).
5044
+ *
5045
+ * The point of the detector is wasted bandwidth, and a track trickling keep-alive packets wastes
5046
+ * none worth an alert. Typical `20_000`–`100_000`: muted or paused tracks sit near zero, a real
5047
+ * video track is hundreds of kbps. Set it to `0` to report every unconsumed track regardless of
5048
+ * cost.
5049
+ */
4637
5050
  minBitrate: number;
4638
- /** Re-arm time (ms) per track. Default `300_000`. */
5051
+ /**
5052
+ * Re-arm time per track (ms). Default `300_000`.
5053
+ *
5054
+ * Long on purpose: an unconsumed track usually *stays* unconsumed, so a short cooldown means a
5055
+ * steady drip of the same finding for the life of the call. Typical `300_000`–`900_000`.
5056
+ */
4639
5057
  cooldownMs: number;
4640
5058
  };
4641
5059
  /**
@@ -5012,17 +5430,57 @@ type SimulcastReceiverReportPayload = ({
5012
5430
  checks: number;
5013
5431
  };
5014
5432
  type SimulcastReceiverValidatorConfig = {
5015
- /** Minimum receivers of a published track before the comparison means anything. */
5433
+ /**
5434
+ * Receivers a published track needs before the comparison means anything. Default `3`.
5435
+ *
5436
+ * The question is whether *one* receiver drags *the others* down, which needs at least one other to
5437
+ * be dragged — so `3` gives a worst receiver plus two to compare against. `2` is the technical
5438
+ * minimum but makes "median of the others" a single number. Sensible range `3`–`5`.
5439
+ */
5016
5440
  minReceivers: number;
5017
- /** How long a track's bitrates are correlated over (ms). */
5441
+ /**
5442
+ * How long bitrates are correlated over (ms). Default `10_000`.
5443
+ *
5444
+ * Long enough to contain a real adaptation response — the publisher's encoder reacting to a
5445
+ * bandwidth estimate takes seconds, not milliseconds. Typical `10_000`–`30_000`. Too short and you
5446
+ * catch transient jitter rather than a sustained relationship; too long and a genuine change is
5447
+ * averaged out by the healthy period around it.
5448
+ */
5018
5449
  windowMs: number;
5019
- /** Samples needed inside the window before it can be judged. */
5450
+ /**
5451
+ * Samples needed inside the window before it can be judged. Default `5`.
5452
+ *
5453
+ * A correlation over two or three points is meaningless. Combined with `windowMs` this implies a
5454
+ * sampling period: 5 samples in 10 s needs clients reporting at least every ~2 s. If your collector
5455
+ * is slower, widen `windowMs` rather than lowering this.
5456
+ */
5020
5457
  minSamples: number;
5021
- /** The worst receiver must be at most this share of the median, or there is nothing to be dragged by. */
5458
+ /**
5459
+ * The worst receiver must be at most this share of the median receiver, `0`–`1`. Default `0.5`.
5460
+ *
5461
+ * The precondition, not the finding: unless somebody is genuinely doing much worse than the rest,
5462
+ * there is nothing for the publisher to be dragged *by* and the check has nothing to look at.
5463
+ * Typical `0.4`–`0.7`. Higher makes the check run more often on weaker evidence; lower means it
5464
+ * rarely finds a qualifying situation at all.
5465
+ */
5022
5466
  outlierRatioThreshold: number;
5023
- /** How closely the publisher must follow the worst receiver to count as dragged. */
5467
+ /**
5468
+ * How closely the publisher must track the worst receiver to count as dragged, `0`–`1`. Default
5469
+ * `0.8`.
5470
+ *
5471
+ * This is the finding: the publisher sending at ≥80% of the *worst* receiver's rate means it has
5472
+ * collapsed to the lowest common denominator instead of serving everyone else properly. Typical
5473
+ * `0.7`–`0.9`. Toward `1` you only catch total collapse; below ~`0.6` normal encoder behaviour can
5474
+ * look like dragging.
5475
+ */
5024
5476
  trackingRatioThreshold: number;
5025
- /** Clean checks required before concluding per-receiver adaptation. One could be luck. */
5477
+ /**
5478
+ * Clean checks required before concluding per-receiver adaptation works. Default `3`.
5479
+ *
5480
+ * One clean check could be luck — the qualifying moment might simply not have been bad enough. This
5481
+ * is what stops a lucky sample from being reported as a pass, so raising it strengthens the verdict
5482
+ * at the cost of taking longer to reach one. Typical `3`–`10`.
5483
+ */
5026
5484
  minChecks: number;
5027
5485
  };
5028
5486
  /**
@@ -5140,13 +5598,38 @@ type RemoteTrackResolverReportPayload = ({
5140
5598
  checks: number;
5141
5599
  };
5142
5600
  type RemoteTrackResolverValidatorConfig = {
5143
- /** Participants a call needs before it can plausibly have publisher↔subscriber links. Default `2`. */
5601
+ /**
5602
+ * Participants a call needs before it can plausibly have publisher↔subscriber links. Default `2`.
5603
+ *
5604
+ * A one-person call has nobody to subscribe to anyone, so counting it would dilute the ratio with
5605
+ * calls that *could not* have produced a link. `2` is the true minimum here and there is little
5606
+ * reason to raise it.
5607
+ */
5144
5608
  minClients: number;
5145
- /** Inbound tracks that must be present in a call before it counts as a check. Default `2`. */
5609
+ /**
5610
+ * Inbound tracks a call must have before it counts as a check. Default `2`.
5611
+ *
5612
+ * Same idea: no subscribed tracks means nothing to link. Sensible range `2`–`5`.
5613
+ */
5146
5614
  minInboundTracks: number;
5147
- /** Share of inbound tracks that must be linked to conclude the resolver works. Default `0.5`. */
5615
+ /**
5616
+ * Share of inbound tracks that must be linked to conclude the resolver works, `0`–`1`. Default
5617
+ * `0.5`.
5618
+ *
5619
+ * Deliberately lenient, because a partially-linked call is normal: tracks arrive before their
5620
+ * publisher is known, and simulcast layers or probing streams may have no publisher at all. The
5621
+ * question is "is this resolver wired up", not "is every track linked". Typical `0.3`–`0.7`. Raising
5622
+ * it toward `1` turns the check into a strictness audit and it will report failure on healthy
5623
+ * systems.
5624
+ */
5148
5625
  linkedRatioThreshold: number;
5149
- /** Eligible calls to observe before concluding either way. One could be a race. Default `3`. */
5626
+ /**
5627
+ * Eligible calls to observe before concluding either way. Default `3`.
5628
+ *
5629
+ * One call could be a race — every track happening to arrive before its publisher. Typical `3`–`10`.
5630
+ * Note that a low value makes a *pass* less trustworthy than a failure: linking nothing repeatedly is
5631
+ * conclusive, linking things once might be luck.
5632
+ */
5150
5633
  minChecks: number;
5151
5634
  };
5152
5635
  /**
@@ -5255,11 +5738,27 @@ type CodecConsistencyValidatorConfig = {
5255
5738
  * case where everyone agrees on the wrong thing — a silent fallback that nothing else notices.
5256
5739
  */
5257
5740
  expected?: Partial<Record<'audio' | 'video', string>>;
5258
- /** Which kinds to inspect. Default both. */
5741
+ /**
5742
+ * Which kinds to inspect. Default `[ 'audio', 'video' ]`.
5743
+ *
5744
+ * Narrow it when only one matters: audio codec splits are the ones that usually cost transcoding,
5745
+ * while video splits are more often a deliberate per-client decision.
5746
+ */
5259
5747
  kinds: ('audio' | 'video')[];
5260
- /** Participants a call needs before disagreement is meaningful. Default `3`. */
5748
+ /**
5749
+ * Participants a call needs before disagreement is meaningful. Default `3`.
5750
+ *
5751
+ * In a 1:1 call "the participants disagree" is two clients differing, which can be a legitimate
5752
+ * negotiation outcome rather than a fault. Sensible range `3`–`5`.
5753
+ */
5261
5754
  minClients: number;
5262
- /** Calls to inspect before concluding. Default `3`. */
5755
+ /**
5756
+ * Calls to inspect before concluding. Default `3`.
5757
+ *
5758
+ * A structural property of your negotiation, so a handful of calls is plenty — but one call could be
5759
+ * an unusual mix of participants. Typical `3`–`10`. Higher delays the verdict without adding much,
5760
+ * since the answer does not vary call to call.
5761
+ */
5263
5762
  minChecks: number;
5264
5763
  };
5265
5764
  /**
@@ -5439,8 +5938,38 @@ type ClientAppDataFactory = (params: {
5439
5938
  acceptCtx?: AcceptContext;
5440
5939
  }) => Record<string, unknown>;
5441
5940
  type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = {
5941
+ /**
5942
+ * Application-owned data for the observer itself. Never read or modified by the library.
5943
+ *
5944
+ * For per-call / per-client data prefer `createCallAppData` / `createClientAppData`, which run at
5945
+ * creation time and can see the `accept()` context.
5946
+ */
5442
5947
  appData?: AppData;
5948
+ /**
5949
+ * Close a client that has not produced a sample for this long (ms). Default `60_000`.
5950
+ *
5951
+ * This is the **liveness timeout for a participant**, so set it from your client's sampling
5952
+ * period, not from taste: a client sampling every 5 s needs several missed samples to look gone.
5953
+ * Sensible range `3x`–`10x` the sampling period; `60_000` suits the usual 2–10 s collectors.
5954
+ *
5955
+ * Too low and a client that merely paused (tab backgrounded, brief network drop) is closed and
5956
+ * then re-created as a *new* client, which restarts its detectors and splits one participant into
5957
+ * two in any summary. Too high and left participants linger, inflating `peak`, the denominators of
5958
+ * every ratio-based detector, and memory. `undefined` disables the timeout — then nothing closes
5959
+ * an abandoned client but your own `client.close()`.
5960
+ */
5443
5961
  closeClientIfIdleForMs?: number;
5962
+ /**
5963
+ * Close a call once it has had zero clients for this long (ms). Default `60_000`.
5964
+ *
5965
+ * The grace period exists so a brief empty moment — everyone reconnecting after a network blip,
5966
+ * the last participant refreshing — does not end the call and start a new one under the same
5967
+ * `callId`. Sensible range `10_000`–`300_000`.
5968
+ *
5969
+ * Too low splits one meeting into several calls, and each split emits its own `call-summary`. Too
5970
+ * high keeps dead calls in `observedCalls`, holding their detectors and summaries in memory.
5971
+ * `undefined` disables it: the call then lives until you call `call.close()`.
5972
+ */
5444
5973
  closeCallIfEmptyForMs?: number;
5445
5974
  /**
5446
5975
  * When `true` (the default), every call update triggers an observer-wide `update()` pass.
@@ -5481,15 +6010,75 @@ type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unk
5481
6010
  * say which. One shape for every call, or none.
5482
6011
  */
5483
6012
  callSummary?: Partial<CallSummaryConfig> | null;
6013
+ /**
6014
+ * Thresholds that mark a **received** track as degraded. Omitted (the default) means no track is
6015
+ * ever marked degraded — `inboundTrack.degraded` stays `false` and `degradationReasons` stays
6016
+ * empty, so this is opt-in and absence is *not* a clean bill of health.
6017
+ *
6018
+ * Each field is an **exclusive upper bound**: the reason is added when the measured value is
6019
+ * strictly greater. All are evaluated on every update and any number can fire at once;
6020
+ * `degradationReasons` lists the ones that did.
6021
+ *
6022
+ * These feed `CallHealthAggregator` and `outboundTrack.degradedRatio` (how many of a publisher's
6023
+ * receivers are unhappy), which is what `TrackDeliveryMismatchDetector` and
6024
+ * `PublisherFaultCorroborationDetector` read. They are **not** a substitute for client issues:
6025
+ * client-monitor already decides "this endpoint is in trouble" with hysteresis and multi-signal
6026
+ * confirmation. Treat these as a coarse per-track flag for cross-participant comparison, and keep
6027
+ * them loose enough that a single bad tick does not trip them.
6028
+ */
5484
6029
  inboundTrackDegradationThresholds?: {
6030
+ /**
6031
+ * Freezes counted **in one sampling period**, not since the start of the call. `1` means "any
6032
+ * freeze at all this tick", which is strict; `2`–`3` tolerates the odd frame hiccup.
6033
+ * Reason: `'freezes'`.
6034
+ */
5485
6035
  deltaFreezeCount: number;
6036
+ /**
6037
+ * Fraction of frames dropped, `0`–`1`. Typical `0.05`–`0.2`; below ~`0.02` you are inside
6038
+ * normal jitter for most decoders. Reason: `'frames-dropped'`.
6039
+ */
5486
6040
  framesDroppedRatio: number;
6041
+ /**
6042
+ * Jitter buffer delay in **ms**. Typical `150`–`500`: audio stays intelligible well past
6043
+ * `200`, while conversation turn-taking suffers beyond ~`400`. Reason:
6044
+ * `'jitter-buffer-delay'`.
6045
+ */
5487
6046
  jitterBufferDelayInMs: number;
6047
+ /**
6048
+ * Concealed (synthesised) audio samples as a fraction, `0`–`1`. Typical `0.05`–`0.15`; above
6049
+ * ~`0.1` is usually audible as robotic or clipped speech. Reason: `'concealment'`.
6050
+ */
5488
6051
  concealmentRatio: number;
6052
+ /**
6053
+ * Round-trip time in **ms**, as reported for this inbound stream. Typical `250`–`500`.
6054
+ * Remember this is absolute, not relative to the participant's own baseline — a genuinely
6055
+ * distant participant will sit permanently above any fixed bound, which is why "RTT got
6056
+ * worse" belongs to client-monitor and not here. Reason: `'rtt'`.
6057
+ */
5489
6058
  rttInMs: number;
5490
6059
  };
6060
+ /**
6061
+ * Thresholds that mark a **published** track as degraded, from what the receivers report back.
6062
+ * Omitted (the default) means neither threshold-based reason can fire.
6063
+ *
6064
+ * Note two reasons are added regardless of this setting, because they need no threshold:
6065
+ * `quality-limited-<reason>` when the encoder reports a `qualityLimitationReason`, and
6066
+ * `no-packets-sent` when an unmuted track sent nothing in a sampling period.
6067
+ *
6068
+ * Also an **exclusive upper bound** per field, evaluated on every update.
6069
+ */
5491
6070
  outboundTrackDegradationThresholds?: {
6071
+ /**
6072
+ * Fraction of packets lost as reported by the remote end, `0`–`1`. Typical `0.02`–`0.1`;
6073
+ * anything under ~`0.01` is normal on the open internet and will fire constantly. Reason:
6074
+ * `'remote-fraction-lost'`.
6075
+ */
5492
6076
  fractionLost: number;
6077
+ /**
6078
+ * Round-trip time in **ms** as reported by the remote end. Typical `250`–`500`, and absolute
6079
+ * rather than baseline-relative — see the note on the inbound `rttInMs`. Reason:
6080
+ * `'remote-rtt'`.
6081
+ */
5493
6082
  rttInMs: number;
5494
6083
  };
5495
6084
  /**
@@ -5861,14 +6450,43 @@ declare class SlidingWindow<T> {
5861
6450
 
5862
6451
  type TrendTesterConfig = {
5863
6452
  /**
5864
- * How many of the most recent values to keep. The one knob that controls how far back "trend"
5865
- * looks, for both tests — there is deliberately no separate window per test.
6453
+ * How many of the most recent values to keep. Default `30`, floored at `2`.
6454
+ *
6455
+ * The one knob controlling how far back "trend" looks, for both tests — there is deliberately no
6456
+ * separate window per test. This counts *samples*, so the time it spans depends on how often you
6457
+ * push. Mann-Kendall needs roughly 8–10 points before its significance test is worth anything, so
6458
+ * below ~`10` it will mostly answer `no-trend`. Typical `20`–`60`: long enough for a stable
6459
+ * baseline, short enough that a sustained change eventually becomes the new normal instead of being
6460
+ * flagged forever.
5866
6461
  */
5867
6462
  size?: number;
5868
- /** Page-Hinkley's drift tolerance and detection threshold — see {@link pageHinkley}. */
6463
+ /**
6464
+ * Page-Hinkley's **drift tolerance** — change smaller than this is treated as noise and never
6465
+ * accumulated. Default `0`. See {@link pageHinkley}.
6466
+ *
6467
+ * Expressed in the units of whatever you push, so there is no universally good value: for RTT in ms
6468
+ * a few ms is a reasonable tolerance. The default `0` accumulates *every* deviation, which is the
6469
+ * most sensitive setting. Too low and ordinary fluctuation accumulates into a false step change; too
6470
+ * high and a real but gradual shift never accumulates at all.
6471
+ */
5869
6472
  pageHinkleyDelta?: number;
6473
+ /**
6474
+ * Page-Hinkley's **detection threshold** — how much accumulated drift counts as a step change.
6475
+ * Default `50`. See {@link pageHinkley}.
6476
+ *
6477
+ * Also in your units, and the direct sensitivity control: lower detects smaller or earlier steps and
6478
+ * false-positives more; higher waits for unmistakable ones. Worth tuning against a recorded series
6479
+ * rather than by intuition, because the right value depends entirely on the scale and noisiness of
6480
+ * the metric you feed it.
6481
+ */
5870
6482
  pageHinkleyLambda?: number;
5871
- /** Mann-Kendall's significance level — see {@link mannKendallVerdict}. */
6483
+ /**
6484
+ * Mann-Kendall's significance level. Default `0.05`. See {@link mannKendallVerdict}.
6485
+ *
6486
+ * Conventional values are `0.01`, `0.05` and `0.1`. This is the probability of claiming a monotonic
6487
+ * trend that is not really there: `0.01` is stricter and slower to call a trend, `0.1` more
6488
+ * sensitive and noisier.
6489
+ */
5872
6490
  mannKendallAlpha?: number;
5873
6491
  };
5874
6492
  /**
@@ -6043,4 +6661,4 @@ declare function createInMemorySink(samples?: ClientSample[]): InMemorySink;
6043
6661
  declare function createDefaultMediasoupRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
6044
6662
  declare function createP2pRemoteTrackResolverFactory(): RemoteTrackResolverFactory;
6045
6663
 
6046
- export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, type ActiveIssueTracker, ActiveIssuesRegistry, type AvailableCallScopeDetectorsConfigs, type AvailableDetectorsConfigs, type AvailableObserverScopeDetectorsConfigs, type AvailableValidatorConfigs, CODEC_MISMATCH_ISSUE, type CallAppDataFactory, CallConcurrentIssueDetector, type CallConcurrentIssueDetectorConfig, type CallConcurrentIssueGroup, CallConcurrentIssueTypes, type CallHealth, CallHealthAggregator, type CallIssue, type CallIssueSpread, type CallScopedEventName, type CallSummary, type CallSummaryClients, CallSummaryCollector, type CallSummaryConfig, type CallSummaryEnricher, type CallSummaryEnrichers, type CallSummaryScores, type CallSummarySection, type CallSummaryTruncation, type CallSummaryTurnServers, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientMetaData, ClientMetaTypes, type ClientPopulation, type ClientPopulationAxis, ClientPopulationIssueDetector, type ClientPopulationIssueDetectorConfig, ClientPopulationIssueTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type CodecConsistencyReportPayload, CodecConsistencyValidator, type CodecConsistencyValidatorConfig, type CodecEvidence, type CorroboratedPublisherFault, type Detector, Detectors, InMemorySink, type Issue, type IssueBase, type IssueConclusion, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, type IssueFaultDomain, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, LOWEST_COMMON_DENOMINATOR_ISSUE, type Logger, type MannKendallResult, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, ObservedClientIssueRegistry, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, ObserverConcurrentIssueDetector, type ObserverConcurrentIssueDetectorConfig, type ObserverConcurrentIssueGroup, ObserverConcurrentIssueTypes, type ObserverEventBase, type ObserverEvents, type ObserverIssue, type ObserverIssueSpread, type ObserverLogger, type PageHinkleyResult, PublisherFaultCorroborationDetector, type PublisherFaultCorroborationDetectorConfig, PublisherFaultTypes, RESOLVED_ISSUE_SUFFIX, type RemoteTrackLinkEvidence, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolverReportPayload, RemoteTrackResolverValidator, type RemoteTrackResolverValidatorConfig, type RemoteTrackResolvers, type ResolvedActiveClientIssue, type RunningValidator, type SampleRejectedReason, type ScoreCalculator, SfuCongestionDetector, type SfuCongestionDetectorBucket, type SfuCongestionDetectorConfig, type SfuCongestionDetectorEvaluation, type SfuCongestionDetectorReport, type SimulcastReceiverEvidence, type SimulcastReceiverReportPayload, SimulcastReceiverValidator, type SimulcastReceiverValidatorConfig, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrendTester, type TrendTesterConfig, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, TurnServerOutageDetector, type TurnServerOutageDetectorConfig, TurnServerOutageTypes, UNRESOLVED_TRACK_LINKS_ISSUE, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, type ValidationReport, type Validator, type ValidatorName, baseIssueType, concludeCallIssue, concludeObserverIssue, correlation, counterDelta, createCallSummary, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultCallSummaryConfig, defaultClientHealthThresholds, isClientIssueResolutionEntry, issuePayloadAsString, mannKendall, mannKendallVerdict, median, medianAbsoluteDeviation, pageHinkley, percentile, percentileOfSorted, robustZScore, setObserverLogger, summarize };
6664
+ export { type AcceptContext, type AcceptMiddleware, type AcceptMiddlewarePayload, type ActiveClientIssue, type ActiveIssueTracker, ActiveIssuesRegistry, type AvailableCallScopeDetectorsConfigs, type AvailableDetectorsConfigs, type AvailableObserverScopeDetectorsConfigs, type AvailableValidatorConfigs, CODEC_MISMATCH_ISSUE, type CallAppDataFactory, CallConcurrentIssueDetector, type CallConcurrentIssueDetectorConfig, type CallConcurrentIssueGroup, CallConcurrentIssueTypes, type CallHealth, CallHealthAggregator, type CallIssue, type CallIssueSpread, type CallScopedEventName, type CallSummary, type CallSummaryClients, CallSummaryCollector, type CallSummaryConfig, type CallSummaryEnricher, type CallSummaryEnrichers, type CallSummaryScores, type CallSummarySection, type CallSummaryTruncation, type CallSummaryTurnServers, type ClientAppDataFactory, type ClientEvent, ClientEventTypes, type ClientHealth, type ClientHealthThresholds, type ClientIssue, type ClientLocation, type ClientLocationResolver, type ClientMetaData, ClientMetaTypes, type ClientPopulation, type ClientPopulationAxis, ClientPopulationIssueDetector, type ClientPopulationIssueDetectorConfig, ClientPopulationIssueTypes, type ClientSample, ClientSampleSink, type ClientSampleSinkEvents, type ClientSampleSinkFactory, type CodecConsistencyReportPayload, CodecConsistencyValidator, type CodecConsistencyValidatorConfig, type CodecEvidence, type CorroboratedPublisherFault, type Detector, Detectors, GEOHASH_CELL_SIZES, InMemorySink, type Issue, type IssueBase, type IssueConclusion, IssueFanOutDetector, type IssueFanOutDetectorConfig, IssueFanOutTypes, type IssueFaultDomain, JsonlFileSink, type JsonlFileSinkFactoryOptions, type JsonlFileSinkOptions, LOWEST_COMMON_DENOMINATOR_ISSUE, type Logger, type MannKendallResult, type MediasoupConsumerSample, type MediasoupConsumerSampleEvent, type MediasoupDataConsumerSample, type MediasoupDataProducerSample, type MediasoupDirectTransportSample, type MediasoupDirectTransportSampleEventMap, type MediasoupPipeTransportSample, type MediasoupPipeTransportSampleEventMap, type MediasoupPlainTransportSample, type MediasoupPlainTransportSampleEventMap, type MediasoupProducerSample, type MediasoupProducerSampleEvent, type MediasoupRouterSample, type MediasoupSampleEnricher, type MediasoupTransportSample, type MediasoupWebRtcTransportSample, type MediasoupWebRtcTransportSampleEventMap, type Middleware, ObservedCall, type ObservedCallScope, ObservedCertificate, ObservedClient, ObservedClientIssueRegistry, type ObservedClientScope, ObservedCodec, ObservedDataChannel, ObservedIceCandidate, ObservedIceCandidatePair, ObservedIceTransport, ObservedInboundRtp, ObservedInboundTrack, ObservedMediaPlayout, ObservedMediaSource, ObservedMediasoupRouter, type ObservedMediasoupRouterEvents, type ObservedMediasoupRouterScope, type ObservedMediasoupRouterSettings, ObservedOutboundRtp, ObservedOutboundTrack, ObservedPeerConnection, type ObservedPeerConnectionScope, ObservedPeerConnectionTransport, ObservedRemoteInboundRtp, ObservedRemoteOutboundRtp, Observer, ObserverConcurrentIssueDetector, type ObserverConcurrentIssueDetectorConfig, type ObserverConcurrentIssueGroup, ObserverConcurrentIssueTypes, type ObserverEventBase, type ObserverEvents, type ObserverIssue, type ObserverIssueSpread, type ObserverLogger, type PageHinkleyResult, PublisherFaultCorroborationDetector, type PublisherFaultCorroborationDetectorConfig, PublisherFaultTypes, RESOLVED_ISSUE_SUFFIX, type RemoteTrackLinkEvidence, RemoteTrackResolver, type RemoteTrackResolverFactory, type RemoteTrackResolverReportPayload, RemoteTrackResolverValidator, type RemoteTrackResolverValidatorConfig, type RemoteTrackResolvers, type ResolvedActiveClientIssue, type RunningValidator, type SampleRejectedReason, type ScoreCalculator, SfuCongestionDetector, type SfuCongestionDetectorBucket, type SfuCongestionDetectorConfig, type SfuCongestionDetectorEvaluation, type SfuCongestionDetectorReport, type SimulcastReceiverEvidence, type SimulcastReceiverReportPayload, SimulcastReceiverValidator, type SimulcastReceiverValidatorConfig, SlidingWindow, type SlidingWindowEntry, type StatsSummary, TrackDeliveryMismatchDetector, type TrackDeliveryMismatchDetectorConfig, TrackDeliveryMismatchTypes, TrendTester, type TrendTesterConfig, type TurnServerHealth, TurnServerHealthDetector, type TurnServerHealthDetectorConfig, TurnServerHealthTypes, TurnServerOutageDetector, type TurnServerOutageDetectorConfig, TurnServerOutageTypes, UNRESOLVED_TRACK_LINKS_ISSUE, UnconsumedTrackDetector, type UnconsumedTrackDetectorConfig, UnconsumedTrackTypes, type ValidationReport, type Validator, type ValidatorName, baseIssueType, concludeCallIssue, concludeObserverIssue, correlation, counterDelta, createCallSummary, createDefaultMediasoupRemoteTrackResolverFactory, createInMemorySink, createJsonlFileSink, createJsonlFileSinkFactory, createLogger, createP2pRemoteTrackResolverFactory, defaultCallSummaryConfig, defaultClientHealthThresholds, geohash, isClientIssueResolutionEntry, issuePayloadAsString, mannKendall, mannKendallVerdict, median, medianAbsoluteDeviation, pageHinkley, percentile, percentileOfSorted, robustZScore, setObserverLogger, summarize };