@observertc/observer-js 1.0.0-beta.2 → 1.0.0-beta.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (226) hide show
  1. package/README.md +1983 -384
  2. package/dist/index.d.mts +6664 -0
  3. package/dist/index.d.ts +6664 -0
  4. package/dist/index.js +7558 -0
  5. package/dist/index.js.map +1 -0
  6. package/dist/index.mjs +7435 -0
  7. package/dist/index.mjs.map +1 -0
  8. package/package.json +44 -18
  9. package/.eslintrc.json +0 -143
  10. package/.prettierignore +0 -5
  11. package/.prettierrc +0 -7
  12. package/jest.config.js +0 -9
  13. package/lib/ObservedCall.d.ts +0 -86
  14. package/lib/ObservedCall.d.ts.map +0 -1
  15. package/lib/ObservedCall.js +0 -225
  16. package/lib/ObservedCallEventMonitor.d.ts +0 -104
  17. package/lib/ObservedCallEventMonitor.d.ts.map +0 -1
  18. package/lib/ObservedCallEventMonitor.js +0 -314
  19. package/lib/ObservedCallSummary.d.ts +0 -18
  20. package/lib/ObservedCallSummary.d.ts.map +0 -1
  21. package/lib/ObservedCallSummary.js +0 -2
  22. package/lib/ObservedCertificate.d.ts +0 -19
  23. package/lib/ObservedCertificate.d.ts.map +0 -1
  24. package/lib/ObservedCertificate.js +0 -38
  25. package/lib/ObservedClient.d.ts +0 -130
  26. package/lib/ObservedClient.d.ts.map +0 -1
  27. package/lib/ObservedClient.js +0 -650
  28. package/lib/ObservedClientEventMonitor.d.ts +0 -91
  29. package/lib/ObservedClientEventMonitor.d.ts.map +0 -1
  30. package/lib/ObservedClientEventMonitor.js +0 -254
  31. package/lib/ObservedClientSummary.d.ts +0 -23
  32. package/lib/ObservedClientSummary.d.ts.map +0 -1
  33. package/lib/ObservedClientSummary.js +0 -2
  34. package/lib/ObservedCodec.d.ts +0 -22
  35. package/lib/ObservedCodec.d.ts.map +0 -1
  36. package/lib/ObservedCodec.js +0 -45
  37. package/lib/ObservedDataChannel.d.ts +0 -30
  38. package/lib/ObservedDataChannel.d.ts.map +0 -1
  39. package/lib/ObservedDataChannel.js +0 -75
  40. package/lib/ObservedIceCandidate.d.ts +0 -29
  41. package/lib/ObservedIceCandidate.d.ts.map +0 -1
  42. package/lib/ObservedIceCandidate.js +0 -59
  43. package/lib/ObservedIceCandidatePair.d.ts +0 -45
  44. package/lib/ObservedIceCandidatePair.d.ts.map +0 -1
  45. package/lib/ObservedIceCandidatePair.js +0 -127
  46. package/lib/ObservedIceTransport.d.ts +0 -36
  47. package/lib/ObservedIceTransport.d.ts.map +0 -1
  48. package/lib/ObservedIceTransport.js +0 -93
  49. package/lib/ObservedInboundRtp.d.ts +0 -92
  50. package/lib/ObservedInboundRtp.d.ts.map +0 -1
  51. package/lib/ObservedInboundRtp.js +0 -193
  52. package/lib/ObservedInboundTrack.d.ts +0 -32
  53. package/lib/ObservedInboundTrack.d.ts.map +0 -1
  54. package/lib/ObservedInboundTrack.js +0 -60
  55. package/lib/ObservedMediaPlayout.d.ts +0 -22
  56. package/lib/ObservedMediaPlayout.d.ts.map +0 -1
  57. package/lib/ObservedMediaPlayout.js +0 -42
  58. package/lib/ObservedMediaSource.d.ts +0 -28
  59. package/lib/ObservedMediaSource.d.ts.map +0 -1
  60. package/lib/ObservedMediaSource.js +0 -55
  61. package/lib/ObservedOutboundRtp.d.ts +0 -62
  62. package/lib/ObservedOutboundRtp.d.ts.map +0 -1
  63. package/lib/ObservedOutboundRtp.js +0 -144
  64. package/lib/ObservedOutboundTrack.d.ts +0 -32
  65. package/lib/ObservedOutboundTrack.d.ts.map +0 -1
  66. package/lib/ObservedOutboundTrack.js +0 -60
  67. package/lib/ObservedPeerConnection.d.ts +0 -207
  68. package/lib/ObservedPeerConnection.d.ts.map +0 -1
  69. package/lib/ObservedPeerConnection.js +0 -773
  70. package/lib/ObservedPeerConnectionTransport.d.ts +0 -17
  71. package/lib/ObservedPeerConnectionTransport.d.ts.map +0 -1
  72. package/lib/ObservedPeerConnectionTransport.js +0 -34
  73. package/lib/ObservedRemoteInboundRtp.d.ts +0 -30
  74. package/lib/ObservedRemoteInboundRtp.d.ts.map +0 -1
  75. package/lib/ObservedRemoteInboundRtp.js +0 -60
  76. package/lib/ObservedRemoteOutboundRtp.d.ts +0 -30
  77. package/lib/ObservedRemoteOutboundRtp.d.ts.map +0 -1
  78. package/lib/ObservedRemoteOutboundRtp.js +0 -60
  79. package/lib/ObservedTURN.d.ts +0 -31
  80. package/lib/ObservedTURN.d.ts.map +0 -1
  81. package/lib/ObservedTURN.js +0 -58
  82. package/lib/ObservedTurnServer.d.ts +0 -25
  83. package/lib/ObservedTurnServer.d.ts.map +0 -1
  84. package/lib/ObservedTurnServer.js +0 -59
  85. package/lib/Observer.d.ts +0 -65
  86. package/lib/Observer.d.ts.map +0 -1
  87. package/lib/Observer.js +0 -166
  88. package/lib/ObserverEventMonitor.d.ts +0 -138
  89. package/lib/ObserverEventMonitor.d.ts.map +0 -1
  90. package/lib/ObserverEventMonitor.js +0 -433
  91. package/lib/ObserverSummary.d.ts +0 -13
  92. package/lib/ObserverSummary.d.ts.map +0 -1
  93. package/lib/ObserverSummary.js +0 -2
  94. package/lib/common/Middleware.d.ts +0 -17
  95. package/lib/common/Middleware.d.ts.map +0 -1
  96. package/lib/common/Middleware.js +0 -60
  97. package/lib/common/SingleExecutor.d.ts +0 -3
  98. package/lib/common/SingleExecutor.d.ts.map +0 -1
  99. package/lib/common/SingleExecutor.js +0 -31
  100. package/lib/common/logger.d.ts +0 -17
  101. package/lib/common/logger.d.ts.map +0 -1
  102. package/lib/common/logger.js +0 -50
  103. package/lib/common/types.d.ts +0 -3
  104. package/lib/common/types.d.ts.map +0 -1
  105. package/lib/common/types.js +0 -3
  106. package/lib/common/utils.d.ts +0 -11
  107. package/lib/common/utils.d.ts.map +0 -1
  108. package/lib/common/utils.js +0 -63
  109. package/lib/detectors/Detector.d.ts +0 -5
  110. package/lib/detectors/Detector.d.ts.map +0 -1
  111. package/lib/detectors/Detector.js +0 -3
  112. package/lib/detectors/Detectors.d.ts +0 -11
  113. package/lib/detectors/Detectors.d.ts.map +0 -1
  114. package/lib/detectors/Detectors.js +0 -34
  115. package/lib/index.d.ts +0 -28
  116. package/lib/index.d.ts.map +0 -1
  117. package/lib/index.js +0 -49
  118. package/lib/mediasoup/ObservedMediaRouter.d.ts +0 -10
  119. package/lib/mediasoup/ObservedMediaRouter.d.ts.map +0 -1
  120. package/lib/mediasoup/ObservedMediaRouter.js +0 -6
  121. package/lib/monitors/TurnUsageMonitor.d.ts +0 -1
  122. package/lib/monitors/TurnUsageMonitor.d.ts.map +0 -1
  123. package/lib/monitors/TurnUsageMonitor.js +0 -146
  124. package/lib/schema/ClientEventTypes.d.ts +0 -228
  125. package/lib/schema/ClientEventTypes.d.ts.map +0 -1
  126. package/lib/schema/ClientEventTypes.js +0 -41
  127. package/lib/schema/ClientMetaTypes.d.ts +0 -34
  128. package/lib/schema/ClientMetaTypes.d.ts.map +0 -1
  129. package/lib/schema/ClientMetaTypes.js +0 -16
  130. package/lib/schema/ClientSample.d.ts +0 -1333
  131. package/lib/schema/ClientSample.d.ts.map +0 -1
  132. package/lib/schema/ClientSample.js +0 -4
  133. package/lib/scores/CalculatedScore.d.ts +0 -6
  134. package/lib/scores/CalculatedScore.d.ts.map +0 -1
  135. package/lib/scores/CalculatedScore.js +0 -2
  136. package/lib/scores/DefaultCallScoreCalculator.d.ts +0 -7
  137. package/lib/scores/DefaultCallScoreCalculator.d.ts.map +0 -1
  138. package/lib/scores/DefaultCallScoreCalculator.js +0 -21
  139. package/lib/scores/ScoreCalculator.d.ts +0 -4
  140. package/lib/scores/ScoreCalculator.d.ts.map +0 -1
  141. package/lib/scores/ScoreCalculator.js +0 -2
  142. package/lib/updaters/CallUpdater.d.ts +0 -5
  143. package/lib/updaters/CallUpdater.d.ts.map +0 -1
  144. package/lib/updaters/CallUpdater.js +0 -2
  145. package/lib/updaters/ObserverUpdater.d.ts +0 -5
  146. package/lib/updaters/ObserverUpdater.d.ts.map +0 -1
  147. package/lib/updaters/ObserverUpdater.js +0 -2
  148. package/lib/updaters/OnAllCallObserverUpdater.d.ts +0 -15
  149. package/lib/updaters/OnAllCallObserverUpdater.d.ts.map +0 -1
  150. package/lib/updaters/OnAllCallObserverUpdater.js +0 -50
  151. package/lib/updaters/OnAllClientCallUpdater.d.ts +0 -15
  152. package/lib/updaters/OnAllClientCallUpdater.d.ts.map +0 -1
  153. package/lib/updaters/OnAllClientCallUpdater.js +0 -53
  154. package/lib/updaters/OnAnyCallObserverUpdater.d.ts +0 -12
  155. package/lib/updaters/OnAnyCallObserverUpdater.d.ts.map +0 -1
  156. package/lib/updaters/OnAnyCallObserverUpdater.js +0 -37
  157. package/lib/updaters/OnAnyClientCallUpdater.d.ts +0 -12
  158. package/lib/updaters/OnAnyClientCallUpdater.d.ts.map +0 -1
  159. package/lib/updaters/OnAnyClientCallUpdater.js +0 -36
  160. package/lib/updaters/OnIntervalUpdater.d.ts +0 -10
  161. package/lib/updaters/OnIntervalUpdater.d.ts.map +0 -1
  162. package/lib/updaters/OnIntervalUpdater.js +0 -17
  163. package/lib/updaters/Updater.d.ts +0 -6
  164. package/lib/updaters/Updater.d.ts.map +0 -1
  165. package/lib/updaters/Updater.js +0 -2
  166. package/lib/utils/MediasoupRemoteTrackResolver.d.ts +0 -25
  167. package/lib/utils/MediasoupRemoteTrackResolver.d.ts.map +0 -1
  168. package/lib/utils/MediasoupRemoteTrackResolver.js +0 -110
  169. package/lib/utils/RemoteTrackResolver.d.ts +0 -7
  170. package/lib/utils/RemoteTrackResolver.d.ts.map +0 -1
  171. package/lib/utils/RemoteTrackResolver.js +0 -2
  172. package/src/ObservedCall.ts +0 -342
  173. package/src/ObservedCallEventMonitor.ts +0 -392
  174. package/src/ObservedCallSummary.ts +0 -22
  175. package/src/ObservedCertificate.ts +0 -43
  176. package/src/ObservedClient.ts +0 -807
  177. package/src/ObservedClientEventMonitor.ts +0 -309
  178. package/src/ObservedClientSummary.ts +0 -24
  179. package/src/ObservedCodec.ts +0 -49
  180. package/src/ObservedDataChannel.ts +0 -85
  181. package/src/ObservedIceCandidate.ts +0 -64
  182. package/src/ObservedIceCandidatePair.ts +0 -134
  183. package/src/ObservedIceTransport.ts +0 -99
  184. package/src/ObservedInboundRtp.ts +0 -205
  185. package/src/ObservedInboundTrack.ts +0 -75
  186. package/src/ObservedMediaPlayout.ts +0 -49
  187. package/src/ObservedMediaSource.ts +0 -60
  188. package/src/ObservedOutboundRtp.ts +0 -156
  189. package/src/ObservedOutboundTrack.ts +0 -74
  190. package/src/ObservedPeerConnection.ts +0 -1088
  191. package/src/ObservedPeerConnectionTransport.ts +0 -38
  192. package/src/ObservedRemoteInboundRtp.ts +0 -64
  193. package/src/ObservedRemoteOutboundRtp.ts +0 -65
  194. package/src/ObservedTURN.ts +0 -86
  195. package/src/ObservedTurnServer.ts +0 -72
  196. package/src/Observer.ts +0 -251
  197. package/src/ObserverEventMonitor.ts +0 -551
  198. package/src/ObserverSummary.ts +0 -15
  199. package/src/common/Middleware.ts +0 -84
  200. package/src/common/SingleExecutor.ts +0 -36
  201. package/src/common/logger.ts +0 -82
  202. package/src/common/types.ts +0 -4
  203. package/src/common/utils.ts +0 -69
  204. package/src/detectors/Detector.ts +0 -6
  205. package/src/detectors/Detectors.ts +0 -38
  206. package/src/index.ts +0 -28
  207. package/src/mediasoup/ObservedMediaRouter.ts +0 -10
  208. package/src/monitors/TurnUsageMonitor.ts +0 -187
  209. package/src/schema/ClientEventTypes.ts +0 -280
  210. package/src/schema/ClientMetaTypes.ts +0 -41
  211. package/src/schema/ClientSample.ts +0 -1682
  212. package/src/scores/CalculatedScore.ts +0 -6
  213. package/src/scores/DefaultCallScoreCalculator.ts +0 -23
  214. package/src/scores/ScoreCalculator.ts +0 -3
  215. package/src/updaters/CallUpdater.ts +0 -5
  216. package/src/updaters/ObserverUpdater.ts +0 -5
  217. package/src/updaters/OnAllCallObserverUpdater.ts +0 -59
  218. package/src/updaters/OnAllClientCallUpdater.ts +0 -61
  219. package/src/updaters/OnAnyCallObserverUpdater.ts +0 -41
  220. package/src/updaters/OnAnyClientCallUpdater.ts +0 -39
  221. package/src/updaters/OnIntervalUpdater.ts +0 -19
  222. package/src/updaters/Updater.ts +0 -5
  223. package/src/utils/MediasoupRemoteTrackResolver.ts +0 -155
  224. package/src/utils/RemoteTrackResolver.ts +0 -12
  225. package/tsconfig.json +0 -19
  226. package/tslint.json +0 -6
package/README.md CHANGED
@@ -1,23 +1,74 @@
1
- # ObserverTC - Observer JS
1
+ # ObserverTC — `@observertc/observer-js`
2
2
 
3
3
  [![NPM version](https://img.shields.io/npm/v/@observertc/observer-js.svg)](https://www.npmjs.com/package/@observertc/observer-js)
4
4
  [![License](https://img.shields.io/npm/l/@observertc/observer-js.svg)](https://github.com/observertc/observer-js/blob/main/LICENSE)
5
5
 
6
- `observer-js` is a Node.js library for monitoring WebRTC client data. It processes statistical samples from clients, organizes them into calls and participants, tracks a wide range of metrics, detects common issues, and calculates quality scores. This enables real-time insights into WebRTC session performance.
6
+ > **In one line:** feed it WebRTC `getStats()` snapshots, and get back a live, queryable model of
7
+ > every call plus a single typed event stream to react to.
8
+
9
+ `observer-js` is a **server-side Node.js library for monitoring WebRTC sessions**. A WebRTC
10
+ application (typically an SFU or a signaling/stats backend) feeds it `ClientSample` objects —
11
+ periodic snapshots of each participant's `RTCPeerConnection.getStats()` output plus
12
+ application events — and `observer-js` maintains a live, in-memory model of every call,
13
+ participant, peer connection, and media stream, derives per-interval and cumulative metrics,
14
+ and emits a single, unified stream of typed events the application can react to.
15
+
16
+ **What you can do with it:**
17
+
18
+ - **Monitor calls live** — a queryable in-memory tree of every call, client, peer connection,
19
+ track, codec, ICE candidate and data channel, each holding current **and** cumulative metrics.
20
+ - **React on one event bus** — subscribe once on the `Observer`; every payload carries its full
21
+ ancestry (`call → client → peer connection → stat`), so you never walk the tree to subscribe.
22
+ - **Get derived metrics for free** — counter-reset-safe per-tick deltas, bitrates, jitter, RTT,
23
+ fraction-lost, remote-RTP (RTCP) correlation, and TURN/TCP usage from the selected candidate pair.
24
+ - **Correlate across an SFU** — link a publisher's outbound track to every subscriber's inbound
25
+ track (`RemoteTrackResolver`), and observe mediasoup routers/transports/producers/consumers
26
+ on the server side.
27
+ - **Detect server-only problems** — cross-client `Detector`s raise `call-issue`s for conditions no
28
+ single client can see (e.g. everyone in a call degrading at once).
29
+ - **Persist every sample** — per-client sinks (JSONL file, in-memory, or your own) for archival,
30
+ streaming, and offline replay.
31
+ - **Drop it in safely** — warn-don't-throw, a pluggable logger, dual **ESM + CommonJS**, and **no**
32
+ media-stack dependency in the core.
33
+
34
+ > **Status:** `1.0.0-beta`. The API described here is current and intended to be implemented
35
+ > against directly. This document is written to be self-sufficient: an engineer (or an AI
36
+ > agent) should be able to integrate the library, or develop it further, from this file alone.
37
+ > A companion doc, [`docs/logging.md`](./docs/logging.md), covers logging integration in depth.
38
+
39
+ > **Packaging:** server-side, **Node.js ≥ 22**, shipped as a **dual ESM + CommonJS** build — so it
40
+ > works whether your project uses `import` (ESM) or `require()` (CommonJS). Everything — including
41
+ > the built-in file sink — is exported from the single `@observertc/observer-js` entry.
42
+
43
+ > **For AI agents:** [`llms.txt`](./llms.txt) is a curated map of these docs (it belongs at the root
44
+ > of the docs site); [`AGENTS.md`](./AGENTS.md) covers build/test commands and the conventions for
45
+ > working **in** this repository.
7
46
 
8
- This library is a core component of the ObserverTC ecosystem, designed to provide robust server-side monitoring capabilities for WebRTC applications.
47
+ ---
9
48
 
10
- ## Features
49
+ ## Table of contents
50
+
51
+ 1. [Installation](#installation)
52
+ 2. [Quick start](#quick-start)
53
+ 3. [Data flow](#data-flow)
54
+ 4. [Entity hierarchy](#entity-hierarchy)
55
+ 5. [Ingestion: `accept()`, context & lifecycle](#ingestion-accept-context--lifecycle)
56
+ 6. [When things update](#when-things-update)
57
+ 7. [The event bus](#the-event-bus) ← the core of the API
58
+ 8. [API reference](#api-reference)
59
+ 9. [Schema types (`ClientSample`)](#schema-types-clientsample)
60
+ 10. [Detectors (server-side extension point)](#detectors-server-side-extension-point)
61
+ 11. [Call summaries](#call-summaries)
62
+ 12. [Remote track resolution (mediasoup / SFU)](#remote-track-resolution-mediasoup--sfu)
63
+ 13. [Mediasoup router observation](#mediasoup-router-observation)
64
+ 14. [Sinks (per-client sample persistence)](#sinks-per-client-sample-persistence)
65
+ 15. [Injecting data into a client](#injecting-data-into-a-client)
66
+ 16. [Logging](#logging)
67
+ 17. [Design notes](#design-notes)
68
+ 18. [Error-handling philosophy](#error-handling-philosophy)
69
+ 19. [Development & extension guide](#development--extension-guide)
11
70
 
12
- - **Hierarchical Data Model**: Organizes data into `Observer` -> `ObservedCall` -> `ObservedClient` -> `ObservedPeerConnection` and further into streams and data channels.
13
- - **Comprehensive Metrics**: Tracks a wide array of WebRTC statistics including RTT, jitter, packet loss, codecs, ICE states, TURN usage, bandwidth, and more.
14
- - **Automatic Entity Management**: Can automatically create and manage call and client entities based on incoming data samples.
15
- - **Issue Detection**: Built-in detectors for common WebRTC problems.
16
- - **Quality Scoring**: Calculates quality scores for calls and clients.
17
- - **Event-Driven**: Emits events for significant state changes, new entities, and detected issues.
18
- - **Configurable Update Policies**: Flexible control over how and when metrics are processed and updated.
19
- - **TypeScript Support**: Written in TypeScript, providing strong typing and intellisense.
20
- - **Extensible**: Supports custom application data (`appData`) and integration with external schema definitions (e.g., `observertc/schemas`).
71
+ ---
21
72
 
22
73
  ## Installation
23
74
 
@@ -27,510 +78,2058 @@ npm install @observertc/observer-js
27
78
  yarn add @observertc/observer-js
28
79
  ```
29
80
 
30
- ## Quick Start
81
+ **Server-side, Node.js ≥ 22, dual ESM + CommonJS.** The package ships both module formats, so it
82
+ works the same whether your project is ESM or CommonJS — your import line is unchanged either way:
31
83
 
32
- ```typescript
33
- import { Observer, ObserverConfig } from '@observertc/observer-js';
34
- import { ClientSample } from '@observertc/schemas'; // Assuming you use the official schemas
84
+ ```ts
85
+ import { Observer, ClientSample, createJsonlFileSinkFactory } from '@observertc/observer-js';
86
+ ```
35
87
 
36
- // 1. Configure the Observer
37
- const observerConfig: ObserverConfig = {
38
- updatePolicy: 'update-on-interval',
39
- updateIntervalInMs: 5000, // Update observer every 5 seconds
40
- defaultCallUpdatePolicy: 'update-on-any-client-updated', // Calls update when any client sends data
41
- };
42
- const observer = new Observer(observerConfig);
43
-
44
- // 2. Listen to events
45
- observer.on('newcall', (call) => {
46
- console.log(`[Observer] New call detected: ${call.callId}`);
47
-
48
- call.on('newclient', (client) => {
49
- console.log(`[Call: ${call.callId}] New client joined: ${client.clientId}`);
50
-
51
- client.on('issue', (issue) => {
52
- console.warn(`[Client: ${client.clientId}] Issue: ${issue.type} - ${issue.severity} - ${issue.description}`);
53
- });
54
- });
55
-
56
- call.on('update', () => {
57
- console.log(
58
- `[Call: ${call.callId}] Metrics updated. Score: ${call.score?.toFixed(1)}, Clients: ${call.numberOfClients}`
59
- );
60
- });
61
- });
62
-
63
- // 3. Process Client Samples
64
- // (ClientSample typically comes from your application after processing getStats() output)
65
- function processClientStats(rawStats: any, callId: string, clientId: string) {
66
- // Transform rawStats into the ClientSample format
67
- // This is a placeholder for your actual transformation logic
68
- const sample: ClientSample = {
69
- callId,
70
- clientId,
71
- timestamp: Date.now(),
72
- // ...populate with transformed stats from rawStats, adhering to the ClientSample schema
73
- // from github.com/observertc/schemas
74
- };
75
- observer.accept(sample);
88
+ In an ESM project this resolves to the `.mjs` build; in a CommonJS project (where TypeScript
89
+ compiles your `import` down to `require()`) it resolves to the `.js` build. Everything is exported
90
+ from the single `@observertc/observer-js` entry. Written in TypeScript; ships type declarations for
91
+ both formats (`dist/index.d.ts` for `require`, `dist/index.d.mts` for `import`). Runtime
92
+ dependencies: `@bufbuild/protobuf`, `events`, `uuid`. The library does **not** bundle a logger or
93
+ any transport — see [Logging](#logging).
94
+
95
+ `ClientSample` and friends are re-exported from this package, and are also published as the
96
+ shared schema in [`@observertc/schemas`](https://github.com/observertc/schemas); samples
97
+ produced on the client (e.g. by `@observertc/client-monitor-js`) conform to the same shape.
98
+
99
+ ---
100
+
101
+ ## Quick start
102
+
103
+ ```ts
104
+ import { Observer, ClientSample } from '@observertc/observer-js';
105
+
106
+ // 1. Create an observer.
107
+ const observer = new Observer({
108
+ // a call updates when any of its clients does, and the observer when any of its calls does —
109
+ // both default to true, so this line is only here to show the knob exists:
110
+ autoUpdateOnCallUpdate: true,
111
+ // optional auto-teardown:
112
+ closeCallIfEmptyForMs: 20_000,
113
+ closeClientIfIdleForMs: 60_000,
114
+ });
115
+
116
+ // 2. Subscribe on the single bus. Every payload is an object with the ancestry.
117
+ observer.on('call-added', ({ observedCall }) => {
118
+ console.log('new call', observedCall.callId);
119
+ });
120
+
121
+ observer.on('client-issue', ({ observedClient, issue }) => {
122
+ console.warn(`[${observedClient.clientId}] ${issue.type}`, issue.payload);
123
+ });
124
+
125
+ observer.on('peer-connection-updated', ({ observedClient, observedPeerConnection }) => {
126
+ console.log(observedClient.clientId, 'RTT(ms):', observedPeerConnection.currentRttInMs);
127
+ });
128
+
129
+ observer.on('sample-rejected', ({ reason, sample }) => {
130
+ console.warn('dropped a sample:', reason);
131
+ });
132
+
133
+ // 3. Feed samples. `context` (optional) is transient per-accept data, carried to the
134
+ // `*-updated` events this accept triggers (never written to appData).
135
+ function onClientStats(sample: ClientSample) {
136
+ observer.accept(sample, { studioVersion: '1.2.3' });
76
137
  }
77
138
 
78
- // Example usage:
79
- // const myRawClientStats = getStatsFromClient();
80
- // processClientStats(myRawClientStats, 'myMeeting123', 'userABC');
139
+ // 4. Tear down.
140
+ process.on('SIGINT', () => observer.close());
141
+ ```
142
+
143
+ ---
144
+
145
+ ## Data flow
81
146
 
82
- // 4. Cleanup when done
83
- // process.on('SIGINT', () => observer.close());
147
+ ```
148
+ client getStats() ──► ClientSample ──► observer.accept(sample, ctx?)
149
+ │
150
+ ┌────────────────────────────────┘
151
+ ▼
152
+ get-or-create ObservedCall ──► get-or-create ObservedClient ──► client.accept(sample, ctx)
153
+ │
154
+ per peerConnections[] in the sample
155
+ ▼
156
+ get-or-create ObservedPeerConnection
157
+ .accept(pcSample, ctx) updates all sub-stats,
158
+ derives deltas/bitrates/RTT, correlates remote RTP
159
+ │
160
+ metrics roll up: PeerConnection → Client → Call → Observer
161
+ │
162
+ events emitted on the Observer bus ──► your handlers
84
163
  ```
85
164
 
165
+ - A sample **must** have `callId` and `clientId` (the library sets them, or the app does). If
166
+ either is missing, the sample is dropped and `sample-rejected` is emitted.
167
+ - Sub-entities that stop appearing in samples are garbage-collected via a "visited"
168
+ mark-and-sweep on each `ObservedPeerConnection.accept()`, emitting the corresponding
169
+ `*-removed` events.
170
+
86
171
  ---
87
172
 
88
- ## Detailed Documentation
173
+ ## Entity hierarchy
89
174
 
90
- The following sections provide a comprehensive guide to `observer-js`.
175
+ | Class | Created by | Keyed on its parent as | Holds |
176
+ |-------|-----------|------------------------|-------|
177
+ | `Observer` | `new Observer(config?)` | — (root) | `observedCalls: Map<string, ObservedCall>`, global counters, the event bus |
178
+ | `ObservedCall` | `observer.createObservedCall(settings)` / lazily by `accept` | `observedCalls` | `observedClients: Map<string, ObservedClient>`, call-wide metrics, `detectors`, `scoreCalculator` |
179
+ | `ObservedClient` | `call.createObservedClient(settings)` / lazily | `observedClients` | `observedPeerConnections: Map<string, ObservedPeerConnection>`, per-client metrics |
180
+ | `ObservedPeerConnection` | lazily, from `sample.peerConnections[]` | `observedPeerConnections` | the 15 sub-stat maps below, transport/RTT/bitrate metrics |
181
+ | Sub-stats | lazily, from the `PeerConnectionSample` | maps on the PC | individual WebRTC stat objects |
91
182
 
92
- ### 1. General Description
183
+ `ObservedPeerConnection` sub-stat maps (all `public readonly`):
93
184
 
94
- (This section is identical to the introductory paragraph at the top of this README)
185
+ ```
186
+ observedCertificates, observedCodecs, observedDataChannels,
187
+ observedIceCandidates, observedIceCandidatesPair, observedIceTransports,
188
+ observedInboundRtps, observedInboundTracks, observedMediaPlayouts,
189
+ observedMediaSources, observedOutboundRtps, observedOutboundTracks,
190
+ observedPeerConnectionTransports, observedRemoteInboundRtps, observedRemoteOutboundRtps
191
+ ```
95
192
 
96
- ### 2. Core Concepts
193
+ Each sub-stat class (`ObservedInboundRtp`, `ObservedOutboundRtp`, `ObservedInboundTrack`,
194
+ `ObservedOutboundTrack`, `ObservedDataChannel`, `ObservedIceCandidate`,
195
+ `ObservedIceCandidatePair`, `ObservedIceTransport`, `ObservedCertificate`, `ObservedCodec`,
196
+ `ObservedMediaSource`, `ObservedMediaPlayout`, `ObservedPeerConnectionTransport`,
197
+ `ObservedRemoteInboundRtp`, `ObservedRemoteOutboundRtp`) mirrors the corresponding stat
198
+ fields from the schema plus derived fields (deltas, bitrates).
97
199
 
98
- #### 2.1. Data Flow
200
+ ---
99
201
 
100
- 1. **Client-Side**: Your application collects WebRTC statistics (e.g., via `RTCPeerConnection.getStats()`).
101
- 2. **Transformation**: These raw stats are transformed into the `ClientSample` schema (ideally from [observertc/schemas](https://github.com/observertc/schemas)).
102
- 3. **Ingestion**: The `ClientSample` is passed to the `observer.accept()` method.
103
- 4. **Processing**: `observer-js` processes the sample, updating or creating relevant entities (`ObservedCall`, `ObservedClient`, `ObservedPeerConnection`, etc.) and their metrics.
104
- 5. **Analysis**: Metrics are analyzed for issue detection and quality scoring.
105
- 6. **Events**: Events are emitted for significant state changes, new issues, or updates.
202
+ ## Ingestion: `accept()`, context & lifecycle
106
203
 
107
- #### 2.2. Entity Hierarchy
204
+ ### `observer.accept(sample, context?)`
108
205
 
109
- - **`Observer`**: The root object, managing multiple calls and global settings.
110
- - **`ObservedCall`**: Represents a distinct call session.
111
- - **`ObservedClient`**: Represents an individual participant within a call.
112
- - **`ObservedPeerConnection`**: Represents a WebRTC RTCPeerConnection of a client.
113
- - **`ObservedInboundRtpStream` / `ObservedOutboundRtpStream`**: Tracks individual media streams.
114
- - **`ObservedDataChannel`**: Tracks data channels.
115
- - **`ObservedTURN`**: Tracks global TURN server usage metrics across the observer.
206
+ The single entry point. It:
116
207
 
117
- #### 2.3. Automatic Entity Creation
208
+ 1. drops + emits `sample-rejected` if the observer is closed;
209
+ 2. runs the sample through the **global accept-middleware chain** (see below);
210
+ 3. (chain terminal) drops + emits `sample-rejected` if `callId`/`clientId` is missing;
211
+ 4. gets or lazily creates the `ObservedCall` and `ObservedClient` (their `appData` comes from the
212
+ configured factories, never from `context`);
213
+ 5. delegates to `client.accept(sample, context)`, which fans out to each
214
+ `ObservedPeerConnection.accept(pcSample, context)`.
118
215
 
119
- When `observer.accept(sample)` is called:
216
+ ### Accept middlewares (global pre-dispatch hook)
120
217
 
121
- - If an `ObservedCall` for `sample.callId` doesn't exist, it's typically created.
122
- - If an `ObservedClient` for `sample.clientId` within that call doesn't exist, it's typically created.
123
- - Peer connections, streams, and data channels are similarly managed based on IDs in the sample.
218
+ `observer.addAcceptMiddleware(...)` registers middlewares run on **every** sample inside
219
+ `accept()`, in order, **before** the sample is dispatched to any call or client. Each middleware
220
+ gets a `{ sample, context }` payload; it can inspect or mutate the sample (set/normalize
221
+ `callId`/`clientId`, enrich, redact) or the context, then call `next(payload)` to continue.
222
+ **Not calling `next` drops the sample** — nothing is created and no event fires. A throwing
223
+ middleware is caught and warns (the sample is dropped), never crashing `accept()`.
124
224
 
125
- #### 2.4. Metrics Aggregation
225
+ ```ts
226
+ import { Observer, AcceptMiddleware } from '@observertc/observer-js';
126
227
 
127
- The library aggregates a wide array of metrics at each level of the hierarchy, including (but not limited to):
228
+ const observer = new Observer();
128
229
 
129
- - RTT, jitter, packet loss
130
- - Bytes sent/received (audio, video, data)
131
- - Codec information
132
- - ICE connection details, TURN usage
133
- - Stream/track states (muted, enabled)
134
- - Frame rates, resolutions
135
- - Bandwidth estimations
230
+ // derive callId/clientId from the app's own attachment, before dispatch
231
+ const route: AcceptMiddleware = ({ sample }, next) => {
232
+ sample.callId ??= sample.attachments?.roomId as string;
233
+ sample.clientId ??= sample.attachments?.peerId as string;
234
+ next({ sample });
235
+ };
136
236
 
137
- #### 2.5. Issue Detection
237
+ // drop samples from a blocklisted client (never dispatched)
238
+ const filter: AcceptMiddleware = (payload, next) => {
239
+ if (blocked.has(payload.sample.clientId)) return; // no next() => dropped
240
+ next(payload);
241
+ };
138
242
 
139
- A `Detectors` system analyzes metrics to identify common WebRTC issues (e.g., high packet loss, low audio levels, frozen video, connection setup problems). Issues are reported via events.
243
+ observer.addAcceptMiddleware(route, filter);
244
+ // observer.removeAcceptMiddleware(route);
245
+ ```
140
246
 
141
- #### 2.6. Quality Scoring
247
+ This is a lightweight global injection point. When no middleware is registered, `accept()`
248
+ dispatches directly with no overhead.
142
249
 
143
- `ScoreCalculator` components assess the quality of calls and clients based on metrics and detected issues, typically resulting in a numerical score (e.g., 0.0 to 5.0).
250
+ ### `context` (the `AcceptContext`)
144
251
 
145
- #### 2.7. Event-Driven Architecture
252
+ ```ts
253
+ type AcceptContext = Record<string, unknown>;
254
+ ```
146
255
 
147
- The library uses Node.js `EventEmitter` to signal various occurrences, allowing applications to react to changes in real-time.
256
+ A single, optional, free-form object threaded down the whole accept chain
257
+ (`Observer → Client → PeerConnection`). It is **transient request-scoped data** — temporary or
258
+ contextual information the application wants available while an update is processed.
148
259
 
149
- ### 3. API Reference
260
+ `context` is **never written to `appData`** and is **not stored** on any entity. The two are
261
+ deliberately distinct:
150
262
 
151
- #### 3.1. `Observer`
263
+ - **`appData`** — application-assigned extra info that identifies/decorates an entity, fixed at
264
+ creation (via `settings.appData` or the `createCallAppData` / `createClientAppData` factories),
265
+ or assigned by the app on the `*-added` events. The library never changes it. The factories
266
+ **receive the context of the `accept()` that triggered the creation**, so a fact carried on the
267
+ context can be baked into `appData` at birth — but it is copied by the factory, deliberately, not
268
+ written across by the library.
269
+ - **`context`** — passed per `accept()`, may differ on every call, and is carried straight
270
+ through to the `*-updated` events that the `accept()` triggers, then discarded.
152
271
 
153
- Manages all monitored calls and global observer state.
272
+ `client-updated` and `peer-connection-updated` carry the exact context of that sample;
273
+ `call-updated` carries the context of the client `accept()` that drove the call update (absent
274
+ for interval- or teardown-driven call updates). When no context is given, the field is absent.
154
275
 
155
- **Configuration (`ObserverConfig`)**
276
+ ### Get-or-create helpers
156
277
 
157
- ```typescript
158
- export type ObserverConfig<AppData extends Record<string, unknown> = Record<string, unknown>> = {
159
- updatePolicy?: 'update-on-any-call-updated' | 'update-when-all-call-updated' | 'update-on-interval';
160
- updateIntervalInMs?: number; // Used if updatePolicy is 'update-on-interval'
161
- defaultCallUpdatePolicy?: ObservedCallSettings['updatePolicy'];
162
- defaultCallUpdateIntervalInMs?: number;
163
- appData?: AppData; // Custom data for this observer instance
164
- };
278
+ If you want to create/configure entities yourself before/without samples:
279
+
280
+ ```ts
281
+ const call = observer.getOrCreateObservedCall({ callId, appData }); // ObservedCall | undefined
282
+ const client = call?.getOrCreateObservedClient({ clientId, appData }); // ObservedClient | undefined
165
283
  ```
166
284
 
167
- **Constructor**
285
+ These return `undefined` (and warn) when the parent is closed; `createObservedCall`/
286
+ `createObservedClient` return the **existing** instance (and warn) if the id already exists.
168
287
 
169
- ```typescript
170
- new Observer<AppData>(config?: ObserverConfig<AppData>)
288
+ ### Automatic teardown
289
+
290
+ - `closeClientIfIdleForMs` — a client with no sample for this long auto-closes.
291
+ - `closeCallIfEmptyForMs` — a call with zero clients for this long auto-closes.
292
+ - Closing cascades down (call → clients → peer connections → sub-stats), unsubscribing
293
+ listeners and emitting the `*-closed` / `*-removed` events.
294
+
295
+ ---
296
+
297
+ ## When things update
298
+
299
+ "Update" means *recompute aggregated metrics, run the detectors, and emit the `*-updated` event* at
300
+ that level. Updates are **event-driven** — there is no built-in timer.
301
+
302
+ The rule is structural rather than configurable:
303
+
304
+ > **A call is updated when any of its clients is updated. The observer is updated when any of its
305
+ > calls is updated.** Composed, that means the observer is updated exactly when any client anywhere
306
+ > is updated.
307
+
308
+ Two booleans, both defaulting to `true`, let you opt out of a link in that chain:
309
+
310
+ | Setting | Where | Effect when `false` |
311
+ |---------|-------|---------------------|
312
+ | `autoUpdateOnClientUpdate` | `ObservedCallSettings` | the call updates only when you call `call.update()` |
313
+ | `autoUpdateOnCallUpdate` | `ObserverConfig` | the observer updates only when you call `observer.update()` |
314
+
315
+ An app that wants a fixed cadence sets both to `false` and drives `observer.update()` from its own
316
+ `setInterval`. Note that **observer-scoped detectors and validators run nowhere else** — if the
317
+ observer never updates, they never run.
318
+
319
+ ```ts
320
+ const observer = new Observer({ autoUpdateOnCallUpdate: false });
321
+
322
+ setInterval(() => observer.update(), 5_000);
323
+ ```
324
+
325
+ > Earlier versions had an `updatePolicy` / `defaultCallUpdatePolicy` enum (`'update-on-any-…'`,
326
+ > `'update-when-all-…'`, `'none'`) and a pluggable `Updater` object. Both are gone. "When all clients
327
+ > have updated" sounds appealing and deadlocks on the first client that stops sending — one silent
328
+ > participant froze the whole call's aggregation until it timed out.
329
+
330
+ ---
331
+
332
+ ## The event bus
333
+
334
+ This is the primary API. **Subscribe on the `Observer` instance** — it is the single emitter
335
+ for the entire hierarchy. The `ObservedCall` / `ObservedClient` / `ObservedPeerConnection`
336
+ objects are themselves `EventEmitter`s too, but those local events are reserved for internal
337
+ lifecycle/teardown wiring (see [Local lifecycle events](#local-lifecycle-events)); application
338
+ code should use the Observer bus.
339
+
340
+ ### Payload shape: ancestry + subject
341
+
342
+ Every Observer event delivers exactly **one argument: a payload object**. The payload always
343
+ contains the ancestry from the observer down to the entity that raised it, plus any event-
344
+ specific subject:
345
+
346
+ ```ts
347
+ type ObserverEventBase = { observer: Observer, context?: AcceptContext };
348
+ type ObservedCallScope = ObserverEventBase & { observedCall: ObservedCall };
349
+ type ObservedClientScope = ObservedCallScope & { observedClient: ObservedClient };
350
+ type ObservedPeerConnectionScope = ObservedClientScope & { observedPeerConnection: ObservedPeerConnection };
171
351
  ```
172
352
 
173
- - `config`: Optional. Defaults: `updatePolicy: 'update-when-all-call-updated'`.
353
+ So a peer-connection-level event hands you the observer, call, client, **and** peer connection:
354
+
355
+ ```ts
356
+ observer.on('inbound-rtp-added', ({ observer, observedCall, observedClient, observedPeerConnection, observedInboundRtp }) => {
357
+ // all five are present and correctly typed
358
+ });
359
+ ```
360
+
361
+ `observer.on/off/once/emit` are fully typed against the event map — the handler argument is
362
+ inferred per event name.
363
+
364
+ ### Event catalogue
365
+
366
+ All payloads include the ancestry for their level (above). The **Extra** column lists the
367
+ additional field(s) on top of that scope.
368
+
369
+ #### Observer level — scope `{ observer }`
370
+
371
+ | Event | Extra payload | Fires when |
372
+ |-------|---------------|-----------|
373
+ | `observer-updated` | — | `observer.update()` ran (see [When things update](#when-things-update)) |
374
+ | `observer-closed` | — | `observer.close()` |
375
+ | `sample-rejected` | `{ reason: 'observer-closed' \| 'missing-callId' \| 'missing-clientId', sample: ClientSample }` | a sample was dropped by `accept()` |
376
+ | `observer-issue` | `{ issue: ObserverIssue }` | `observer.addIssue(...)` — a cross-call / SFU-wide finding (see [observer-level detectors](#observer-level-detectors-cross-call--sfu-wide)) |
377
+ | `validation-ready` | `{ validator: string, report: ValidationReport }` | a [validator](#validators--one-shot-structural-checks) settled — fires once per check, not per tick |
378
+
379
+ #### Mediasoup level — scope `{ observer, observedMediasoupRouter }`
380
+
381
+ | Event | Extra | Fires when |
382
+ |-------|-------|-----------|
383
+ | `mediasoup-router-added` | — | `observer.createObservedMediasoupRouter(...)` registered a router |
384
+ | `mediasoup-router-matched-with-peer-connection` | `{ observedCall, observedClient, observedPeerConnection }` | a newly added peer connection's id matched one of the router's WebRTC transport ids. **Opt-in** via `matchPeerConnectionByWebRtcTransportId: true`. |
385
+ | `mediasoup-router-removed` | — | the underlying mediasoup router closed (its `router.observer` `close` fired) |
386
+
387
+ See [Mediasoup router observation](#mediasoup-router-observation) for the full design and examples.
388
+
389
+ #### Call level — scope `{ observer, observedCall }`
390
+
391
+ | Event | Extra | Fires when |
392
+ |-------|-------|-----------|
393
+ | `call-added` | — | a call is created |
394
+ | `call-updated` | `{ context?: AcceptContext }` | `call.update()` ran |
395
+ | `call-closed` | — | the call closed |
396
+ | `call-empty` | — | last client left the call |
397
+ | `call-not-empty` | — | first client joined a previously-empty call |
398
+ | `call-issue` | `{ issue: CallIssue }` | `call.addIssue(...)` (server-side detector finding) |
399
+ | `call-summary` | `{ summary: CallSummary }` | the call is closing and a [summary](#call-summaries) was configured. Emitted **inside** `close()`, while the call is still reachable |
400
+
401
+ #### Client level — scope `{ observer, observedCall, observedClient }`
402
+
403
+ | Event | Extra | Fires when |
404
+ |-------|-------|-----------|
405
+ | `client-added` | — | a client is created |
406
+ | `client-sink-created` | `{ sink: ClientSampleSink }` | a per-client sink was created (only when `createClientSink` returns one); fires right after `client-added` |
407
+ | `client-updated` | `{ sample: ClientSample, elapsedTimeInMs: number, context?: AcceptContext }` | the client processed a sample |
408
+ | `client-closed` | — | the client closed |
409
+ | `client-joined` | — | first `CLIENT_JOINED` event seen |
410
+ | `client-left` | — | `CLIENT_LEFT` seen (or inferred on close) |
411
+ | `client-rejoined` | `{ timestamp: number }` | a later `CLIENT_JOINED` after an earlier join |
412
+ | `client-issue` | `{ issue: ClientIssue }` | a client-reported issue arrived, or `client.addIssue(...)`. A keyed issue also opens an entry in `observedClient.activeIssues` |
413
+ | `client-issue-resolved` | `{ resolvedIssue: ResolvedActiveClientIssue }` | a stateful issue ended — the client sent its `<type>-resolved` companion, or the observer force-closed it. Carries the finished interval (`durationInMs`, `resolvedBy`) — see [client issues](#client-issues-the-lifecycle-and-the-division-of-labour) |
414
+ | `client-metadata` | `{ metaData: ClientMetaData }` | a client meta item arrived |
415
+ | `client-extension-stats` | `{ extensionStats: ExtensionStat }` | an app-defined extension stat arrived |
416
+ | `client-event` | `{ event: ClientEvent }` | any client event was processed |
417
+
418
+ #### Peer-connection level — scope `{ observer, observedCall, observedClient, observedPeerConnection }`
419
+
420
+ | Event | Extra | Notes |
421
+ |-------|-------|-------|
422
+ | `peer-connection-added` / `peer-connection-closed` | — | lifecycle of the PC |
423
+ | `peer-connection-updated` | `{ context?: AcceptContext }` | the PC processed a sample |
424
+ | `ice-connection-state-changed` / `ice-gathering-state-changed` / `connection-state-changed` | `{ state: string }` | driven by client events |
425
+ | `inbound-track-added` / `-updated` / `-removed` / `-muted` / `-unmuted` | `{ observedInboundTrack }` | |
426
+ | `outbound-track-added` / `-updated` / `-removed` / `-muted` / `-unmuted` | `{ observedOutboundTrack }` | |
427
+ | `inbound-rtp-added` / `-updated` / `-removed` | `{ observedInboundRtp }` | `-updated` fires every tick |
428
+ | `outbound-rtp-added` / `-updated` / `-removed` | `{ observedOutboundRtp }` | `-updated` fires every tick |
429
+ | `remote-inbound-rtp-added` / `-updated` / `-removed` | `{ observedRemoteInboundRtp }` | |
430
+ | `remote-outbound-rtp-added` / `-updated` / `-removed` | `{ observedRemoteOutboundRtp }` | |
431
+ | `data-channel-added` / `-updated` / `-removed` | `{ observedDataChannel }` | |
432
+ | `ice-candidate-added` / `-updated` / `-removed` | `{ observedIceCandidate }` | |
433
+ | `ice-candidate-pair-added` / `-updated` / `-removed` | `{ observedIceCandidatePair }` | |
434
+ | `ice-transport-added` / `-updated` / `-removed` | `{ observedIceTransport }` | |
435
+ | `codec-added` / `-updated` / `-removed` | `{ observedCodec }` | |
436
+ | `media-source-added` / `-updated` / `-removed` | `{ observedMediaSource }` | |
437
+ | `media-playout-added` / `-updated` / `-removed` | `{ observedMediaPlayout }` | |
438
+ | `peer-connection-transport-added` / `-updated` / `-removed` | `{ observedPeerConnectionTransport }` | |
439
+ | `certificate-added` / `-updated` / `-removed` | `{ observedCertificate }` | |
440
+
441
+ > **Volume note.** The `*-updated` sub-stat events fire on every peer-connection `accept()`
442
+ > (i.e. per sample, per stream). For high-throughput servers, subscribe only to what you need,
443
+ > or read fields off the entities on `client-updated` / `call-updated` instead.
444
+
445
+ ### Local lifecycle events
446
+
447
+ These remain on the individual entities (not the bus), for teardown/coordination. You can
448
+ listen to them, but prefer the bus equivalents above for application logic.
449
+
450
+ | Entity | Local events |
451
+ |--------|--------------|
452
+ | `ObservedCall` | `update`, `newclient`, `empty`, `not-empty`, `close` |
453
+ | `ObservedClient` | `update` (`sample`, `elapsedTimeInMs`), `close`, `joined`, `left` |
454
+ | `ObservedPeerConnection` | `removed-inbound-track`, `removed-outbound-track`, `close` |
455
+
456
+ ---
457
+
458
+ ## API reference
459
+
460
+ ### `Observer`
174
461
 
175
- **Key Properties**
462
+ ```ts
463
+ new Observer<AppData>(config?: ObserverConfig<AppData>)
176
464
 
177
- - `observedCalls: Map<string, ObservedCall>`: Active calls.
178
- - `observedTURN: ObservedTURN`: Aggregated TURN metrics.
179
- - `appData: AppData | undefined`: Custom application data.
180
- - `closed: boolean`: True if `close()` has been called.
181
- - Counters: `totalAddedCall`, `totalRemovedCall`, RTT buckets, `totalClientIssues`, `numberOfClientsUsingTurn`, `numberOfClients`, `numberOfPeerConnections`, etc.
465
+ type ObserverConfig<AppData = Record<string, unknown>> = {
466
+ // a call updates when any client does; the observer when any call does. Default true.
467
+ autoUpdateOnCallUpdate?: boolean;
468
+ appData?: AppData;
469
+ closeClientIfIdleForMs?: number;
470
+ closeCallIfEmptyForMs?: number;
471
+ // accumulate a per-call summary (see Call summaries). Absent or null = off, and nothing
472
+ // subscribes to anything. `{}` is valid: a summary with no built-in sections.
473
+ callSummary?: Partial<CallSummaryConfig> | null;
474
+ // appData factories — run when an entity is created without explicit appData
475
+ // (incl. lazily by accept()). appData is application-owned; accept `context` never touches it.
476
+ createCallAppData?: (p: { callId: string; observer: Observer; acceptCtx?: AcceptContext }) => Record<string, unknown>;
477
+ createClientAppData?: (p: { clientId: string; observedCall: ObservedCall; acceptCtx?: AcceptContext }) => Record<string, unknown>;
478
+ // sink factory — produces a per-client sink that receives every accepted sample (see Sinks).
479
+ createClientSink?: (p: { clientId: string; observedCall: ObservedCall }) => ClientSampleSink | undefined;
480
+ // remote-track-resolver factory — produces a call's RemoteTrackResolver (see Remote track resolution).
481
+ createRemoteTrackResolver?: (observedCall: ObservedCall) => RemoteTrackResolver | undefined;
482
+ };
483
+ ```
182
484
 
183
- **Key Methods**
485
+ **appData factories.** Instead of pre-creating a call/client (or assigning on `call-added` /
486
+ `client-added`) just to enrich its `appData`, register a factory once. It runs whenever the entity
487
+ is created without an explicit `settings.appData` — including the lazy creation inside `accept()`.
488
+ The `client` factory receives the already-created parent `observedCall`, so it can derive fields
489
+ from it.
490
+
491
+ Both also receive **`acceptCtx`**: the [`AcceptContext`](#context-the-acceptcontext) of the
492
+ `accept()` that caused the creation, or `undefined` when you created the entity yourself. This is
493
+ what lets an [accept middleware](#accept-middlewares-global-pre-dispatch-hook) resolve something
494
+ once — a tenant, a trace id — and have it land in `appData` at birth, instead of every factory
495
+ re-deriving it from the sample.
496
+
497
+ ```ts
498
+ const observer = new Observer({
499
+ createCallAppData: ({ callId, acceptCtx }) => ({ callId, startedAt: Date.now(), tenant: acceptCtx?.tenant }),
500
+ createClientAppData: ({ clientId, observedCall }) => ({ clientId, tenant: observedCall.appData.tenant }),
501
+ });
184
502
 
185
- - `createObservedCall<T>(settings: ObservedCallSettings<T>): ObservedCall<T>`
186
- - `getObservedCall<T>(callId: string): ObservedCall<T> | undefined`
187
- - `accept(sample: ClientSample): void`: A convenience method to feed WebRTC stats. If `sample.callId` and `sample.clientId` are provided, it will route the sample to the appropriate `ObservedCall` and `ObservedClient`, creating them if they don't exist. The core sample processing for an existing client happens within the `ObservedClient`'s own `accept` or update mechanism.
188
- - `update(): void`: Manually trigger an update cycle (behavior depends on `updatePolicy`).
189
- - `close(): void`: Cleans up resources for the observer and all its calls.
190
- - `createEventMonitor<CTX>(ctx?: CTX): ObserverEventMonitor<CTX>`: For contextual event listening.
503
+ observer.accept(sample, { tenant: 'acme' });
504
+ ```
191
505
 
192
- **Events (`ObserverEvents`)**
506
+ `appData` stays application-owned: the context is *offered* to the factory, never written across by
507
+ the library, and it is still not stored on any entity.
508
+
509
+ Key members:
510
+
511
+ - `accept(sample: ClientSample, context?: AcceptContext): void`
512
+ - `addAcceptMiddleware(...mw: AcceptMiddleware[]): this` / `removeAcceptMiddleware(...mw): this` — global pre-dispatch sample hooks (see [Accept middlewares](#accept-middlewares-global-pre-dispatch-hook))
513
+ - `getObservedCall<T>(callId): ObservedCall<T> | undefined`
514
+ - `createObservedCall<T>(settings, acceptCtx?): ObservedCall<T> | undefined`
515
+ - `getOrCreateObservedCall<T>(settings, acceptCtx?): ObservedCall<T> | undefined`
516
+ - `addIssue(issue: Omit<ObserverIssue, 'scope'>): void` — raise an **observer-level** finding → emits `observer-issue`. `scope` is stamped for you
517
+ - `update(): void` — force an aggregation/`observer-updated` tick
518
+ - `addObserverDetector(name, config?): this` — build a cross-call detector onto `observer.detectors`
519
+ - `addCallDetector(name, config?): this` — register a call-scoped detector for every call created
520
+ from now on
521
+ - `removeCallDetector(name, { includeOpenCalls? }): number` — stop building it, and (by default) drop
522
+ it from calls already open. Returns how many live instances were removed
523
+ - `removeObserverDetector(name): number` — remove an observer-scoped detector. For one specific
524
+ instance use `observer.detectors.remove(detector)`
525
+ - `addValidator(name, config?): this` — start a one-shot structural check
526
+ - `cancelValidator(name | validator, reason?): number` — stop a running check; it finishes
527
+ `inconclusive` with the reason and emits `validation-ready`
528
+ - `close(): void`
529
+ - `readonly detectors: Detectors` — observer-scoped registry. **Starts empty**; nothing is implicit
530
+ - `readonly callDetectorConfigs: Map<name, config>` — what `addCallDetector` recorded
531
+ - `readonly callSummaryCollector?: CallSummaryCollector` — owns the resolved `config.callSummary`,
532
+ the summary subscriptions, and the summaries. `undefined` when summaries are off, which is the
533
+ only place that answer lives
534
+ - `readonly validators: Set<RunningValidator>` — normally empty; each removes itself on finishing
535
+ - `readonly activeIssuesRegistry: ActiveIssuesRegistry` — the fleet's open client issues
536
+ - `readonly observedCalls: Map<string, ObservedCall>`
537
+ - `readonly observedTURN: ObservedTURN`
538
+ - `get appData()`, `get numberOfCalls()`
539
+ - counters: `numberOfClients`, `numberOfClientsUsingTurn`, `numberOfInboundRtpStreams`,
540
+ `numberOfOutboundRtpStreams`, `numberOfDataChannels`, `numberOfPeerConnections`,
541
+ `totalAddedCall`, `totalRemovedCall`, `closed`
542
+ - `on/off/once/emit` typed against the [event map](#event-catalogue)
543
+
544
+ ### `ObservedCall`
545
+
546
+ ```ts
547
+ type ObservedCallSettings<AppData = Record<string, unknown>> = {
548
+ // update this call whenever one of its clients accepts a sample. Default true.
549
+ autoUpdateOnClientUpdate?: boolean;
550
+ callId: string;
551
+ appData?: AppData;
552
+ closeCallIfEmptyForMs?: number;
553
+ };
554
+ ```
193
555
 
194
- - `'newcall' (call: ObservedCall)`
195
- - `'call-updated' (call: ObservedCall)`
196
- - `'client-event' (client: ObservedClient, event: ClientEvent)`
197
- - `'client-issue' (client: ObservedClient, issue: ClientIssue)`
198
- - `'client-metadata' (client: ObservedClient, metadata: ClientMetaData)`
199
- - `'client-extension-stats' (client: ObservedClient, stats: ExtensionStat)`
200
- - `'update' ()`
201
- - `'close' ()`
556
+ Key members:
557
+
558
+ - `readonly callId: string`, `appData: AppData`
559
+ - `readonly observedClients: Map<string, ObservedClient>`, `get numberOfClients()`
560
+ - `getObservedClient<T>(clientId)`, `createObservedClient<T>(settings, acceptCtx?)`, `getOrCreateObservedClient<T>(settings, acceptCtx?)` (all `… | undefined`)
561
+ - `addIssue(issue: Omit<CallIssue, 'scope'>): void` — raise a **call-level** finding → emits `call-issue`. `scope` is stamped for you
562
+ - `addDetector(name, config?): this` — build a call-scoped detector onto this call only
563
+ - `removeDetector(name): number` — remove it from this call, `close()`ing it. For one specific
564
+ instance use `call.detectors.remove(detector)`
565
+ - `readonly detectors: Detectors` — server-side detector registry (empty by default; see [Detectors](#detectors-server-side-extension-point))
566
+ - `readonly activeIssuesRegistry: ActiveIssuesRegistry` — this call's open client issues, propagating into the observer's
567
+ - `readonly unconsumedOutboundTracks: Set<ObservedOutboundTrack>` — maintained by the resolver
568
+ - `scoreCalculator: ScoreCalculator`, `get score()`, `readonly calculatedScore`
569
+ - `remoteTrackResolver?: RemoteTrackResolver` — set from `ObserverConfig.createRemoteTrackResolver` at call creation (see [Remote track resolution](#remote-track-resolution-mediasoup--sfu))
570
+ - aggregates: `numberOfIssues`, `numberOfPeerConnections`, `numberOfInboundRtpStreams`,
571
+ `numberOfOutboundRtpStreams`, `numberOfDataChannels`, `maxNumberOfClients`,
572
+ `clientsUsedTurn: Set<string>`, `startedAt?`, `endedAt?`, `closedAt?`, `closed`
573
+ - `summary?: CallSummary` — the live record of this call, when [summaries](#call-summaries) are on
574
+ - `update()`, `close()`
575
+
576
+ ### `ObservedClient`
577
+
578
+ ```ts
579
+ type ObservedClientSettings<AppData = Record<string, unknown>> = {
580
+ clientId: string;
581
+ appData?: AppData;
582
+ closeClientIfIdleForMs?: number;
583
+ };
584
+ ```
202
585
 
203
- #### 3.2. `ObservedCall`
586
+ Key members:
587
+
588
+ - `readonly clientId: string`, `appData: AppData`, `readonly call: ObservedCall`
589
+ - `readonly observedPeerConnections: Map<string, ObservedPeerConnection>`
590
+ - `readonly sink?: ClientSampleSink` — the per-client sink (see [Sinks](#sinks-per-client-sample-persistence)), if `createClientSink` is configured; listen on it for `close`/`error`
591
+ - **Injection API** (queue app data to be merged into the next sample processing):
592
+ `injectEvent(ClientEvent)`, `injectIssue(ClientIssue)`, `injectMetaData(ClientMetaData)`,
593
+ `injectExtensionStat(ExtensionStat)`, `injectAttachment(attachments: Record<string, unknown>)`
594
+ - **Direct add API** (process immediately): `addIssue(ClientIssue)`, `addMetadata(ClientMetaData)`,
595
+ `addExtensionStats(ExtensionStat)`
596
+ - Metrics (current/derived): `currentAvgRttInMs?`, `currentMinRttInMs?`, `currentMaxRttInMs?`,
597
+ `receivingAudioBitrate`, `receivingVideoBitrate`, `sendingAudioBitrate`, `sendingVideoBitrate`,
598
+ `usingTURN`, `usingTCP`, `availableIncomingBitrate`, `availableOutgoingBitrate`
599
+ - Counts: `numberOfInboundRtpStreams`, `numberOfOutboundRtpStreams`, `numberOfInbundTracks`,
600
+ `numberOfOutboundTracks`, `numberOfDataChannels`, `numberOfPeerConnections`
601
+ - Per-tick deltas: `deltaReceivedAudioBytes`, `deltaSentAudioBytes`, … (see source for the full set)
602
+ - Lifecycle: `joinedAt?`, `leftAt?`, `closedAt?`, `closed`, `get score()`
603
+ - Metadata: `browser?`, `engine?`, `platform?`, `operationSystem?`, `mediaDevices`, `mediaConstraints`
604
+ - `accept(sample, context?)`, `close()`
605
+
606
+ ### `ObservedPeerConnection`
607
+
608
+ Key members:
609
+
610
+ - `readonly peerConnectionId: string`, `readonly client: ObservedClient`, `appData?`
611
+ - The 15 `observed*` sub-stat `Map`s (listed [above](#entity-hierarchy)), plus array getters:
612
+ `codecs`, `inboundRtps`, `outboundRtps`, `remoteInboundRtps`, `remoteOutboundRtps`,
613
+ `mediaSources`, `mediaPlayouts`, `dataChannels`, `peerConnectionTransports`, `iceTransports`,
614
+ `iceCandidates`, `iceCandidatePairs`, `certificates`, `selectedIceCandidatePairs`,
615
+ `selectedIceCandiadtePairForTurn`
616
+ - State: `connectionState?`, `iceConnectionState?`, `iceGatheringState?`, `usingTURN`, `usingTCP`
617
+ - Metrics: `currentRttInMs?`, `iceRttInMs?`, `rtcpRttInMs?`, `sfuHopRttInMs?`, `currentJitter?`,
618
+ `availableIncomingBitrate`, `availableOutgoingBitrate`, sending/receiving bitrates, packet rates,
619
+ and `total*` / `delta*` byte/packet counters
620
+ - `accept(pcSample, context?)`, `close()`, `get score()`
621
+
622
+ **Two different round trips — don't mix them.** `iceRttInMs` comes from ICE/STUN consent checks and
623
+ measures the trip to *whatever terminates ICE*: **in an SFU topology that is the SFU**, so it is the
624
+ client↔SFU leg. `rtcpRttInMs` comes from RTCP receiver reports and is an **end-to-end** media-path
625
+ round trip. They are not interchangeable, and averaging them together produces a number that moves
626
+ as streams come and go for reasons unrelated to the network. `currentRttInMs` therefore *prefers*
627
+ RTCP and falls back to ICE — always one kind within a tick, never a blend. `sfuHopRttInMs`
628
+ (`rtcp − ice`) estimates everything past the SFU, which separates "this client's last mile is slow"
629
+ from "the path beyond the SFU is slow".
630
+
631
+ **Counter-reset boundaries.** Chrome resets an SSRC's cumulative counters when the codec switches
632
+ ([crbug/webrtc/5361](https://bugs.chromium.org/p/webrtc/issues/detail?id=5361), open since 2015),
633
+ which otherwise shows up as a sawtooth spike or a negative bitrate. `ObservedInboundRtp` /
634
+ `ObservedOutboundRtp` therefore set **`counterResetBoundary`** on any tick where `codecId`,
635
+ `encoder`/`decoderImplementation` or `scalabilityMode` changed, and suppress every delta for that
636
+ tick. Without this, a room-wide codec rollout fires a synchronized fake-degradation alert across
637
+ every participant at once.
638
+
639
+ **Remote-RTP correlation (derived).** During `accept()`, receiver/sender reports are linked
640
+ to the local streams by `remoteId` (fallback SSRC) and surfaced as fields:
641
+
642
+ - on `ObservedOutboundRtp`: `remoteRttInMs?`, `remoteFractionLost?`, `remoteJitter?`, `remotePacketsLost?`
643
+ - on `ObservedInboundRtp`: `remoteRttInMs?`, `remoteBytesSent?`, `remotePacketsSent?`, `remoteTimestamp?`
644
+
645
+ These are reset each tick and only set when the matching remote report is present.
204
646
 
205
- Represents a single call session.
647
+ ---
206
648
 
207
- **Configuration (`ObservedCallSettings`)**
649
+ ## Schema types (`ClientSample`)
650
+
651
+ The shape of an accepted sample (re-exported from this package; identical to
652
+ `@observertc/schemas`). Only the top level is shown — each stat object mirrors the standard
653
+ WebRTC `getStats()` dictionaries plus a few extensions.
654
+
655
+ ```ts
656
+ type ClientSample = {
657
+ timestamp: number; // client wall-clock (ms epoch)
658
+ callId?: string; // set by you or the library
659
+ clientId?: string; // set by you or the library
660
+ score?: number; // optional client-computed score (0..5)
661
+ attachments?: Record<string, unknown>;
662
+ peerConnections?: PeerConnectionSample[];
663
+ clientEvents?: ClientEvent[];
664
+ clientIssues?: ClientIssue[];
665
+ clientMetaItems?: ClientMetaData[];
666
+ extensionStats?: ExtensionStat[];
667
+ };
208
668
 
209
- ```typescript
210
- export type ObservedCallSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
211
- callId: string;
212
- appData?: AppData;
213
- updatePolicy?: 'update-on-any-client-updated' | 'update-when-all-client-updated' | 'update-on-interval';
214
- updateIntervalInMs?: number; // Used if updatePolicy is 'update-on-interval'
215
- remoteTrackResolvePolicy?: 'mediasoup-sfu'; // For specific SFU integration
669
+ type PeerConnectionSample = {
670
+ peerConnectionId: string;
671
+ attachments?: Record<string, unknown>; // e.g. { direction: 'send'|'recv', producerId, consumerId, label }
672
+ score?: number;
673
+ inboundTracks?; outboundTracks?;
674
+ codecs?;
675
+ inboundRtps?; remoteInboundRtps?;
676
+ outboundRtps?; remoteOutboundRtps?;
677
+ mediaSources?; mediaPlayouts?;
678
+ peerConnectionTransports?; dataChannels?;
679
+ iceTransports?; iceCandidates?; iceCandidatePairs?;
680
+ certificates?;
216
681
  };
682
+
683
+ type ClientEvent = { type: string; payload?: string; timestamp?: number; /* +ids */ };
684
+ type ClientIssue = { type: string; payload?: string; timestamp?: number };
685
+ type ClientMetaData = { type: string; payload?: string; timestamp?: number; /* +ids */ };
686
+ type ExtensionStat = { type: string; payload?: string };
217
687
  ```
218
688
 
219
- **Key Properties**
689
+ `payload` fields are JSON strings; the library parses the ones it understands.
690
+
691
+ **`ClientEventTypes`** (enum of known `event.type` values): `CLIENT_JOINED`, `CLIENT_LEFT`,
692
+ `PEER_CONNECTION_OPENED/CLOSED/STATE_CHANGED`, `MEDIA_TRACK_ADDED/REMOVED/MUTED/UNMUTED/RESUMED`,
693
+ `ICE_GATHERING_STATE_CHANGED`, `ICE_CONNECTION_STATE_CHANGED`, `DATA_CHANNEL_OPEN/CLOSED/ERROR`,
694
+ `NEGOTIATION_NEEDED`, `SIGNALING_STATE_CHANGE`, `ICE_CANDIDATE`, `ICE_CANDIDATE_ERROR`, and the
695
+ mediasoup set `PRODUCER_*` / `CONSUMER_*` / `DATA_PRODUCER_*` / `DATA_CONSUMER_*`.
696
+
697
+ **`ClientMetaTypes`** (enum of known meta `type` values): `MEDIA_CONSTRAINT`, `MEDIA_DEVICE`,
698
+ `MEDIA_DEVICES_SUPPORTED_CONSTRAINTS`, `USER_MEDIA_ERROR`, `LOCAL_SDP`, `OPERATION_SYSTEM`,
699
+ `ENGINE`, `PLATFORM`, `BROWSER`.
700
+
701
+ ### Worked example: a real `ClientSample`
702
+
703
+ Two consecutive samples from one participant ("Guest" in room `qq0iwfnd`) of an
704
+ edumeet/mediasoup call show what actually flows through `accept()`: a rich **join snapshot**,
705
+ then lean **steady-state ticks**.
706
+
707
+ **Sample 1 — the join snapshot.** Carries the one-off lifecycle `clientEvents` and device
708
+ `clientMetaItems` alongside the first stats. (Abbreviated; ids and times are from the real log.)
709
+
710
+ ```jsonc
711
+ {
712
+ "timestamp": 1780572332518,
713
+ "callId": "d3dbf2f5-79be-4cb8-9d43-fb404f07ef27",
714
+ "clientId": "c926983c-4468-4046-ae8c-a9cabe1a1868",
715
+ "score": 0, // no quality measured yet on the join tick
716
+ "attachments": { "displayName": "Guest", "roomId": "qq0iwfnd", "actualSessionId": "d3dbf2f5-…" },
717
+
718
+ "clientEvents": [ // chronological lifecycle (12 in the real sample)
719
+ { "type": "CLIENT_JOINED", "timestamp": 1780572324515 },
720
+ { "type": "PEER_CONNECTION_OPENED", "timestamp": 1780572326790 }, // pc=b81c8d9d (media)
721
+ { "type": "ICE_GATHERING_STATE_CHANGED", "timestamp": 1780572326811 }, // → gathering
722
+ { "type": "PEER_CONNECTION_STATE_CHANGED", "timestamp": 1780572326812 }, // → connecting
723
+ { "type": "PRODUCER_ADDED", "timestamp": 1780572326821 }, // producer=1abdaf82 (audio)
724
+ { "type": "MEDIA_TRACK_ADDED", "timestamp": 1780572326821 }, // track=36ae42df (audio)
725
+ { "type": "PEER_CONNECTION_STATE_CHANGED", "timestamp": 1780572326827 }, // → connected
726
+ { "type": "PRODUCER_ADDED", "timestamp": 1780572326837 }, // producer=ba06a35b (video)
727
+ { "type": "DATA_PRODUCER_CREATED", "timestamp": 1780572326853 }
728
+ ],
729
+
730
+ "clientMetaItems": [ // environment & devices, one-off (10 in the real sample)
731
+ { "type": "USER_AGENT_DATA", "payload": "{…Chrome 148 / macOS…}" },
732
+ { "type": "MEDIA_DEVICE", "payload": "{…\"BRIO 4K Stream Edition\"…}" }
733
+ // …mic / camera / speaker devices…
734
+ ],
735
+
736
+ "peerConnections": [
737
+ {
738
+ "peerConnectionId": "b81c8d9d-…", // the media PC — Guest publishes to the SFU
739
+ "outboundRtps": [ /* audio + video */ ],
740
+ "outboundTracks": [ /* mic + camera: label, settings, capabilities */ ],
741
+ "remoteInboundRtps": [ /* RTCP feedback from the SFU */ ],
742
+ "codecs": [ /* … */ ], "iceTransports": [ /* … */ ],
743
+ "iceCandidatePairs": [ /* … */ ], "dataChannels": [ /* … */ ]
744
+ },
745
+ { "peerConnectionId": "8635acb7-…", "peerConnectionTransports": [ /* … */ ] } // signaling-only PC
746
+ ]
747
+ }
748
+ ```
220
749
 
221
- - `callId: string`
222
- - `appData: AppData | undefined`
223
- - `numberOfClients: number`
224
- - `score: number | undefined`: Overall call quality score.
225
- - `observedClients: Map<string, ObservedClient>`
226
- - Counters: `totalAddedClients`, `totalRemovedClients`, `numberOfIssues`, RTT buckets, total bytes sent/received (audio/video/data), etc.
750
+ What `accept()` does with it, in order — each step emits on the bus with full ancestry:
227
751
 
228
- **Key Methods**
752
+ 1. lazily creates the `ObservedCall` → **`call-added`**;
753
+ 2. creates the `ObservedClient` → **`client-added`**, then **`client-joined`** (from `CLIENT_JOINED`);
754
+ 3. creates an `ObservedPeerConnection` per entry → **`peer-connection-added`** (×2 here);
755
+ 4. creates an `ObservedOutboundTrack` per track → **`outbound-track-added`**, plus the matching
756
+ **`outbound-rtp-added`**;
757
+ 5. replays the device list as **`client-metadata`** events and the lifecycle items as
758
+ **`client-event`**; and finally **`client-updated`** for the whole tick.
229
759
 
230
- - `createObservedClient<T>(settings: ObservedClientSettings<T>): ObservedClient<T>`
231
- - `getObservedClient<T>(clientId: string): ObservedClient<T> | undefined`
232
- - `update(): void`
233
- - `close(): void`
234
- - `createEventMonitor<CTX>(ctx?: CTX): ObservedCallEventMonitor<CTX>`
760
+ `attachments.roomId` lands on `observedClient.attachments` (read it on `client-updated`, **not** at
761
+ creation — see [Ingestion](#ingestion-accept-context--lifecycle)).
235
762
 
236
- **Events (Emitted via `ObservedCall` instance)**
763
+ **Sample 2 — a steady-state tick** (~8 s later): same `callId` / `clientId`, **no** new
764
+ `clientEvents` or `clientMetaItems`, just refreshed `peerConnections` stats. Each PC now scores `5`
765
+ and the aggregate client `score` is `4.74` — a healthy call. This is the shape of nearly every
766
+ sample: each tick refreshes metrics and fires the `*-updated` events, while the heavy join
767
+ snapshot happens only once.
237
768
 
238
- - `'newclient' (client: ObservedClient)`
239
- - `'empty' ()`: When the last client leaves.
240
- - `'not-empty' ()`: When the first client joins an empty call.
241
- - `'update' ()`
242
- - `'close' ()`
769
+ ---
243
770
 
244
- #### 3.3. `ObservedClient`
771
+ ## Detectors (server-side extension point)
245
772
 
246
- Represents a participant in a call.
773
+ `observer-js` ships [ten detectors](#built-in-detectors), each an opt-in extension you register
774
+ explicitly with `addObserverDetector` / `addCallDetector` / `addDetector` (see
775
+ [Registering detectors](#registering-detectors)) — none are created automatically. All of them
776
+ correlate **across** the clients of a call or the calls of a fleet, because that is the only thing a
777
+ server can do better than a browser: per-client signals — packet loss, jitter, RTT, freezes — are
778
+ already detected on the client and arrive on samples as `clientIssues` (surfaced via `client-issue`).
247
779
 
248
- **Configuration (`ObservedClientSettings`)**
780
+ Findings are raised as **`CallIssue`** or **`ObserverIssue`** — both share `IssueBase`: `{ type,
781
+ timestamp, conclusion?, payload? }` — and the payload is the **object**, not a JSON string. A
782
+ server-raised finding is delivered to an in-process handler, so there is nothing to serialise for:
249
783
 
250
- ```typescript
251
- export type ObservedClientSettings<AppData extends Record<string, unknown> = Record<string, unknown>> = {
252
- clientId: string;
253
- appData?: AppData;
254
- // Potentially other client-specific settings
255
- };
784
+ ```ts
785
+ observer.on('call-issue', ({ observedCall, issue }) => {
786
+ issue.payload; // the object; no JSON.parse
787
+ issue.conclusion?.faultDomain; // a first-class field, not payload.conclusion
788
+ issuePayloadAsString(issue); // only at an edge that needs text (log, HTTP, queue)
789
+ });
256
790
  ```
257
791
 
258
- **Key Properties**
792
+ (`ClientIssue`, the type on samples, keeps its string payload — that one really is a wire format.)
259
793
 
260
- - `clientId: string`
261
- - `call: ObservedCall`: Reference to the parent call.
262
- - `appData: AppData | undefined`
263
- - `score: number | undefined`: Client quality score.
264
- - `numberOfPeerConnections: number`
265
- - `usingTURN: boolean`
266
- - `observedPeerConnections: Map<string, ObservedPeerConnection>`
267
- - Counters: `numberOfIssues`, RTT buckets, total bytes sent/received, `availableIncomingBitrate`, `availableOutgoingBitrate`, etc.
794
+ The registry is also an open extension point, on **`ObservedCall`**:
268
795
 
269
- **Key Methods**
796
+ ```ts
797
+ import { Observer, Detector } from '@observertc/observer-js';
270
798
 
271
- - `accept(sample: ClientSample): void`: (Or a similar internal update method called by `Observer.accept` or `ObservedCall`) Processes a `ClientSample` specific to this client, updating its metrics, peer connections, streams, etc. This is the primary point where a client's detailed WebRTC statistics are processed.
272
- - `createObservedPeerConnection<T>(settings: ObservedPeerConnectionSettings<T>): ObservedPeerConnection<T>`
273
- - `getObservedPeerConnection<T>(peerConnectionId: string): ObservedPeerConnection<T> | undefined`
274
- - `update(): void`
275
- - `close(): void`
276
- - `createEventMonitor<CTX>(ctx?: CTX): ObservedClientEventMonitor<CTX>`
799
+ class MyCrossClientDetector implements Detector {
800
+ readonly name = 'my-detector';
801
+ constructor(private readonly call /* : ObservedCall */) {}
802
+ update() { // called on every call.update()
803
+ // …inspect this.call.observedClients across participants…
804
+ if (/* condition only visible server-side */ false) {
805
+ this.call.addIssue({ type: this.name, payload: { /* … */ }, timestamp: Date.now() });
806
+ // → emitted on the bus as 'call-issue'
807
+ }
808
+ }
809
+ }
810
+
811
+ const observer = new Observer();
812
+ observer.on('call-added', ({ observedCall }) => {
813
+ observedCall.detectors.add(new MyCrossClientDetector(observedCall));
814
+ });
815
+ observer.on('call-issue', ({ observedCall, issue }) => { /* react */ });
816
+ ```
817
+
818
+ ### Client issues: the lifecycle, and the division of labour
277
819
 
278
- **Events (Emitted via `ObservedClient` instance)**
820
+ The most important thing to understand about detection in this library is **what it deliberately
821
+ does not do**. A client running
822
+ [`client-monitor-js`](https://github.com/ObserveRTC/client-monitor-js) already ships ~20 detectors
823
+ that decide *what is wrong with that endpoint* — `congestion`, `cpulimitation`, `audio-concealment`,
824
+ `freezed-video-track`, `keyframe-storm`, `video-decoder-overloaded`, `stuck-decoder`,
825
+ `ice-disconnected`, and so on. Those verdicts are better than anything re-derived from raw counters
826
+ server-side, because they carry hysteresis and multi-signal confirmation: `audio-concealment`
827
+ subtracts silent concealment (raw `concealedSamples` rises during ordinary silence, so a naive
828
+ detector flags every quiet moment); `audio-jitter-buffer-stress` requires the buffer to be grown
829
+ **and** NetEQ to be time-stretching (a grown buffer alone means NetEQ is *succeeding*);
830
+ `ice-disconnected` only fires once `disconnected` has persisted, so the blips ICE heals on its own
831
+ never surface.
279
832
 
280
- - `'joined' ()`
281
- - `'left' ()`
282
- - `'update' ()`
283
- - `'close' ()`
284
- - `'newpeerconnection' (pc: ObservedPeerConnection)`
285
- - `'issue' (issue: ClientIssue)` (and other specific issue events)
833
+ **observer-js does not repeat that work.** Its job is the question no browser can answer: *who else
834
+ is in this state right now, what do they have in common, and where in publisher → SFU → subscriber
835
+ does the fault begin?*
286
836
 
287
- #### 3.4. `ObservedPeerConnection`
837
+ #### The wire format
288
838
 
289
- Represents an `RTCPeerConnection`.
839
+ From client-monitor-js **4.6.0** the whole issue lifecycle reaches the server. A stateful issue
840
+ arrives as two `clientIssues[]` entries sharing a `key`:
841
+
842
+ ```
843
+ raise: { type: 'stuck-decoder', key, payload, timestamp: raisedAt }
844
+ resolution: { type: 'stuck-decoder-resolved', key, payload: { raisedAt, comment, …final }, timestamp: resolvedAt }
845
+ ```
290
846
 
291
- - Tracks ICE connection state, data channel stats, stream stats.
292
- - Holds `ObservedInboundRtpStream`, `ObservedOutboundRtpStream`, and `ObservedDataChannel` instances.
847
+ The observer opens an entry in `observedClient.activeIssues` on the raise and closes it on the
848
+ matching key, emitting **`client-issue-resolved`** with the finished interval. Handled for you:
849
+
850
+ - the `-resolved` **suffix** is stripped, so both entries share one logical `type`;
851
+ - a **re-raise** of a live key refreshes the payload without restarting `raisedAt`;
852
+ - **keyless** entries are one-shot — reported via `client-issue`, never tracked;
853
+ - issues still open when a client closes are **force-resolved** (`resolvedBy: 'client-closed'`), and
854
+ the registry additionally expires stale entries, so a crashed participant can't leave an issue
855
+ "active" forever.
856
+
857
+ #### Why intervals beat windows
858
+
859
+ This turns point-in-time symptom reports into **intervals**, and that is the whole game. *"Several
860
+ clients reported congestion in the last 10 seconds"* is a heuristic that has to guess whether the
861
+ symptoms are still happening. *"Several clients are congested **right now, simultaneously**"* is
862
+ ground truth, because the client says when the episode ends. Overlapping intervals are far stronger
863
+ evidence of a shared cause than near-in-time reports.
864
+
865
+ ```ts
866
+ observer.on('client-issue', ({ observedClient, issue }) => { /* opened (or one-shot) */ });
867
+ observer.on('client-issue-resolved', ({ resolvedIssue }) => {
868
+ resolvedIssue.type; // 'stuck-decoder' — suffix stripped
869
+ resolvedIssue.durationInMs; // how long the episode lasted
870
+ resolvedIssue.resolvedBy; // 'client' | 'timeout' | 'client-closed'
871
+ });
293
872
 
294
- #### 3.5. `ObservedInboundRtpStream` / `ObservedOutboundRtpStream`
873
+ // the live per-client mirror
874
+ observedClient.activeIssues; // ObservedClientIssueRegistry, keyed by issue.key
875
+ ```
295
876
 
296
- - Track metrics for individual media streams (audio/video) like codec, packets lost/received, jitter, bytes, etc.
877
+ > **client-monitor-js >= 4.6.0 is required** for every issue-driven detector. There is no fallback
878
+ > path that infers these conditions from raw counters — the client decides better, and maintaining a
879
+ > worse second implementation to be polite to old clients is how both end up wrong. Issues without a
880
+ > `key` have no lifecycle (nothing could ever close them), so they stay one-shot: reported on
881
+ > `client-issue`, never registered.
297
882
 
298
- #### 3.6. `ObservedDataChannel`
883
+ #### `ActiveIssuesRegistry` — issues are **pushed**, not polled
299
884
 
300
- - Tracks metrics for data channels like state, messages sent/received, bytes.
885
+ A detector does not go looking for the issues it cares about. It implements `ActiveIssueTracker` and
886
+ registers for the types it consumes; the registry hands them over as they open and close.
301
887
 
302
- #### 3.7. `ClientSample` (Schema)
888
+ ```ts
889
+ observedCall.activeIssuesRegistry // this meeting
890
+ observer.activeIssuesRegistry // the fleet; every call's registry propagates into it
303
891
 
304
- This is the primary input data structure passed to `observer.accept()`. It's a comprehensive object that should mirror the information obtainable from WebRTC `getStats()` and other client-side states. Key fields include:
892
+ observer.activeIssuesRegistry.addIssueTracker('congestion', myDetector);
893
+ observer.activeIssuesRegistry.removeIssueTracker(myDetector);
305
894
 
306
- - `callId`, `clientId`, `timestamp`
307
- - `peerConnections: RTCPeerConnectionStats[]`
308
- - `inboundRtpStreams: RTCInboundRtpStreamStats[]`
309
- - `outboundRtpStreams: RTCOutboundRtpStreamStats[]`
310
- - `remoteInboundRtpStreams: RTCRemoteInboundRtpStreamStats[]`
311
- - `remoteOutboundRtpStreams: RTCRemoteOutboundRtpStreamStats[]`
312
- - `dataChannels: RTCDataChannelStats[]`
313
- - `iceLocalCandidates: RTCIceCandidateStats[]`, `iceRemoteCandidates: RTCIceCandidateStats[]`, `iceCandidatePairs: RTCIceCandidatePairStats[]`
314
- - `mediaSources: RTCAudioSourceStats[] / RTCVideoSourceStats[]`
315
- - `tracks: RTCMediaStreamTrackStats[]`
316
- - `certificates: RTCCertificateStats[]`
317
- - `codecs: RTCCodecStats[]`
318
- - `transports: RTCIceTransportStats[]` (or similar depending on spec version)
319
- - `browser`, `engine`, `platform`, `os` (client environment metadata)
320
- - `userMediaErrors`, `iceConnectionStates`, `connectionStates` (client-reported events/states)
321
- - `extensionStats` (for custom data)
895
+ registry.values(); // the open issues in this scope, oldest first
896
+ registry.size; // how many
897
+ ```
322
898
 
323
- _(Refer to the [observertc/schemas](https://github.com/observertc/schemas) repository, particularly the `ClientSample.ts` definition, for the exact and complete structure.)_
899
+ The cost of a detector is then proportional to the issues it actually receives, not to the number of
900
+ participants: a healthy 500-client fleet does no per-tick work at all, because nothing was pushed.
324
901
 
325
- ### 4. Configuration Possibilities
902
+ **There is no wildcard.** A tracker names its types and sees nothing else. "Feed me everything and
903
+ I'll work out what matters" moves the decision from the application — which knows its client build
904
+ and its issue vocabulary — onto a detector that has to guess, and it makes the cost of a
905
+ subscription unbounded and invisible. If a detector should watch five types, list five types.
326
906
 
327
- #### 4.1. Update Policies
907
+ Onset spread is measured on the **observer clock**, never the client's. `raisedAt` comes from each
908
+ participant's own machine, and comparing those across clients makes clock skew look like a
909
+ synchronized infrastructure event.
328
910
 
329
- Control how frequently entities re-calculate metrics and emit `update` events.
911
+ ### Observer-level detectors (cross-call / SFU-wide)
330
912
 
331
- **Observer Level (`ObserverConfig.updatePolicy`)**
913
+ Some findings only exist **above** call scope — "many calls on the same SFU degraded at once" is far
914
+ more actionable than fifty individual client alerts. The same detector registry exists on the `Observer`,
915
+ runs on every `observer.update()`, and raises findings through `observer.addIssue(...)`, surfaced on
916
+ the bus as **`observer-issue`**:
332
917
 
333
- - `'update-on-any-call-updated'`: Observer updates if any of its calls update.
334
- - `'update-when-all-call-updated'`: Observer updates after all its calls update. (Default)
335
- - `'update-on-interval'`: Observer updates at `ObserverConfig.updateIntervalInMs`.
918
+ ```ts
919
+ observer.detectors.add({
920
+ name: 'sfu-wide-degradation',
921
+ update: () => {
922
+ const degradedCalls = [ ...observer.observedCalls.values() ].filter(isDegraded);
336
923
 
337
- **Call Level (`ObservedCallSettings.updatePolicy` or `ObserverConfig.defaultCallUpdatePolicy`)**
924
+ if (observer.numberOfCalls > 3 && degradedCalls.length / observer.numberOfCalls > 0.6) {
925
+ observer.addIssue({ type: 'SFU_WIDE_QUALITY_DEGRADATION', timestamp: Date.now() });
926
+ }
927
+ },
928
+ });
338
929
 
339
- - `'update-on-any-client-updated'`: Call updates if any of its clients update.
340
- - `'update-when-all-client-updated'`: Call updates after all its clients update.
341
- - `'update-on-interval'`: Call updates at its `updateIntervalInMs`.
930
+ observer.on('observer-issue', ({ issue }) => alert(issue));
931
+ ```
342
932
 
343
- #### 4.2. Intervals
933
+ ### Publisher → subscribers: the resolver links
344
934
 
345
- - `ObserverConfig.updateIntervalInMs`
346
- - `ObserverConfig.defaultCallUpdateIntervalInMs`
347
- - `ObservedCallSettings.updateIntervalInMs`
935
+ The question a single browser can never answer is *"did **everyone** receiving Alice see the same
936
+ degradation?"*. The join is the publisher↔subscriber links maintained by a
937
+ [`RemoteTrackResolver`](#remote-track-resolution-mediasoup--sfu), and detectors walk them directly:
348
938
 
349
- #### 4.3. Application Data (`appData`)
939
+ ```ts
940
+ outboundTrack.remoteInboundTracks; // Set<ObservedInboundTrack> — every subscriber of this source
941
+ inboundTrack.remoteOutboundTrack; // the publisher, or undefined if unlinked
942
+ inboundTrack.getInboundRtp(); // that receiver's RTP stats
943
+ observedCall.unconsumedOutboundTracks; // published tracks with no subscriber at all
944
+ ```
350
945
 
351
- Associate custom context with `Observer`, `ObservedCall`, and `ObservedClient` instances using generics.
946
+ > A `TrackDistributionAggregator` class used to sit in front of these links and summarise every
947
+ > published track against all of its receivers, on every tick. It is gone. It scanned the majority
948
+ > (all published tracks) to find the interesting minority, which is the wrong axis — the detectors
949
+ > now start from the handful of *affected* tracks the issue registry pushed at them and resolve only
950
+ > those. The statistics helpers it used (`percentile`, `median`, `summarize`, `counterDelta`,
951
+ > `robustZScore`, `SlidingWindow`, `TrendTester`) are all still exported for building your own.
352
952
 
353
- ```typescript
354
- interface MyCallAppData {
355
- meetingTitle: string;
356
- scheduledAt: Date;
357
- }
358
- const call = observer.createObservedCall<MyCallAppData>({
359
- callId: 'call1',
360
- appData: { meetingTitle: 'Team Sync', scheduledAt: new Date() },
953
+ ### Call health: `CallHealthAggregator`
954
+
955
+ The **client** axis. Where the resolver links answer "how was *this source* delivered?", this asks
956
+ "how is *each participant* doing, sending vs receiving?":
957
+
958
+ ```ts
959
+ import { CallHealthAggregator } from '@observertc/observer-js';
960
+
961
+ const health = new CallHealthAggregator(observedCall).aggregate();
962
+
963
+ health.degradedRatio; // 0.82 — the number that distinguishes shared faults from individual ones
964
+ health.inboundDegradedRatio; // receiving side → egress/downstream suspicion
965
+ health.outboundDegradedRatio; // sending side → ingress suspicion
966
+ health.rttInMs?.median; // percentile rollups, never means
967
+ health.qualityLimitation; // { cpu, bandwidth, other } client counts
968
+ health.clients; // per-client entries with `reasons`, direction flags, TURN/TCP
969
+ ```
970
+
971
+ ### Registering detectors
972
+
973
+ **Nothing is created implicitly.** A `new Observer()` has zero detectors. There is no detector
974
+ configuration in `ObserverConfig` and no default set — an application says what it wants to watch, or
975
+ it watches nothing.
976
+
977
+ ```ts
978
+ const observer = new Observer({
979
+ createRemoteTrackResolver: createDefaultMediasoupRemoteTrackResolverFactory(),
361
980
  });
362
- console.log(call.appData?.meetingTitle);
981
+
982
+ // observer-scoped (cross-call) — built immediately onto `observer.detectors`
983
+ observer.addObserverDetector('observer-concurrent-issue-detector', {
984
+ issueTypes: [ 'congestion', 'ice-disconnected', 'ice-connection-failed' ],
985
+ minAffectedCalls: 3,
986
+ });
987
+ observer.addObserverDetector('turn-server-outage-detector', { minClientsAtPeak: 10 });
988
+
989
+ // call-scoped — recorded in `observer.callDetectorConfigs`, applied to every call created AFTER this
990
+ observer.addCallDetector('call-concurrent-issue-detector', {
991
+ issueTypes: [ 'congestion', 'ice-disconnected' ],
992
+ });
993
+ // one specific call
994
+ observedCall.addDetector('issue-fan-out-detector', { issueTypes: [ 'freezed-video-track' ] });
995
+ ```
996
+
997
+ Every `add*` is **chainable** — it returns the owning entity:
998
+
999
+ ```ts
1000
+ observer
1001
+ .addObserverDetector('turn-server-health-detector')
1002
+ .addObserverDetector('turn-server-outage-detector', { minClientsAtPeak: 10 })
1003
+ .addValidator('remote-track-resolver');
1004
+ ```
1005
+
1006
+ #### Removing them
1007
+
1008
+ By **name**, on the entity — which removes *every* instance under that name:
1009
+
1010
+ ```ts
1011
+ observer.removeObserverDetector('turn-server-outage-detector'); // → 1
1012
+ observer.removeCallDetector('call-concurrent-issue-detector'); // stops it everywhere
1013
+ observedCall.removeDetector('issue-fan-out-detector'); // this call only
363
1014
  ```
364
1015
 
365
- #### 4.4. `appData` vs. Attachments
1016
+ By **instance**, through the registry — which is where instances live, since `add*` returns the
1017
+ entity rather than the detector:
1018
+
1019
+ ```ts
1020
+ observer
1021
+ .addObserverDetector('client-population-issue-detector', { issueTypes: [ 'cpulimitation' ], groupBy: 'browser' })
1022
+ .addObserverDetector('client-population-issue-detector', { issueTypes: [ 'cpulimitation' ], groupBy: 'operationSystem' });
366
1023
 
367
- The `observer-js` library provides two primary ways to associate custom information with its entities: `appData` and `attachments`. Understanding their distinct purposes is key for effective use.
1024
+ const [ byBrowser, byOs ] = observer.detectors.getAll('client-population-issue-detector');
1025
+
1026
+ observer.detectors.remove(byOs); // keeps the browser axis running
1027
+ ```
1028
+
1029
+ `Detectors` is a small collection: `instances` (a copy, in registration order), `listOfNames`,
1030
+ `size`, `get(name)`, `getAll(name)`, `has(name)`, `add(detector)`, `remove(detector)`,
1031
+ `removeByName(name)`, `clear()`, and it is iterable — `for (const detector of call.detectors)`.
1032
+ `instances` being a copy is deliberate: removing while iterating the live array would skip entries,
1033
+ and "drop the ones that look like X" is the most natural thing to want to write.
1034
+
1035
+ Two things worth knowing:
1036
+
1037
+ - **By name removes every instance under it**, not the first. A name can legitimately be registered
1038
+ more than once — `ClientPopulationIssueDetector` is meant to be added once per `groupBy` axis — and
1039
+ "remove whichever is first in the array" is not something a caller can predict from a name. Go via
1040
+ `detectors.getAll(name)` + `detectors.remove(instance)` when you mean one of them.
1041
+ - **`removeCallDetector` affects calls already open, by default.** Otherwise whether a detector runs
1042
+ would depend on when a call happened to join, which is not a state anyone can reason about. Pass
1043
+ `{ includeOpenCalls: false }` to change only what future calls are built with.
1044
+
1045
+ Every removal path calls the detector's `close()`, so it unsubscribes from the issue registry and
1046
+ drops any timers or bus listeners. A detector removed without closing would keep being fed matching
1047
+ issues for the life of the call — invisible, unbounded, and it would still look healthy if you
1048
+ inspected it.
1049
+
1050
+ Detectors are named by their kebab-case `NAME`, and the name types the config — an unknown name or a
1051
+ key that belongs to a different detector will not compile. Each detector owns its defaults in its own
1052
+ constructor, beside the doc explaining what each threshold means; there is no central table to keep
1053
+ in sync.
1054
+
1055
+ > **Why no defaults?** A detector nobody asked for is a detector nobody will act on. It costs time on
1056
+ > every tick and raises findings into a handler that was not written to expect them. Earlier versions
1057
+ > auto-created everything from a three-state config slot; the result was applications receiving
1058
+ > finding types they had never heard of.
1059
+
1060
+ Every issue-driven detector takes an explicit, non-empty `issueTypes` (or
1061
+ `publisherIssueTypes`/`receiverIssueTypes`). There is no "watch everything" option — see
1062
+ [the registry](#activeissuesregistry--issues-are-pushed-not-polled).
1063
+
1064
+ > **🔗 marks detectors that require a
1065
+ > [`RemoteTrackResolver`](#remote-track-resolution-mediasoup--sfu).** They reason about a published
1066
+ > track and its subscribers, so without the publisher↔subscriber links they see nothing and stay
1067
+ > **silent forever** — which looks exactly like "no problems found". Configure
1068
+ > `ObserverConfig.createRemoteTrackResolver`, and start the
1069
+ > [`remote-track-resolver` validator](#validators--one-shot-structural-checks) to prove it is wired.
1070
+
1071
+ ### Built-in detectors
1072
+
1073
+ They consume the verdicts `client-monitor-js` >= 4.6.0 already ships (raise + `<type>-resolved`) and
1074
+ add only the cross-participant conclusion. **None re-derives a per-endpoint verdict from raw
1075
+ counters** — that is the rule the whole design hangs on:
1076
+
1077
+ > **If a condition is detectable on the client, the client's issue is the source of truth.**
1078
+
1079
+ | Detector | 🔗 | Scope | Raises |
1080
+ |----------|:--:|-------|--------|
1081
+ | `CallConcurrentIssueDetector` | | call | `CONCURRENT_CLIENT_ISSUES`, `ISSUE_ONSET_BURST` |
1082
+ | `ObserverConcurrentIssueDetector` | | **observer** | `CROSS_CALL_CONCURRENT_ISSUES`, `CROSS_CALL_ISSUE_ONSET_BURST` |
1083
+ | `IssueFanOutDetector` | 🔗 | call | `PUBLISHED_TRACK_ISSUE_FAN_OUT`, `SINGLE_RECEIVER_ISSUE` |
1084
+ | `PublisherFaultCorroborationDetector` | 🔗 | call | `CORROBORATED_PUBLISHER_FAULT` |
1085
+ | `TrackDeliveryMismatchDetector` | 🔗 | call | `PUBLISHED_TRACK_NOT_DELIVERED`, `RECEIVER_TRACK_NOT_DELIVERED`, `PUBLISHER_TRACK_DRY` |
1086
+ | `UnconsumedTrackDetector` | 🔗 | call | `UNCONSUMED_PUBLISHED_TRACK` |
1087
+ | `ClientPopulationIssueDetector` | | **observer** | `CLIENT_POPULATION_ISSUE` |
1088
+ | `SfuCongestionDetector` | | **observer** | `sfu-congestion` |
1089
+ | `TurnServerHealthDetector` | | **observer** | `TURN_SERVER_DEGRADED` |
1090
+ | `TurnServerOutageDetector` | | **observer** | `TURN_SERVER_OUTAGE` |
1091
+
1092
+ What each adds that no endpoint can know:
1093
+
1094
+ - **`CallConcurrentIssueDetector`** — *who else in this meeting is in this state right now?* The
1095
+ difference between "one person's Wi-Fi" and "this room is broken".
1096
+ - **`ObserverConcurrentIssueDetector`** — *is our infrastructure in trouble?* A **separate class**,
1097
+ not the call one with a bigger denominator, because it is a different question with different
1098
+ gates. It requires the group to span at least `minAffectedCalls` **independent calls** (default
1099
+ `2`) and raises its own `CROSS_CALL_*` types. Without that gate, one thirty-person meeting where
1100
+ everyone is congested clears every client threshold and pages you for a single bad room the
1101
+ call-scoped detector already reported. Clients in different calls share no room, no publisher and
1102
+ no host — only the servers, which is what makes the finding conclusive. Note there is deliberately
1103
+ **no participant ratio** at this scope: six broken calls out of forty is a small share of all
1104
+ clients, and a ratio gate would hide exactly the event you want.
1105
+ - **`IssueFanOutDetector`** — *does this issue follow one published source, or one receiver?*
1106
+ - **`PublisherFaultCorroborationDetector`** — *do **both ends** of one track agree the source is at
1107
+ fault?* Fan-out sees one end and infers; this sees the publisher reporting `encoder-bottleneck`
1108
+ about its own send path *while* its subscribers report `freezed-video-track` about receiving it.
1109
+ Two independent parties, one conclusion, nothing left to deduce — hence the highest confidence in
1110
+ the library. Run both: fan-out is broader and catches the case where the publisher is fine and the
1111
+ SFU's forwarding is not.
1112
+ - **`ClientPopulationIssueDetector`** — *is this concentrated on one **kind of client**?* The one
1113
+ correlation here that is neither per-call nor per-server. Every other observer-scoped detector
1114
+ reasons "clients in unrelated calls share only the infrastructure, so it must be us" — right for
1115
+ network symptoms, **wrong for endpoint ones**. `cpulimitation` across six unrelated calls is not an
1116
+ SFU event; CPU is owned by the endpoint, so what those endpoints share is a browser version or a
1117
+ client release. Groups by `browser` / `engine` / `platform` / `operationSystem` / `location`, one
1118
+ axis per instance. The gate is **relative risk**, not share: "30% of Chrome 141 is unhappy" means
1119
+ nothing if 30% of everyone is, and a share-based rule simply indicts whichever browser is most
1120
+ popular. See [the `location` axis](#grouping-by-place-the-location-axis) for the geographic form.
1121
+ - **`SfuCongestionDetector`** — *is congestion spiking across the fleet right now?* Counts distinct
1122
+ clients reporting congestion in fixed wall-clock buckets and compares each bucket against a
1123
+ median+MAD baseline of the ones before it. Buckets rather than update ticks on purpose: the tick is
1124
+ unevenly spaced and shorter than a client's sampling period, so counting on it compares windows of
1125
+ different lengths and calls the difference a signal. Only add it when the observer's calls all come
1126
+ from the **same SFU** — the finding's meaning is "these clients share only that server".
1127
+ - **`TrackDeliveryMismatchDetector`** — *are the two ends of a track disagreeing?*
1128
+ - **`UnconsumedTrackDetector`** — *is anyone actually subscribed?* (reads the resolver's silence)
1129
+ - **`TurnServerHealthDetector`** — *does trouble cluster on one relay?*
1130
+ - **`TurnServerOutageDetector`** — covers the case the health detector structurally cannot. The
1131
+ health detector groups clients by the server relaying them and asks how many report issues — it
1132
+ needs clients *on* the server to ask. When a TURN server dies, allocation fails: existing sessions
1133
+ drop and new clients never obtain a relay candidate through it, so they are never attributed to it
1134
+ at all. Its population goes to zero and the health detector falls silent for the worst possible
1135
+ reason. **Degradation makes clients unhappy; an outage makes them disappear.** Absence is a
1136
+ dangerous signal, so the **control group** is the heart of the design: a call ending, everyone
1137
+ leaving at 6pm, and a fleet-wide network event all look identical to an outage. It refuses to blame
1138
+ a server unless clients *not* relayed through it are demonstrably still connected
1139
+ (`requireControlGroup`, on by default).
1140
+
1141
+ #### There is no ICE detector
1142
+
1143
+ ICE trouble is reported by `client-monitor-js` >= 4.6.0 as the keyed issues `ice-disconnected`,
1144
+ `ice-connection-failed`, `ice-transport-stalled` and `unstable-ice-path`, each with hysteresis and
1145
+ multi-signal confirmation behind it. An `IceDisruptionDetector` used to re-derive that server-side
1146
+ from raw state transitions; it has been removed, because the server sees less and guesses more. The
1147
+ client knows whether `disconnected` persisted or healed in 200 ms; the observer does not.
1148
+
1149
+ Correlating ICE trouble is now configuration, not a class:
1150
+
1151
+ ```ts
1152
+ observer.addObserverDetector('observer-concurrent-issue-detector', {
1153
+ issueTypes: [ 'ice-disconnected', 'ice-connection-failed', 'ice-transport-stalled' ],
1154
+ });
1155
+ ```
368
1156
 
369
- **`appData` (Application Data)**
1157
+ ### Grouping by place: the `location` axis
370
1158
 
371
- - **Purpose**: `appData` is designed to hold structured, typed, and relatively static metadata about an entity (`Observer`, `ObservedCall`, `ObservedClient`). This data is typically set at the time of entity creation and is directly accessible as a property of the entity instance.
372
- - **Typing**: It is strongly typed using generics (e.g., `Observer<MyObserverAppData>`). This provides type safety and autocompletion in TypeScript environments.
373
- - **Mutability**: While technically mutable (if the object assigned is mutable), it's generally intended for information that defines or describes the entity and doesn't change frequently during its lifecycle.
374
- - **Accessibility**: Directly accessible via `entity.appData`.
375
- - **Use Cases**:
376
- - Storing application-specific identifiers (e.g., `userId`, `roomId`, `meetingType`).
377
- - Configuration flags relevant to how your application interprets this entity.
378
- - Descriptive information (e.g., `clientDeviceType`, `callRegion`).
1159
+ If your clients report coordinates, `ClientPopulationIssueDetector` can group by **where they are**
1160
+ instead of what they run — which is the grouping network symptoms actually cluster by:
379
1161
 
380
- **`attachments` (Arbitrary Attachments)**
1162
+ ```ts
1163
+ observer.addObserverDetector('client-population-issue-detector', {
1164
+ issueTypes: [ 'congestion', 'ice-disconnected' ],
1165
+ groupBy: 'location',
1166
+ locationPrecision: 3, // geohash chars: 3 ~156 km, 4 ~39 km, 5 ~5 km
1167
+ resolveClientLocation: (client) => client.attachments?.geo as { latitude: number, longitude: number },
1168
+ });
1169
+ ```
381
1170
 
382
- - **Purpose**: `attachments` (if implemented as a `Map<string, unknown>` or similar mechanism on entities) are meant for associating arbitrary, often dynamic, or less structured data with an entity. This can be useful for temporary state, inter-plugin communication, or data that doesn't fit neatly into a predefined `appData` schema.
383
- - **Typing**: Typically less strictly typed (e.g., `unknown` or `any` values in a Map). Consumers of attachments need to perform their own type checks or assertions.
384
- - **Mutability**: Designed to be more dynamic. Attachments can be added, updated, or removed throughout the entity's lifecycle.
385
- - **Accessibility**: Accessed via methods like `entity.setAttachment(key, value)`, `entity.getAttachment(key)`, `entity.removeAttachment(key)`.
386
- - **Use Cases**:
387
- - Storing temporary state calculated by one part of your application to be read by another (e.g., a custom issue detector plugin might attach intermediate findings).
388
- - Caching results of expensive computations related to the entity.
389
- - Allowing different modules or plugins to associate their own private data with an observer entity without needing to modify its core `appData` type.
390
- - Storing large binary data or complex objects that are not part of the core descriptive metadata.
1171
+ **The client still owns "RTT jumped".** client-monitor's `CongestionDetector` compares each peer
1172
+ connection's RTT against its own EWMA baseline and requires a bandwidth-limitation corroboration
1173
+ before raising `congestion`. Absolute RTT is not comparable between clients — someone 200 ms away is
1174
+ *always* 200 ms away, so the only signal is deviation from that client's own baseline, which is
1175
+ exactly what the client measures. The observer's contribution is the part no endpoint can see: that
1176
+ many of the affected clients are **in the same place at the same time**.
1177
+
1178
+ Three things to know:
1179
+
1180
+ - **Cells, not radii.** The population is a geohash prefix. "Within N km" is a clustering problem —
1181
+ order-dependent, no stable group name, pairwise cost — and a detector needs the *same* group key on
1182
+ every tick for its cooldown and control group to mean anything. The cost is that a cell boundary can
1183
+ split two adjacent clients, which biases towards missing a finding rather than inventing one.
1184
+ - **Only the cell key is reported.** `payload.population` is the geohash; coordinates never enter the
1185
+ issue. These payloads get archived into [call summaries](#call-summaries), so that matters.
1186
+ - **Geography is confounded with your topology.** The control group is "everyone outside this cell",
1187
+ which cannot separate *"the path into this region degraded"* from *"the SFU serving this region
1188
+ degraded"*. If a region maps largely onto one deployment, both hypotheses fit the same evidence —
1189
+ so the finding concludes `infrastructure` and points you at `SfuCongestionDetector` /
1190
+ `TurnServerHealthDetector`, which answer whether clients elsewhere on the same server also
1191
+ degraded. It does not claim an attribution it cannot support.
1192
+
1193
+ Coordinates are not in `ClientSample`, so `resolveClientLocation` is required; without it the detector
1194
+ logs a warning at construction and finds nothing, rather than quietly reporting no findings forever.
1195
+
1196
+ ### Validators — one-shot structural checks
1197
+
1198
+ Every detector above answers *"is something wrong right now?"* and runs on every tick, because the
1199
+ answer legitimately changes. A **validator** answers *"is this deployment built correctly?"* — which
1200
+ only changes when you deploy. So it is not configured on and left running: you **start** one, it runs
1201
+ until it can decide, reports once, and the observer drops it.
1202
+
1203
+ ```ts
1204
+ observer.addValidator('simulcast-receivers', { minChecks: 5 });
1205
+
1206
+ observer.on('validation-ready', ({ validator, report }) => {
1207
+ if (!report.ready) return;
1208
+ if (report.verdict === 'layer-decided-lowest-common-denominator') page(validator, report);
1209
+ });
391
1210
 
392
- **When to Use Which:**
1211
+ onDeploy(() => observer.addValidator('simulcast-receivers')); // check again
1212
+ ```
393
1213
 
394
- - Use **`appData`** for:
395
- - Core, descriptive metadata that is known at creation time or changes infrequently.
396
- - Data that benefits from strong typing and is integral to your application's understanding of the entity.
397
- - Use **`attachments`** for:
398
- - Dynamic, temporary, or loosely structured data.
399
- - Data added by different, potentially independent, parts of your system or plugins.
400
- - Information that doesn't need to be part of the primary, typed `appData` schema.
1214
+ `observer.validators` is the set currently running — normally empty, since each removes itself on
1215
+ finishing. There is no revalidation timer: a deploy, not elapsed time, is what makes a structural
1216
+ verdict stale, so re-checking means starting another.
401
1217
 
402
- If `attachments` are not yet a formal feature, this section can serve as a design consideration or be adapted if you introduce such a mechanism. If `attachments` are already present, ensure the description matches their actual implementation.
1218
+ **Cancelling.** A check that has not decided can be stopped, by name or by instance:
403
1219
 
404
- #### 4.5. Remote Track Resolution
1220
+ ```ts
1221
+ observer.cancelValidator('simulcast-receivers', 'sfu redeployed');
405
1222
 
406
- For SFU scenarios, especially with MediaSoup:
407
- `ObservedCallSettings.remoteTrackResolvePolicy: 'mediasoup-sfu'`
1223
+ // or one specific instance — `observer.validators` holds what is running
1224
+ for (const validator of observer.validators) observer.cancelValidator(validator, 'shutting down');
1225
+ ```
408
1226
 
409
- ### 5. Examples
1227
+ Cancelling is **not** silent discarding. The validator finishes `inconclusive` with your reason,
1228
+ emits `validation-ready` like any other completion, and removes itself. That matters twice over:
1229
+ anything waiting on the verdict would otherwise wait forever, and *"we stopped asking"* is a
1230
+ materially different outcome from *"we asked and learned nothing"* — which is exactly what an
1231
+ `inconclusive` carrying a reason records. Pass a real reason; the default tells the reader nothing
1232
+ they could not already infer. `observer.close()` cancels whatever is still running with
1233
+ `'observer closed'`.
1234
+
1235
+ | Validator | `addValidator` name | Question | Also raises |
1236
+ |-----------|---------------------|----------|-------------|
1237
+ | `SimulcastReceiverValidator` 🔗 | `simulcast-receivers` | Does the SFU pick layers per receiver, or drag the publisher down to the worst one? | `WORST_RECEIVER_CONTAGION` |
1238
+ | `RemoteTrackResolverValidator` | `remote-track-resolver` | Is the resolver actually linking anything? | `REMOTE_TRACK_LINKS_UNRESOLVED` |
1239
+ | `CodecConsistencyValidator` | `codec-consistency` | Is everyone on the same codec — and is it the one you think you negotiated? | `CODEC_INCONSISTENCY` |
1240
+
1241
+ **`SimulcastReceiverValidator`** — simulcast (or SVC) exists so one slow
1242
+ participant doesn't set everyone's quality: with several encodings the server hands the struggling
1243
+ receiver a lower layer and leaves the rest alone. Without it — or with a server that relays RTCP end
1244
+ to end, so the publisher's bandwidth estimate collapses to the slowest receiver — the only way to
1245
+ serve them is to make the *source* send less. Both causes look identical from outside; what the check
1246
+ establishes is whether per-receiver adaptation happens at all.
1247
+
1248
+ | `verdict` | meaning |
1249
+ |-----------|---------|
1250
+ | `layer-decided-per-receiver` | verified — a receiver fell far behind and the publisher carried on |
1251
+ | `layer-decided-lowest-common-denominator` | the publisher followed its worst receiver; everyone gets the slowest participant's quality |
1252
+ | `inconclusive` | cancelled, or the observer closed, before it could decide |
1253
+
1254
+ **Not finishing is not a pass.** The check only runs when a publisher has 3+ receivers and one is at
1255
+ most half the median; plenty of healthy deployments never present that. A validator that never sees it
1256
+ simply keeps running and never reports — it does not quietly succeed. `report.checks` counts the times
1257
+ the check genuinely ran, so an `inconclusive` with `checks: 0` says plainly that nothing was verified.
1258
+
1259
+ **`RemoteTrackResolverValidator`** exists because of a specific, nasty failure mode. Four things here
1260
+ are built on publisher↔subscriber links — `IssueFanOutDetector`,
1261
+ `PublisherFaultCorroborationDetector`, `TrackDeliveryMismatchDetector`, `UnconsumedTrackDetector`
1262
+ (and `SimulcastReceiverValidator`) — and every one of them correctly does *nothing* when the links
1263
+ are missing rather than guessing. So a resolver wired to the wrong id field leaves all of them
1264
+ permanently silent, and **silence is what a healthy deployment looks like too**: you would conclude
1265
+ your calls were clean when in fact nothing was ever examined. Verdicts: `links-resolved` /
1266
+ `no-links-resolved` / `inconclusive`. Run it at start-up and after changing the resolver or the SFU's
1267
+ id scheme.
1268
+
1269
+ **`CodecConsistencyValidator`** answers two things at once. A **split** — participants of one call on
1270
+ different codecs — is a real fault with a confusing symptom: an SFU that forwards without transcoding
1271
+ cannot serve them all, so some pairs see each other and some do not, with no error anywhere. Only
1272
+ something holding every participant at once can see it. The quieter half is the silent fallback: a
1273
+ deployment configured for VP9 or AV1 drops to VP8 whenever one endpoint cannot negotiate the
1274
+ preference, the call keeps working at a higher bitrate than budgeted, and the team believes it
1275
+ shipped AV1 months ago. Give it `expected` and it says so. Verdicts: `codec-consistent` /
1276
+ `codec-split` / `unexpected-codec` / `inconclusive`.
1277
+
1278
+ ```ts
1279
+ observer.addValidator('remote-track-resolver');
1280
+ observer.addValidator('codec-consistency', { expected: { video: 'video/VP9', audio: 'audio/opus' } });
1281
+ ```
410
1282
 
411
- #### 5.1. Basic Observer Setup & Sample Ingestion
1283
+ #### `CallIssue` vs `ObserverIssue`
412
1284
 
413
- ```typescript
414
- // filepath: /path/to/your/app.ts
415
- import { Observer, ObserverConfig } from '@observertc/observer-js'; // Adjust path
416
- import { ClientSample } from '@observertc/schemas'; // Adjust path if using official schemas
1285
+ Server-raised findings come in two kinds, distinguished by the scope that raised them:
417
1286
 
418
- const observerConfig: ObserverConfig = {
419
- updatePolicy: 'update-on-interval',
420
- updateIntervalInMs: 5000,
421
- defaultCallUpdatePolicy: 'update-on-any-client-updated',
422
- };
423
- const observer = new Observer(observerConfig);
424
-
425
- observer.on('newcall', (call) => {
426
- console.log(`[Observer] New call: ${call.callId}`);
427
- call.on('update', () => {
428
- console.log(`[Call: ${call.callId}] Updated. Clients: ${call.numberOfClients}, Score: ${call.score?.toFixed(1)}`);
429
- });
430
- call.on('newclient', (client) => {
431
- console.log(`[Call: ${call.callId}] New client: ${client.clientId}`);
432
- client.on('update', () => {
433
- // console.log(`[Client: ${client.clientId}] Updated. Score: ${client.score?.toFixed(1)}`);
434
- });
435
- client.on('issue', (issue) => {
436
- console.warn(`[Client: ${client.clientId}] Issue: ${issue.type} - ${issue.severity} - ${issue.description}`);
437
- });
438
- });
439
- });
440
-
441
- // Function to transform your app's WebRTC stats to ClientSample
442
- function mapStatsToClientSample(appStats: any, callId: string, clientId: string): ClientSample {
443
- // Detailed mapping logic here based on ClientSample.ts schema
444
- // from github.com/observertc/schemas
445
- return {
446
- callId,
447
- clientId,
448
- timestamp: Date.now(),
449
- // ... map all relevant stats fields ...
450
- } as ClientSample; // Ensure all required fields are present
1287
+ | | raised by | delivered as | `scope` |
1288
+ |---|---|---|---|
1289
+ | `CallIssue` | `observedCall.addIssue(...)` | `call-issue` | `'call'` |
1290
+ | `ObserverIssue` | `observer.addIssue(...)` | `observer-issue` | `'observer'` |
1291
+
1292
+ Both share `IssueBase` — `type`, `timestamp`, `conclusion?`, `payload?` — and `Issue` is the union,
1293
+ discriminated on `scope`.
1294
+
1295
+ ```ts
1296
+ observer.on('call-issue', ({ observedCall, issue }) => {
1297
+ issue.scope; // 'call'
1298
+ observedCall.callId; // the call — NOT repeated in the payload
1299
+ issue.conclusion?.faultDomain;
1300
+ issue.payload; // evidence only
1301
+ });
1302
+
1303
+ observer.on('observer-issue', ({ issue }) => {
1304
+ issue.scope; // 'observer'
1305
+ });
1306
+ ```
1307
+
1308
+ `scope` is stamped by `addIssue` rather than asked of the detector: it is a fact about *where the
1309
+ finding was raised*, which the entity knows and a detector should not have to restate. Having it on
1310
+ the issue — not merely implied by which event fired — keeps a finding self-describing once it leaves
1311
+ the bus, into a shared handler, a log line or a queue.
1312
+
1313
+ **The payload is evidence and nothing else.** It no longer repeats `type`, `scope`, or the `callId`
1314
+ already carried by the event, and `conclusion` was lifted out of it to a first-class field. A payload
1315
+ that restates its own envelope invites the two to disagree — and they did, because nothing kept them
1316
+ in step. `payload` is always an object (the `string` form is gone, along with `issuePayloadOf`); use
1317
+ `issuePayloadAsString(issue)` at a boundary that genuinely needs text.
1318
+
1319
+ #### Conclusions
1320
+
1321
+ Every issue-driven finding carries a `conclusion` — the interpretation step, so the person reading
1322
+ the alert doesn't have to perform it. It sits **beside** the evidence, not inside it:
1323
+
1324
+ ```jsonc
1325
+ {
1326
+ "type": "CROSS_CALL_ISSUE_ONSET_BURST",
1327
+ "scope": "observer",
1328
+ "timestamp": 1739812345678,
1329
+ "conclusion": {
1330
+ "faultDomain": "infrastructure",
1331
+ "summary": "network congestion is open across independent calls at the same time — 6 of 40 calls (11/300 clients)",
1332
+ "recommendation": "check SFU egress bandwidth and host network saturation before looking at any single participant",
1333
+ "confidence": 0.85
1334
+ },
1335
+ "payload": {
1336
+ "issueType": "congestion",
1337
+ "calls": 40, "affectedCalls": 6,
1338
+ "perCall": [ { "callId": "…", "affectedClients": 4, "totalClients": 9 } ]
1339
+ }
451
1340
  }
1341
+ ```
452
1342
 
453
- // Example: Receiving stats and processing
454
- const rawStatsFromClient = {
455
- /* ... your client's getStats() output ... */
456
- };
457
- const callId = 'meeting-alpha-123';
458
- const clientId = 'user-xyz-789';
459
- const sample = mapStatsToClientSample(rawStatsFromClient, callId, clientId);
460
- observer.accept(sample);
1343
+ `faultDomain` is one of `infrastructure`, `call`, `published-track`, `endpoint`, `client-population`
1344
+ or `unknown`, and it comes from the **spread**, not the issue type — congestion in one call is a
1345
+ meeting problem, congestion in six calls is a server problem, and the client reported the identical
1346
+ symptom in both.
1347
+
1348
+ One case is worth knowing about because it inverts the usual reading: **`cpu-limitation` spread
1349
+ across many independent calls concludes `client-population`, not `infrastructure`.** Endpoint CPU is
1350
+ owned by the endpoint, so breadth there points at what those endpoints share — a recent client
1351
+ release, a browser version, shared VDI hardware — and paging the SFU on-call would be wrong. The
1352
+ conclusion table encodes that so nobody has to rediscover it during an incident.
1353
+
1354
+ Unknown issue types (your own custom client detectors) still produce a structurally valid conclusion
1355
+ from the spread alone; they just get generic wording.
1356
+
1357
+ Two functions are exported, one per scope: `concludeCallIssue()` and `concludeObserverIssue()`. They
1358
+ are separate because a detector already knows its scope, and a single generic function forced every
1359
+ caller to pass the other scope's fields as placeholders — call-scoped detectors passing
1360
+ `affectedCalls: 1, totalCalls: 1` forever, observer-scoped ones passing a participant ratio that was
1361
+ deliberately never read. Placeholders like that invite being read as if they meant something.
1362
+
1363
+ #### Cost
1364
+
1365
+ Detectors run inside `call.update()`, on your event loop, so their cost matters. Two things keep it
1366
+ off the participant axis:
1367
+
1368
+ - **Issues are pushed, not polled.** A detector holds only what the registry handed it, so an
1369
+ `update()` that finds `size === 0` — the overwhelmingly common case — costs one comparison,
1370
+ whatever the participant count. Nothing iterates clients looking for trouble.
1371
+ - **So are unconsumed tracks.** `observedCall.unconsumedOutboundTracks` is maintained by the resolver
1372
+ as tracks gain and lose subscribers, so `UnconsumedTrackDetector` reads a set that is normally
1373
+ empty instead of walking every published track (529 µs → 65 µs per tick at 1 200 tracks).
1374
+ - **Track lookups start from the affected minority.** A detector resolving an issue to its published
1375
+ track searches the *reporting client's* peer connections (typically one or two), not the call.
1376
+
1377
+ At 20 calls × 12 participants (2 640 subscriptions) the whole detector pass costs ~1.3 ms per tick.
1378
+ `yarn bench` prints a per-detector breakdown for your own shape.
1379
+
1380
+ #### Worked examples
1381
+
1382
+ [`examples/detectors.ts`](./examples/detectors.ts) (`yarn example:detectors`) runs one scenario per
1383
+ detector — the question it answers, its full config, the synthetic traffic that makes it fire, and
1384
+ the finding with its conclusion. It asserts every expected finding is produced, so it doubles as a
1385
+ smoke test. [`examples/sfu-observer.ts`](./examples/sfu-observer.ts) (`yarn example`) is the end-to-end
1386
+ tour instead: ingest → correlate → react, with the mediasoup wiring alongside.
1387
+
1388
+ #### `TrackDeliveryMismatchDetector` — resolving an ambiguous symptom
1389
+
1390
+ A dry track ("no bytes are arriving") is the clearest symptom there is and, on its own, completely
1391
+ ambiguous. A receiver seeing silence cannot distinguish *the camera was switched off* from *the SFU
1392
+ stopped forwarding* from *my own consumer wedged* — all three look identical from the browser.
1393
+
1394
+ Joining the two ends of the published track resolves it:
1395
+
1396
+ | publisher | subscribers | verdict |
1397
+ |---|---|---|
1398
+ | sending | **all** dry | `PUBLISHED_TRACK_NOT_DELIVERED` — the forwarding path |
1399
+ | sending | **some** dry | `RECEIVER_TRACK_NOT_DELIVERED` — those consumers (in mediasoup: recreate them) |
1400
+ | dry | any dry | `PUBLISHER_TRACK_DRY` — the source stopped; **not** an SFU fault |
1401
+
1402
+ The publisher side is judged from both available signals: its own `dry-outbound-track` issue when the
1403
+ client reports one, and the observed outbound RTP (`deltaPacketsSent`) as fallback and corroboration.
1404
+ That combination is what makes the first row trustworthy — the server can state that packets
1405
+ demonstrably left the publisher during the same interval in which every receiver got nothing.
1406
+
1407
+ This is the "SFU forwarding mismatch" check, and it needs **no mediasoup instrumentation at all** —
1408
+ the clients' own dry-track verdicts plus the resolver links are sufficient.
1409
+
1410
+ #### `UnconsumedTrackDetector` — reading the resolver's silence
1411
+
1412
+ The one detector where the *absence* of links is the signal: a track still pushing packets whose
1413
+ `remoteInboundTracks` set is empty, i.e. uplink and SFU ingress spent on media nobody receives
1414
+ (everyone has the publisher hidden, a simulcast layer no viewer selects, or an app that forgot to
1415
+ stop a track). It waits `minUnconsumedDurationInMs` first, since a gap between publishing and the
1416
+ first subscription is normal at join time.
1417
+
1418
+ Note the trap this one has to guard against, and why it checks `call.remoteTrackResolver` at runtime
1419
+ rather than trusting the flag alone: **"no subscribers" and "no resolver configured" produce the
1420
+ identical observation.** Without a resolver it would report every published track in the call as
1421
+ unconsumed.
1422
+
1423
+ ## Call summaries
1424
+
1425
+ Everything else in this library is about *now*. Detectors answer "is something wrong right now",
1426
+ validators answer a structural question once, and both read state the call throws away when it ends.
1427
+ A **call summary** is the one thing that outlives the call: who was in it, what was raised against
1428
+ it, how it scored — the questions asked *after* the meeting, by support, by billing, by whoever is
1429
+ writing the incident note.
1430
+
1431
+ It is configured on the observer, at construction:
1432
+
1433
+ ```ts
1434
+ const observer = new Observer({
1435
+ callSummary: {
1436
+ include: [ 'clients', 'issues', 'turnServers', 'scores' ],
1437
+ },
1438
+ });
1439
+
1440
+ observer.on('call-summary', ({ summary }) => archive(summary));
1441
+ ```
461
1442
 
462
- // Later, on application shutdown:
463
- // observer.close();
1443
+ Omit `callSummary`, or set it to `null`, and there are no summaries and **not one extra bus
1444
+ subscription**. Pass an object — `{}` is valid — and every call this observer creates carries one.
1445
+
1446
+ > **Why construction-time, when detectors are added per call?** A summary is a record of what
1447
+ > happened, and a record you can switch on halfway through is a record with a hole in it. Calls that
1448
+ > started before the switch would carry different sections from calls that started after, with
1449
+ > nothing on either to say which. One shape for every call, or none.
1450
+
1451
+ ### Sections are opt-in, and absence means "not collected"
1452
+
1453
+ `include` picks from four built-ins, and **the default is `[]`** — none of them:
1454
+
1455
+ | Section | Contains |
1456
+ |---------|----------|
1457
+ | `clients` | `clientIds` (join order), `peak`, `joined`, `left`. Identifiers and counts only |
1458
+ | `issues` | `CallIssue[]`, in the order raised, capped by `maxIssues` |
1459
+ | `turnServers` | `serverUrls` that carried media, and `clientsRelayed` |
1460
+ | `scores` | `min` / `max` / `median` of the call score, and `samples` |
1461
+
1462
+ **A missing section means it was never collected — never "nothing happened".** Reading
1463
+ `summary.issues === undefined` as "this call was clean" is the one misreading this type invites, so
1464
+ there is no default-empty section to make it easy. This is the same rule as `inconclusive` on a
1465
+ validator: silence is not success.
1466
+
1467
+ The `clients` section is deliberately identifiers and counts. Anything *about* a client — browser,
1468
+ platform, region — is already on `observedClient` while the call is live, and belongs in
1469
+ `attachments` via an enricher if you want it kept; see below.
1470
+
1471
+ ### Enrichers: fold in anything, from any call-scoped event
1472
+
1473
+ ```ts
1474
+ new Observer({
1475
+ callSummary: {
1476
+ include: [ 'issues' ],
1477
+ enrich: {
1478
+ 'client-joined': (summary, { observedClient }) => {
1479
+ // serialisable facts only — the region string, never the live object it came from
1480
+ ((summary.attachments.regions ??= []) as string[]).push(String(observedClient.appData.region));
1481
+ },
1482
+ },
1483
+ },
1484
+ });
464
1485
  ```
465
1486
 
466
- #### 5.2. Manual Call and Client Creation
1487
+ Each enricher is typed against its own event's payload. Only **call-scoped** events are accepted —
1488
+ the ones carrying an `observedCall`. An enricher on `observer-issue` or `validation-ready` will not
1489
+ compile, because there is no single call to attribute a fleet-wide fact to, and quietly writing it
1490
+ into every open summary would be worse than a type error.
1491
+
1492
+ The library never writes to `summary.attachments`, so nothing you put there can collide with a
1493
+ section added in a future version.
1494
+
1495
+ > **Why `attachments` and not `appData`.** `appData` is live working state hung off an entity for
1496
+ > that entity's lifetime, and it may hold things that cannot be serialised — a mediasoup router, a
1497
+ > socket. A summary is the opposite: it outlives the call so it can be **shipped**, and it reaches
1498
+ > you on `call-summary` while the call it describes is being torn down, so an unserialisable value
1499
+ > in it points at something already gone. Same contract as `attachments` on a `ClientSample`: read
1500
+ > the live object off `observedCall` / `observedClient` in the enricher, attach what serialises —
1501
+ > the router's `id`, not the router. An enricher that throws is logged and skipped — a summary is a
1502
+ side-channel, and nothing about a call should break because a field could not be recorded.
1503
+
1504
+ ### Caps announce what they dropped
1505
+
1506
+ `maxIssues` (default `500`) and `maxClientIds` (default `10_000`) bound the two unbounded lists.
1507
+ When either bites, `summary.truncated` appears with the shortfall — present **only** when something
1508
+ was actually dropped. That is what makes dropping safe: the true count is recoverable as
1509
+ `issues.length + (truncated?.issues ?? 0)`. A silently truncated summary is worse than no summary,
1510
+ because someone will count `issues.length` and report it as the issue count.
467
1511
 
468
- ```typescript
469
- // ... observer setup ...
1512
+ `issues` is the plain array, with no derived tallies alongside it. A count is `issues.length` and a
1513
+ per-type count is one `filter` — both cheaper at the call site than kept correct here.
470
1514
 
471
- const call = observer.createObservedCall({
472
- callId: 'scheduled-webinar-456',
473
- updatePolicy: 'update-on-interval',
474
- updateIntervalInMs: 10000,
1515
+ ### Reading it
1516
+
1517
+ `observedCall.summary` is live: read it at any point during the call. It is also delivered once on
1518
+ `call-summary`, emitted **inside** `close()` while the call is still in `observer.observedCalls` —
1519
+ after that the call is gone and there is nothing left to ask. `observer.close()` closes its calls
1520
+ first and its collector afterwards, so every summary still makes it out.
1521
+
1522
+ Cost is **one bus listener per subscribed event type, for the whole observer** — not one per call. A
1523
+ per-call design would be quadratic in concurrent calls: at 500 calls and eight events, 4 000
1524
+ listeners each doing 500 no-op invocations per event. Percentiles are computed once, at close.
1525
+
1526
+ ## Remote track resolution (mediasoup / SFU)
1527
+
1528
+ In an SFU, one participant's **outbound** track is delivered to other participants as **inbound**
1529
+ tracks (one **publisher** → many **subscribers**). Correlation is **opt-in** per observer: set
1530
+ `ObserverConfig.createRemoteTrackResolver`, a factory invoked when each call is created that returns the
1531
+ call's `RemoteTrackResolver` (or `undefined` for none).
1532
+
1533
+ `RemoteTrackResolver` is a generic, strategy-driven class. It subscribes to the bus (filtered to
1534
+ its call) and links tracks by **publisher id** — the link key — maintaining the links directly on
1535
+ the tracks: `inboundTrack.remoteOutboundTrack` and `outboundTrack.remoteInboundTracks: Set`.
1536
+
1537
+ ```ts
1538
+ import { Observer, createDefaultMediasoupRemoteTrackResolverFactory } from '@observertc/observer-js';
1539
+
1540
+ const observer = new Observer({
1541
+ createRemoteTrackResolver: createDefaultMediasoupRemoteTrackResolverFactory(),
475
1542
  });
476
1543
 
477
- const client1 = call.createObservedClient({ clientId: 'presenter-01' });
478
- // Samples for 'presenter-01' in call 'scheduled-webinar-456' will update this client.
1544
+ // later, given tracks (links are kept up to date as tracks come and go):
1545
+ const source = inboundTrack.remoteOutboundTrack; // the publishing ObservedOutboundTrack
1546
+ const receivers = [ ...outboundTrack.remoteInboundTracks ]; // the subscribing ObservedInboundTrack[]
1547
+ ```
1548
+
1549
+ Two built-in factories ship: `createDefaultMediasoupRemoteTrackResolverFactory()` (publisher =
1550
+ `attachments.producerId`, subscriber = `attachments.consumerId`) and
1551
+ `createP2pRemoteTrackResolverFactory()` (matches by RTP **SSRC**, preserved end-to-end in p2p).
1552
+
1553
+ For any other topology, build a `RemoteTrackResolver` with your own key resolvers — the publisher
1554
+ id is just whatever links a subscribed track to the published one:
1555
+
1556
+ ```ts
1557
+ import { Observer, RemoteTrackResolver } from '@observertc/observer-js';
1558
+
1559
+ const observer = new Observer({
1560
+ createRemoteTrackResolver: (observedCall) => new RemoteTrackResolver(observedCall, {
1561
+ resolveOutboundTrackPublisherId: (out) => out.attachments?.mediaId as string | undefined,
1562
+ resolveInboundTrackPublisherId: (inb) => inb.attachments?.mediaId as string | undefined,
1563
+ resolveInboundTrackSubscriberId: (inb) => inb.attachments?.subId as string | undefined, // optional
1564
+ }),
1565
+ });
479
1566
  ```
480
1567
 
481
- #### 5.3. Using Event Monitors for Contextual Logging
1568
+ For the mediasoup factory, the application puts `producerId` / `consumerId` (and optionally
1569
+ `direction`, `label`) into the track `attachments`.
1570
+
1571
+ ### Resolution stands alone — you do not need router observation
1572
+
1573
+ Track resolution and [mediasoup router observation](#mediasoup-router-observation) are two
1574
+ **independent** opt-ins that happen to share a vendor name. Nothing in `RemoteTrackResolver` reads
1575
+ `ObservedMediasoupRouter`, and nothing in `ObservedMediasoupRouter` touches calls, clients or tracks.
1576
+
1577
+ So if your application already builds its own per-router report and you only want the detectors that
1578
+ need publisher↔subscriber links, set `createRemoteTrackResolver` and simply never call
1579
+ `observer.createObservedMediasoupRouter(...)`. No `MediasoupRouterSample` is created, nothing
1580
+ accumulates, and these keep working:
1581
+
1582
+ `UnconsumedTrackDetector`, `PublisherFaultCorroborationDetector`, `TrackDeliveryMismatchDetector`,
1583
+ `IssueFanOutDetector`, `RemoteTrackResolverValidator`, `SimulcastReceiverValidator`.
1584
+
1585
+ ### Backing a strategy with your own mapping
1586
+
1587
+ The three resolvers are plain functions returning a link key, so they can read a table your own
1588
+ report already maintains rather than something the client attached. The only requirement is a key
1589
+ **visible on both sides**. SSRC is the useful one, because it needs no client cooperation at all —
1590
+ mediasoup knows each consumer's `rtpParameters.encodings[].ssrc` server-side, and the subscriber's
1591
+ inbound RTP reports the same value:
1592
+
1593
+ ```ts
1594
+ // your own table, filled where you already create consumers
1595
+ const ssrcToProducerId = new Map<number, string>();
1596
+
1597
+ const observer = new Observer({
1598
+ createRemoteTrackResolver: (observedCall) => new RemoteTrackResolver(observedCall, {
1599
+ resolveOutboundTrackPublisherId: (out) => out.attachments?.producerId as string | undefined,
1600
+ resolveInboundTrackPublisherId: (inb) => {
1601
+ const ssrc = inb.getInboundRtp()?.ssrc;
1602
+
1603
+ return ssrc === undefined ? undefined : ssrcToProducerId.get(ssrc);
1604
+ },
1605
+ }),
1606
+ });
1607
+ ```
1608
+
1609
+ > **A key that arrives late still links.** A track announces itself once, but its `attachments` are
1610
+ > replaced on every sample and a table like the one above is inherently racy against sample arrival.
1611
+ > Tracks whose key does not resolve at first sight are held and retried on their own
1612
+ > `*-track-updated`, i.e. exactly when new stats arrive for them — so a key that appears on the second
1613
+ > sample links then, rather than being lost for the track's lifetime. `resolver.pendingTrackCounts`
1614
+ > reports how many are still waiting; in a healthy setup it is `{ inbound: 0, outbound: 0 }`.
1615
+
1616
+ If a strategy resolves *nothing* the failure is quiet — "no subscribers" and "no links resolved" look
1617
+ identical from the outside. That is what
1618
+ [`RemoteTrackResolverValidator`](#validators--one-shot-structural-checks) is for; run it once in
1619
+ staging after wiring up a custom strategy.
1620
+
1621
+ ---
1622
+
1623
+ ## Mediasoup router observation
1624
+
1625
+ Everything above is built from the **client-reported** `ClientSample`. When you run a
1626
+ [mediasoup](https://mediasoup.org) SFU you also have the **server's own** ground truth — its
1627
+ routers, transports, producers, consumers and data channels, with exact lifetimes and state
1628
+ transitions. `ObservedMediasoupRouter` captures that server-side view into a
1629
+ **`MediasoupRouterSample`**, completely independent of the client sample pipeline.
1630
+
1631
+ ### The concept
1632
+
1633
+ You hand the observer a live mediasoup `Router`; it attaches to mediasoup's own `observer` API and,
1634
+ from then on, **passively tracks** the router's topology and lifecycle — with no polling and no
1635
+ changes to your media code:
1636
+
1637
+ - new transports (`webrtc` / `plain` / `pipe` / `direct`), their selected `tuple`, ICE/DTLS/SCTP
1638
+ state transitions and `connectedAt`;
1639
+ - producers (codec, SSRCs/RIDs, `pause`/`resume`) and consumers (`pause`/`resume`,
1640
+ `producerPaused`/`producerResumed`);
1641
+ - data producers and data consumers;
1642
+ - `createdAt` / `closedAt` for every entity above.
1643
+
1644
+ It keeps all of this **in memory**, in a single `MediasoupRouterSample` exposed as
1645
+ `observedRouter.sample` — see [`src/schema/MediasoupRouter.ts`](./src/schema/MediasoupRouter.ts). The
1646
+ sample **accumulates for the life of the router**: closed transports/producers/consumers are kept
1647
+ (with their `closedAt` set), not removed. Read it whenever you like — it's a plain object you own.
1648
+
1649
+ ### Memory & large meetings
1650
+
1651
+ This is intentionally the **simplest** approach — everything lives in memory and nothing is sampled
1652
+ or evicted for you. That's fine for typical rooms, but be aware of the cost at scale:
1653
+
1654
+ - **Consumers grow as O(N²)** on a single flat router: with `N` participants each producing audio +
1655
+ video and consuming everyone else, the sample holds roughly `2·N·(N−1)` consumer records (≈ 19,800
1656
+ for `N` = 100).
1657
+ - The sample is **cumulative** — closed entities and their `history` are retained — so it also grows
1658
+ with call duration and churn (renegotiation, simulcast layer changes, rejoins).
1659
+
1660
+ A 100-participant flat router can therefore reach tens of MB and keep growing. There is **no built-in
1661
+ sink, snapshotting, or eviction** — by design. **If you run large meetings, do your own sampling:**
1662
+ on your own cadence read `observedRouter.sample` (snapshot/serialize/persist what you need), drop what
1663
+ you don't, and close routers you no longer track. (mediasoup also typically shards routers across
1664
+ workers/cores, which keeps any one router small.)
1665
+
1666
+ ### Extending the sample, and building your own report
1667
+
1668
+ The sample is yours to annotate. Every entity — the router, each transport, producer, consumer, data
1669
+ producer and data consumer — has an `attachments?: Record<string, unknown>` slot, and there are three
1670
+ ways to fill it, from most declarative to most ad-hoc.
1671
+
1672
+ **1. `enrich` — mirror mediasoup's own `appData`.** The common case: your application already keeps
1673
+ `participantId`, `purpose` and similar on the mediasoup objects, and you want them on the sample.
1674
+ Runs once per entity at creation, before the corresponding event:
1675
+
1676
+ ```ts
1677
+ observer.createObservedMediasoupRouter({
1678
+ router,
1679
+ enrich: {
1680
+ producer: (producer) => ({ participantId: producer.appData.participantId, purpose: producer.appData.purpose }),
1681
+ consumer: (consumer) => ({ subscriberId: consumer.appData.subscriberId }),
1682
+ transport: (transport) => ({ role: transport.appData.role }),
1683
+ },
1684
+ });
1685
+ ```
1686
+
1687
+ A throwing enricher is caught and logged — it can't take the router's bookkeeping down with it.
1688
+
1689
+ **2. Lifecycle events — enrich on the fly.** Each entity announces itself as
1690
+ `<entity>-sample-added` and `<entity>-sample-closed`, carrying **the live sample object** (not a
1691
+ copy) plus the mediasoup object it came from. Mutating it in the handler is the intended pattern:
1692
+
1693
+ ```ts
1694
+ observedRouter.on('producer-sample-added', ({ sample, producer, transport }) => {
1695
+ sample.attachments = { ...sample.attachments, participantId: lookup(producer.id) };
1696
+ });
1697
+
1698
+ observedRouter.on('producer-sample-closed', ({ sample }) => {
1699
+ archive(sample); // its `closedAt` is set
1700
+ });
1701
+ ```
1702
+
1703
+ Events: `transport-sample-added` / `-closed`, `producer-sample-added` / `-closed`,
1704
+ `consumer-sample-added` / `-closed`, `data-producer-sample-added` / `-closed`,
1705
+ `data-consumer-sample-added` / `-closed`.
1706
+
1707
+ **3. `attachTo(id, attachments)` — annotate later, from anywhere.** When the knowledge arrives after
1708
+ the entity did (a signalling message, a database lookup that resolved):
1709
+
1710
+ ```ts
1711
+ observedRouter.attachTo(producerId, { participantId, joinedFrom: 'mobile' }); // merges
1712
+ ```
1713
+
1714
+ Ids are unique across mediasoup entity kinds, so one method covers all of them. It returns `false`
1715
+ for an unknown id rather than failing quietly — which matters when application events race the
1716
+ mediasoup ones. For direct access there are typed accessors: `getTransportSample(id)`,
1717
+ `getProducerSample(id)`, `getConsumerSample(id)`, `getDataProducerSample(id)`,
1718
+ `getDataConsumerSample(id)`. They index the *same* objects the arrays hold, so a lookup is O(1)
1719
+ instead of a `sample.producers.find(...)` scan.
1720
+
1721
+ #### Building your own report
1722
+
1723
+ `observedRouter.sample` is live — arrays grow and `history` entries are appended as the router runs,
1724
+ so a report built directly on it keeps changing after you think you're done. Use **`snapshot()`** for
1725
+ a detached deep copy:
1726
+
1727
+ ```ts
1728
+ const report = {
1729
+ ...observedRouter.snapshot(), // never moves again
1730
+ generatedAt: Date.now(),
1731
+ region: process.env.REGION,
1732
+ };
1733
+ ```
482
1734
 
483
- ```typescript
484
- const call = observer.getObservedCall('meeting-alpha-123');
485
- if (call) {
486
- const callMonitor = call.createEventMonitor({ callId: call.callId, started: new Date() });
487
- callMonitor.on('client-joined', (client, context) => {
488
- console.log(`EVENT_MONITOR (${context.callId}): Client ${client.clientId} joined at ${new Date()}`);
489
- });
490
- callMonitor.on('issue-detected', (client, issue, context) => {
491
- console.error(`EVENT_MONITOR (${context.callId}): Issue on ${client.clientId} - ${issue.description}`);
492
- });
1735
+ > **Note on typing.** The sample types no longer carry a `Record<string, unknown>` index signature.
1736
+ > That signature allowed arbitrary top-level keys but also silently accepted typos on real fields and
1737
+ > weakened autocomplete. Custom data belongs in `attachments`, which is typed as such. If you were
1738
+ > assigning ad-hoc keys directly onto a sample object, move them into `attachments`.
1739
+
1740
+ ### Matching peer connections — by **event**, not by storage
1741
+
1742
+ The observer correlates the SFU side with the client side **at the peer-connection level**: a
1743
+ mediasoup WebRTC transport and a client's `RTCPeerConnection` share the same id, so whenever an
1744
+ observed peer connection's id matches one of the router's WebRTC transport ids, that's a match.
1745
+
1746
+ **The observer does not store the router (or its sample) on any entity.** Instead, for **every**
1747
+ matching peer connection it emits **`mediasoup-router-matched-with-peer-connection`** and steps
1748
+ back — *your application* decides what the pairing means. The payload carries the full peer-connection
1749
+ ancestry, so you get the router **and** the matched `observedPeerConnection`, `observedClient` and
1750
+ `observedCall` in one place. Stamp the `routerId` into the peer connection's / client's `appData`,
1751
+ build your own index, attach the server sample to the call in your database — whatever fits.
1752
+
1753
+ This matching is **opt-in**: pass `matchPeerConnectionByWebRtcTransportId: true` to
1754
+ `createObservedMediasoupRouter`. When enabled, as peer connections are observed
1755
+ (`peer-connection-added`) the observer checks whether the peer connection's id is one of the router's
1756
+ WebRTC transport ids; on a hit it emits — once per matching peer connection — and keeps watching, so a
1757
+ router serving many participants emits one match per participant's transport. When the flag is omitted
1758
+ or `false`, no matching is performed and the event never fires. The internal listener is removed
1759
+ automatically when the router closes or the observer closes.
1760
+
1761
+ ### Ordering contract — observe the router first
1762
+
1763
+ Matching is **forward-only by design**, and that is sufficient because the lifecycle ordering is
1764
+ **guaranteed, not racy**:
1765
+
1766
+ - `ObservedMediasoupRouter` works purely by **subscribing to mediasoup's `observer` API**, so it can
1767
+ only see events that happen *after* it is created. You therefore create it the moment the router
1768
+ exists — **before** any transport is added to it — and it captures the rest going forward.
1769
+ - A mediasoup transport is always created **on the server first**; only then can the client connect
1770
+ to it, produce/consume, and begin shipping `ClientSample`s. So a peer connection — and the
1771
+ `peer-connection-added` event it triggers — can never appear before its server-side WebRTC
1772
+ transport already exists (and has been observed by the router).
1773
+
1774
+ Put together: by the time a `peer-connection-added` fires, the router has already recorded that
1775
+ transport's id in `webrtcTransportIds`, so a single forward-looking listener catches every match. No
1776
+ back-scan of existing peer connections and no re-check on transport creation are needed — the
1777
+ observer deliberately does **not** look backwards.
1778
+
1779
+ **Your responsibility:** call `createObservedMediasoupRouter(...)` as early as the router exists
1780
+ (before transports are added or samples are accepted). If you register the router *after* its
1781
+ transports are created or after the client's first sample, those events are already in the past and
1782
+ the corresponding matches are missed.
1783
+
1784
+ When the underlying mediasoup router closes, its `close` propagates to `ObservedMediasoupRouter`,
1785
+ which sets the sample's `closedAt` and emits **`mediasoup-router-removed`** — your cue to read /
1786
+ persist the final `observedRouter.sample` and drop your reference to it.
1787
+
1788
+ ### Options — `observer.createObservedMediasoupRouter(settings)`
1789
+
1790
+ | Field | Type | Required | Meaning |
1791
+ |-------|------|----------|---------|
1792
+ | `router` | `mediasoup.types.Router` | yes | the live router to observe; the observer attaches to `router.observer`. `.id` and the sample's `routerId` come from `router.id` |
1793
+ | `appData` | `Record<string, unknown>` | no | application-owned bag on the `ObservedMediasoupRouter` |
1794
+ | `attachments` | `Record<string, unknown>` | no | free-form data; carried on `sample.attachments` |
1795
+ | `matchPeerConnectionByWebRtcTransportId` | `boolean` | no | opt in to peer-connection matching: emit `mediasoup-router-matched-with-peer-connection` for each peer connection whose id matches one of the router's WebRTC transport ids. Omitted / `false` → no matching, the event never fires |
1796
+
1797
+ Peer-connection matching is **off by default**; enable it with
1798
+ `matchPeerConnectionByWebRtcTransportId: true`. Returns the `ObservedMediasoupRouter`, or `undefined`
1799
+ if the observer is closed (a router with the same id returns the existing instance — both warn).
1800
+
1801
+ Useful members on the returned object: `.sample` (the in-memory `MediasoupRouterSample`, with
1802
+ `createdAt` / `closedAt?` on it), `.appData`, `.attachments`, `.webrtcTransportIds: Set<string>`,
1803
+ `.id`, `.close()`.
1804
+
1805
+ ### Example
1806
+
1807
+ ```ts
1808
+ import { Observer, InMemorySink } from '@observertc/observer-js';
1809
+ import type { ObservedMediasoupRouterScope, ObservedPeerConnectionScope } from '@observertc/observer-js';
1810
+
1811
+ const observer = new Observer();
1812
+
1813
+ // 1) Feed client samples as usual so the observer knows about calls, clients & peer connections.
1814
+ // (e.g. transport-layer: observer.accept(clientSample, context))
1815
+
1816
+ // 2) Observe the SFU side; opt in to peer-connection matching. State accumulates in `.sample`.
1817
+ const router = /* your mediasoup router */ undefined as any;
1818
+ const observedRouter = observer.createObservedMediasoupRouter({
1819
+ router,
1820
+ matchPeerConnectionByWebRtcTransportId: true,
1821
+ });
1822
+
1823
+ // For large meetings, sample it yourself on your own cadence (see "Memory & large meetings"):
1824
+ // setInterval(() => persist(observedRouter.sample), 10_000);
1825
+
1826
+ // 3) Every peer connection whose id matches one of the router's WebRTC transport ids fires this —
1827
+ // WE decide what to do with each pairing. The payload carries the full ancestry.
1828
+ observer.on('mediasoup-router-matched-with-peer-connection',
1829
+ ({ observedMediasoupRouter, observedCall, observedPeerConnection }:
1830
+ ObservedMediasoupRouterScope & ObservedPeerConnectionScope) => {
1831
+ (observedPeerConnection.appData ??= {}).routerId = observedMediasoupRouter.id;
1832
+ myStore.linkRouterToCall(observedCall.callId, observedMediasoupRouter.id);
1833
+ },
1834
+ );
1835
+
1836
+ // 4) The router closed — read/persist the final state, then drop your reference.
1837
+ observer.on('mediasoup-router-removed', ({ observedMediasoupRouter }: ObservedMediasoupRouterScope) => {
1838
+ persist(observedMediasoupRouter.sample); // its `closedAt` is set
1839
+ });
1840
+ ```
1841
+
1842
+ ### Why event-driven matching instead of storing on the call
1843
+
1844
+ - **Loose coupling.** The call model stays about client telemetry; the SFU view lives on its own
1845
+ `ObservedMediasoupRouter` and is associated only if and how *you* choose.
1846
+ - **You own the association.** One router serves many peer connections (across clients and calls),
1847
+ and the right place to keep that mapping is application-specific — so the observer hands you each
1848
+ peer-connection match and gets out of the way.
1849
+ - **You own the sampling.** The router sample is plain in-memory state you read on your own terms;
1850
+ for large meetings, sample/persist it yourself (see [Memory & large meetings](#memory--large-meetings))
1851
+ rather than relying on the library to evict — it deliberately doesn't.
1852
+
1853
+ ---
1854
+
1855
+ ## Sinks (per-client sample persistence)
1856
+
1857
+ A **sink** receives the samples a client accepts — for archival, streaming, or later offline
1858
+ replay. Each `ObservedClient` gets its **own** sink, produced by the
1859
+ `ObserverConfig.createClientSink` factory when the client is created (return `undefined` for no
1860
+ sink). The client pushes every accepted sample to its sink, and `end()`s it on close.
1861
+
1862
+ ### The `ClientSampleSink` base class
1863
+
1864
+ `ClientSampleSink` is an **abstract base class** (a typed `EventEmitter`). You create a sink by
1865
+ **subclassing it** and implementing `write` and `end`. It is **object-mode**: `write` receives
1866
+ the `ClientSample` *object*, so each sink decides how (or whether) to serialize it — JSON line,
1867
+ protobuf, a remote POST body, an in-memory push, etc.
1868
+
1869
+ ```ts
1870
+ import { ClientSampleSink, ClientSample } from '@observertc/observer-js';
1871
+
1872
+ abstract class ClientSampleSink /* extends EventEmitter */ {
1873
+ abstract write(sample: ClientSample): boolean; // accept one sample; `false` = backpressure
1874
+ abstract end(): void; // flush; emit `close` when the destination is ready
1875
+
1876
+ // typed events (inherited): the listener signature is inferred from the event name
1877
+ on(event: 'close' | 'finish' | 'drain', listener: () => void): this;
1878
+ on(event: 'error', listener: (err: Error) => void): this;
1879
+ // ...and the matching `once` / `off` / `emit`
493
1880
  }
494
1881
  ```
495
1882
 
496
- ### 6. Best Practices
1883
+ | Event | Meaning |
1884
+ |-------|---------|
1885
+ | `close` | the destination is fully written and closed (e.g. a file flushed and its fd closed) — "ready" |
1886
+ | `error` | the destination failed |
1887
+ | `finish` | `end()` was processed and queued data flushed (before `close`) |
1888
+ | `drain` | the buffer drained after backpressure; safe to write more |
1889
+
1890
+ The library calls `write(sample)` **synchronously** per accepted sample (it is not awaited),
1891
+ `end()`s the sink when the client closes, and attaches an `error` listener so a failing sink
1892
+ can't crash the process (it also catches throws from `write`/`end`). The application — which
1893
+ created the sink — listens for `close` (destination ready) and `error`. Because `write` isn't
1894
+ awaited in the `accept()` hot path, **backpressure and batching are the sink's concern**.
1895
+
1896
+ ### Built-in sinks
1897
+
1898
+ ```ts
1899
+ import { Observer, createJsonlFileSinkFactory } from '@observertc/observer-js';
1900
+
1901
+ const observer = new Observer({
1902
+ // one ./stats/<callId>__<clientId>.jsonl per client
1903
+ createClientSink: createJsonlFileSinkFactory({ directory: './stats' }),
1904
+ });
1905
+
1906
+ // React when a sink is created for a client:
1907
+ observer.on('client-sink-created', ({ observedClient, sink }) => {
1908
+ sink.on('close', () => {
1909
+ // the file is fully flushed and its fd closed — ready to upload, move, etc.
1910
+ });
1911
+ });
1912
+ ```
1913
+
1914
+ | Export | Signature | Notes |
1915
+ |--------|-----------|-------|
1916
+ | `createJsonlFileSinkFactory` | `({ directory, flags?, getFileName?, serializeSample? }) => ClientSampleSinkFactory` | per-client JSONL files; path defaults to `${callId}__${clientId}.jsonl` under `directory` (which **must exist**) |
1917
+ | `createJsonlFileSink` | `({ path, flags?, serializeSample? }) => ClientSampleSink` | a single JSONL file; wraps `fs.WriteStream` and re-emits its `close`/`finish`/`drain`/`error` |
1918
+ | `JsonlFileSink` | `class extends ClientSampleSink` | the underlying class; exposes `readonly path` so a `close` handler knows which file is ready |
1919
+ | `createInMemorySink` / `InMemorySink` | `(samples?: ClientSample[]) => InMemorySink` | collects the accepted **sample objects** into `.samples: ClientSample[]`; emits `close` on `end()` |
497
1920
 
498
- - **Resource Management**: Always call `observer.close()`, `call.close()`, and `client.close()` when entities are no longer needed to free resources and stop timers.
499
- - **Error Handling**: Wrap calls to library methods in `try...catch` blocks where appropriate, especially for operations that might throw errors based on state (e.g., creating an entity that already exists if not using `getOrCreate` patterns).
500
- - **Event Listener Cleanup**: If dynamically adding/removing listeners, ensure they are properly removed (e.g., using `emitter.off()` or `emitter.removeListener()`) to prevent memory leaks, especially for short-lived monitored entities.
501
- - **`ClientSample` Accuracy**: The quality of monitoring heavily depends on the completeness and correctness of the `ClientSample` data provided. Ensure thorough mapping from `getStats()`.
502
- - **Update Policies**: Choose update policies carefully based on the desired granularity of updates and performance considerations.
1921
+ `serializeSample?: (sample: ClientSample) => string` overrides the default `JSON.stringify` for
1922
+ the JSONL sinks (e.g. to redact or reshape before writing).
503
1923
 
504
- ### 7. Troubleshooting
1924
+ ### Reading sink-specific info (e.g. the file path)
505
1925
 
506
- - **Memory Leaks**: Ensure `close()` is called on all entities. Check for unremoved event listeners.
507
- - **No Events / Missing Updates**:
508
- - Verify `observer.accept()` is being called with correctly formatted `ClientSample` data.
509
- - Ensure `callId` and `clientId` in samples match expectations.
510
- - Check if `updatePolicy` and `updateIntervalInMs` are configured as intended.
511
- - **Debugging**: Utilize `console.log` within event handlers at different levels (Observer, Call, Client) to trace data flow and state changes. Use `appData` to add correlation IDs for easier debugging.
1926
+ The bus hands you the sink as the base `ClientSampleSink`. To read information specific to a sink
1927
+ type — for a file sink, where it was written — **narrow with `instanceof`** and read the sink's
1928
+ public fields. `JsonlFileSink` exposes `path`:
512
1929
 
513
- ### 8. TypeScript Support
1930
+ ```ts
1931
+ import { JsonlFileSink } from '@observertc/observer-js';
514
1932
 
515
- The library is written in TypeScript and provides type definitions.
516
- Use generics with `Observer`, `ObservedCall`, and `ObservedClient` to type your custom `appData`.
1933
+ observer.on('client-sink-created', ({ observedClient, sink }) => {
1934
+ if (sink instanceof JsonlFileSink) {
1935
+ const { path } = sink; // the file this client's samples go to
1936
+ sink.once('close', () => uploadFile(path)); // close = flushed & fd closed → ready
1937
+ }
1938
+ });
1939
+ ```
1940
+
1941
+ The general pattern: each concrete sink exposes whatever it wants as `public readonly` fields, and
1942
+ consumers narrow (`instanceof YourSink`) to read them. Your own sinks do the same.
1943
+
1944
+ ### Writing your own sink
517
1945
 
518
- ```typescript
519
- interface MyClientAppData {
520
- userId: string;
521
- role: 'admin' | 'user';
1946
+ Subclass `ClientSampleSink` and emit the lifecycle events yourself — for any non-file
1947
+ destination (a remote endpoint, a message queue, an object store, …):
1948
+
1949
+ ```ts
1950
+ import { ClientSampleSink, ClientSample, ClientSampleSinkFactory } from '@observertc/observer-js';
1951
+
1952
+ class HttpSink extends ClientSampleSink {
1953
+ private buffer: ClientSample[] = [];
1954
+ constructor(private readonly url: string) { super(); }
1955
+
1956
+ write(sample: ClientSample): boolean {
1957
+ this.buffer.push(sample); // batch; decide your own backpressure
1958
+ return true;
1959
+ }
1960
+ end(): void {
1961
+ fetch(this.url, { method: 'POST', body: JSON.stringify(this.buffer) })
1962
+ .then(() => this.emit('close')) // signal "destination ready"
1963
+ .catch((err) => this.emit('error', err));
1964
+ }
522
1965
  }
523
- const client = call.createObservedClient<MyClientAppData>({
524
- clientId: 'user1',
525
- appData: { userId: 'u-123', role: 'admin' },
1966
+
1967
+ const createClientSink: ClientSampleSinkFactory = ({ clientId, observedCall }) =>
1968
+ new HttpSink(`https://stats.example.com/${observedCall.callId}/${clientId}`);
1969
+
1970
+ const observer = new Observer({ createClientSink });
1971
+ ```
1972
+
1973
+ `observedClient.sink?` exposes the created sink; the `client-sink-created` event delivers it on
1974
+ the bus with full ancestry. `ClientSampleSinkFactory` is
1975
+ `(p: { clientId: string; observedCall: ObservedCall }) => ClientSampleSink | undefined`.
1976
+
1977
+ ---
1978
+
1979
+ ## Injecting data into a client
1980
+
1981
+ Sometimes the application holds data that belongs on a client's record but isn't part of the
1982
+ client-reported `ClientSample` — a room id or display name, an application-level event
1983
+ (*"recording started"*), a server-detected issue, an extension stat, or a device/meta item.
1984
+ `ObservedClient` exposes **injection** methods that merge such data into the client's sample stream,
1985
+ so it updates the live model **and** is persisted to the client's
1986
+ [sink](#sinks-per-client-sample-persistence) exactly like sampled data.
1987
+
1988
+ | Method | Adds to the sample's | Surfaces as |
1989
+ |--------|----------------------|-------------|
1990
+ | `injectAttachment(attachments)` | `attachments` (merged via `Object.assign`) | `observedClient.attachments` |
1991
+ | `injectEvent(event: ClientEvent)` | `clientEvents` | `client-event` (plus any state the event drives) |
1992
+ | `injectIssue(issue: ClientIssue)` | `clientIssues` | `client-issue` |
1993
+ | `injectMetaData(meta: ClientMetaData)` | `clientMetaItems` | `client-metadata` |
1994
+ | `injectExtensionStat(stat: ExtensionStat)` | `extensionStats` | `client-extension-stats` |
1995
+
1996
+ ### When the injected data lands
1997
+
1998
+ Injection is timing-aware so nothing is dropped, regardless of *when* you call it:
1999
+
2000
+ - **During a sample's processing** — e.g. from inside a `client-updated` / `client-event` handler,
2001
+ which run within `accept()` — the data is applied to the **current** sample immediately: reflected
2002
+ in entity state and written to the sink as part of that sample.
2003
+ - **Between samples** — the data is buffered and merged into the **next** `accept()`'s sample.
2004
+ - **On `close()` with pending injections and no further sample** — the buffer is flushed as a final
2005
+ synthetic sample (applied to state and written to the sink) before the sink is ended, so a
2006
+ last-moment injection is never lost.
2007
+
2008
+ In every case the injected data both updates the live `ObservedClient` and reaches the per-client
2009
+ sink — the sink always receives the final, **injection-merged** sample (the sink write happens at the
2010
+ end of `accept()`, after the merge).
2011
+
2012
+ ### Example
2013
+
2014
+ ```ts
2015
+ // Enrich at creation from your app's knowledge of the participant. Injecting in `client-added`
2016
+ // (which runs just before the first accept) lands on the first sample.
2017
+ observer.on('client-added', ({ observedClient }) => {
2018
+ observedClient.injectAttachment({ roomId: lookupRoomId(observedClient.clientId) });
2019
+ });
2020
+
2021
+ // Application-level signals at any time:
2022
+ const client = observer.getObservedCall(callId)?.getObservedClient(clientId);
2023
+ client?.injectEvent({ type: 'RECORDING_STARTED', timestamp: Date.now() });
2024
+ client?.injectIssue({ type: 'app-kicked-participant', timestamp: Date.now() });
2025
+ ```
2026
+
2027
+ `attachments` are latest-wins (like sampled `attachments`): injecting a key overwrites its previous
2028
+ value. `appData` is unaffected — injections flow into the sample/telemetry, not the app-owned
2029
+ `appData` bag (see [Ingestion](#ingestion-accept-context--lifecycle)).
2030
+
2031
+ ## Logging
2032
+
2033
+ `observer-js` logs through a single, swappable sink. Out of the box it writes `debug` and
2034
+ above to `console` (verbose — install your own sink for production). Funnel everything into your
2035
+ logger:
2036
+
2037
+ ```ts
2038
+ import { setObserverLogger, type ObserverLogger } from '@observertc/observer-js';
2039
+
2040
+ setObserverLogger({
2041
+ trace: (m, ...a) => myLogger.trace(`[${m}]`, ...a),
2042
+ debug: (m, ...a) => myLogger.debug(`[${m}]`, ...a),
2043
+ info: (m, ...a) => myLogger.info(`[${m}]`, ...a),
2044
+ warn: (m, ...a) => myLogger.warn(`[${m}]`, ...a),
2045
+ error: (m, ...a) => myLogger.error(`[${m}]`, ...a),
526
2046
  });
527
- // client.appData will be typed as MyClientAppData | undefined
528
2047
  ```
529
2048
 
530
- ### 9. Contributing
2049
+ `createLogger(moduleName)` is also exported for your own modules. See
2050
+ **[`docs/logging.md`](./docs/logging.md)** for pino / winston / console recipes, level
2051
+ filtering, per-module routing, and full silencing.
2052
+
2053
+ ---
2054
+
2055
+ ## Design notes
2056
+
2057
+ **[`docs/design-notes.md`](./docs/design-notes.md)** covers the reasoning behind the library rather
2058
+ than its API: why client-detectable conditions are never re-derived server-side, why each shipped
2059
+ detector exists, what was deliberately *not* built and why, the WebRTC domain facts that shaped the
2060
+ implementation (ICE-Lite disconnect waves, counter resets, why ICE and RTCP RTT must never be
2061
+ blended), and an operational threshold reference.
2062
+
2063
+ ---
2064
+
2065
+ ## Error-handling philosophy
2066
+
2067
+ The library **warns and degrades; it does not throw** on operational problems:
2068
+
2069
+ - `createObservedCall` / `createObservedClient` on a closed parent → warn + return `undefined`.
2070
+ - Duplicate id → warn + return the **existing** instance.
2071
+ - `accept()` on a closed client → warn + no-op.
2072
+ - Sample missing `callId`/`clientId`, or observer closed → `sample-rejected` event.
2073
+ - A throwing accept-middleware → warn + drop that sample (never crashes `accept()`).
2074
+
2075
+ Therefore `create*` and `getOrCreate*` return `T | undefined`; **guard the result.** The
2076
+ `Middleware` utility's internal invariants (e.g. calling `next()` twice) throw, but those throws
2077
+ are caught by `accept()` and surfaced as a warning.
2078
+
2079
+ ---
2080
+
2081
+ ## Development & extension guide
2082
+
2083
+ ```bash
2084
+ yarn install
2085
+ yarn build # tsup → dist/ (dual ESM .mjs + CJS .js, single entry, .d.ts/.d.mts + sourcemaps)
2086
+ yarn lint # eslint -c .eslintrc.json "src/**/*.ts"
2087
+ yarn typecheck # tsc --noEmit
2088
+ yarn test # jest
2089
+ ```
2090
+
2091
+ The build is driven by [`tsup`](https://tsup.egoist.dev) (config in `tsup.config.ts`): a single
2092
+ entry (`src/index.ts`), dual ESM + CommonJS output to `dist/` (`index.mjs` / `index.js`) with
2093
+ `.d.mts` / `.d.ts` types and sourcemaps, targeting Node 20. CI (`.github/workflows/ci.yml`) runs
2094
+ lint + typecheck + **build** + test on every push/PR.
2095
+
2096
+ **Project layout** (`src/`): `Observer.ts`, `ObservedCall.ts`, `ObservedClient.ts`,
2097
+ `ObservedPeerConnection.ts`, the `Observed*` sub-stat classes, `ObserverEvents.ts` (the typed
2098
+ event map + scope types), `detectors/` (`Detector`, `Detectors`, and one file per detector),
2099
+ `validators/` (`Validator`, `Validators`, one file per validator), `issues/` (`ActiveClientIssue`,
2100
+ `ActiveIssueTracker`, `ActiveIssuesRegistry`, `ObservedClientIssueRegistry`), `scores/`,
2101
+ `resolvers/` (remote-track resolvers), `utils/` (`stats`, `SlidingWindow`, `TrendTester`,
2102
+ `CallHealthAggregator`), `common/` (`logger`, `utils`, `Middleware`), `schema/`
2103
+ (sample/event/meta types), and `sinks/` (the `ClientSampleSink` base + `JsonlFileSink` /
2104
+ `InMemorySink`, re-exported from the package root).
2105
+
2106
+ **Conventions to follow when developing further:**
2107
+
2108
+ - *Single event bus.* New consumer-facing events go in `ObserverEvents.ts` with an object
2109
+ payload `[<Scope> & { …subject }]`, and are emitted via the component's
2110
+ `_notify(type, { ...this.eventScope, …subject })`. Each component has a precomputed
2111
+ `eventScope` field and a thin `_notify` wrapper around the right emitter. Keep purely internal
2112
+ coordination as **local** EventEmitter events (and remember to `off` them on close).
2113
+ - *Warn, don't throw* on operational/edge conditions; return `undefined` where a value can't be produced.
2114
+ - *Counter-reset-safe deltas.* When computing a delta from a cumulative counter, never emit a
2115
+ negative value (guard `curr >= prev`), to survive counter resets / SSRC reuse.
2116
+ - *Explicit accumulation.* The per-sample metric accumulation in `accept()` is intentionally
2117
+ explicit and not abstracted — match that style.
2118
+ - *Detectors are server-side.* Add cross-client detectors on `ObservedCall.detectors`; don't
2119
+ re-implement client-detectable signals.
2120
+
2121
+ **Recipes:**
2122
+
2123
+ - *Add an event:* add the key + payload to `ObserverEvents`; in the owning component call
2124
+ `this._notify('my-event', { ...this.eventScope, subject })`.
2125
+ - *Add a per-stream metric:* add the field to the relevant `Observed*Rtp`/track class, populate
2126
+ it in its `update()` (reset at the top of `update()` if it's per-tick), and read it from a
2127
+ `*-updated` handler.
2128
+ - *Add a detector:* implement `Detector`, register it on `call-added` via
2129
+ `observedCall.detectors.add(...)`, surface findings with `observedCall.addIssue(...)`.
531
2130
 
532
- (Placeholder for contribution guidelines - e.g., link to CONTRIBUTING.md, coding standards, pull request process)
2131
+ ---
533
2132
 
534
- ### 10. License
2133
+ ## License
535
2134
 
536
- This project is licensed under the [MIT License](LICENSE).
2135
+ Apache-2.0. Part of the [ObserverTC](https://github.com/observertc) ecosystem.