pinecall-protocol 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,303 @@
1
+ """Generated from schema/metrics.json: every livekit-agents 1.8 metric, verbatim."""
2
+
3
+ from typing import Annotated, Literal
4
+
5
+ from pydantic import Field
6
+
7
+ from pinecall_protocol._base import WireModel
8
+
9
+
10
+ # livekit's Metadata, nested exactly as it nests it.
11
+ class Metadata(WireModel):
12
+ """Which model and provider produced a block."""
13
+
14
+ model_name: str | None = None
15
+ model_provider: str | None = None
16
+
17
+
18
+ # Two arrive for one reply when a tool ran in between; they share the speech_id.
19
+ class LLMMetrics(WireModel):
20
+ """One LLM request, measured by the session's llm node."""
21
+
22
+ type: Literal["llm_metrics"] = "llm_metrics"
23
+ label: str
24
+ request_id: str
25
+ timestamp: float
26
+ duration: float
27
+ ttft: float
28
+ cancelled: bool
29
+ completion_tokens: int
30
+ prompt_tokens: int
31
+ prompt_cached_tokens: int
32
+ cache_creation_tokens: int | None = None
33
+ reasoning_tokens: int | None = None
34
+ total_tokens: int
35
+ tokens_per_second: float
36
+ speech_id: str | None = None
37
+ metadata: Metadata | None = None
38
+
39
+
40
+ # A streaming STT reports one per connection segment, with duration 0.
41
+ class STTMetrics(WireModel):
42
+ """One speech-to-text request, measured by the session's stt node."""
43
+
44
+ type: Literal["stt_metrics"] = "stt_metrics"
45
+ label: str
46
+ request_id: str
47
+ timestamp: float
48
+ duration: float
49
+ audio_duration: float
50
+ input_tokens: int | None = None
51
+ output_tokens: int | None = None
52
+ streamed: bool
53
+ acquire_time: float | None = None
54
+ connection_reused: bool | None = None
55
+ metadata: Metadata | None = None
56
+
57
+
58
+ # One reply may produce several, one per sentence segment.
59
+ class TTSMetrics(WireModel):
60
+ """One text-to-speech request, measured by the session's tts node."""
61
+
62
+ type: Literal["tts_metrics"] = "tts_metrics"
63
+ label: str
64
+ request_id: str
65
+ timestamp: float
66
+ ttfb: float
67
+ duration: float
68
+ audio_duration: float
69
+ cancelled: bool
70
+ characters_count: int
71
+ input_tokens: int | None = None
72
+ output_tokens: int | None = None
73
+ streamed: bool
74
+ acquire_time: float | None = None
75
+ connection_reused: bool | None = None
76
+ segment_id: str | None = None
77
+ speech_id: str | None = None
78
+ metadata: Metadata | None = None
79
+
80
+
81
+ # Not a per-turn measure.
82
+ class VADMetrics(WireModel):
83
+ """The voice activity detector's health, reported about once a second while it runs."""
84
+
85
+ type: Literal["vad_metrics"] = "vad_metrics"
86
+ label: str
87
+ timestamp: float
88
+ idle_time: float
89
+ inference_duration_total: float
90
+ inference_count: int
91
+ metadata: Metadata | None = None
92
+
93
+
94
+ # One per user turn.
95
+ class EOUMetrics(WireModel):
96
+ """How long the session took to decide that the caller had finished."""
97
+
98
+ type: Literal["eou_metrics"] = "eou_metrics"
99
+ timestamp: float
100
+ end_of_utterance_delay: float
101
+ transcription_delay: float
102
+ on_user_turn_completed_delay: float
103
+ speech_id: str | None = None
104
+ metadata: Metadata | None = None
105
+
106
+
107
+ # It listens to the audio and says whether the caller is done.
108
+ class EOTInferenceMetrics(WireModel):
109
+ """One prediction by the end-of-turn model."""
110
+
111
+ type: Literal["eot_inference_metrics"] = "eot_inference_metrics"
112
+ timestamp: float
113
+ total_duration: float
114
+ detection_delay: float
115
+ prediction_duration: float
116
+ num_requests: int | None = None
117
+ metadata: Metadata | None = None
118
+
119
+
120
+ # Reported when the detector runs, not per turn.
121
+ class InterruptionMetrics(WireModel):
122
+ """The interruption detector's latest inference and its running counts."""
123
+
124
+ type: Literal["interruption_metrics"] = "interruption_metrics"
125
+ timestamp: float
126
+ total_duration: float
127
+ prediction_duration: float
128
+ detection_delay: float
129
+ num_interruptions: int
130
+ num_backchannels: int
131
+ num_requests: int
132
+ metadata: Metadata | None = None
133
+
134
+
135
+ class RealtimeCachedTokenDetails(WireModel):
136
+ """Of a realtime model's cached input, how much was audio, text or image."""
137
+
138
+ audio_tokens: int | None = None
139
+ text_tokens: int | None = None
140
+ image_tokens: int | None = None
141
+
142
+
143
+ class RealtimeInputTokenDetails(WireModel):
144
+ """What a realtime model read, by kind, with the cached part broken out."""
145
+
146
+ audio_tokens: int | None = None
147
+ text_tokens: int | None = None
148
+ image_tokens: int | None = None
149
+ cached_tokens: int | None = None
150
+ cached_tokens_details: RealtimeCachedTokenDetails | None = None
151
+
152
+
153
+ class RealtimeOutputTokenDetails(WireModel):
154
+ """What a realtime model produced, by kind."""
155
+
156
+ text_tokens: int | None = None
157
+ audio_tokens: int | None = None
158
+ image_tokens: int | None = None
159
+
160
+
161
+ # Pinecall's default pipeline never emits it; a realtime provider would.
162
+ class RealtimeModelMetrics(WireModel):
163
+ """One response from a speech-to-speech model, which replaces STT, LLM and TTS at once."""
164
+
165
+ type: Literal["realtime_model_metrics"] = "realtime_model_metrics"
166
+ label: str | None = None
167
+ request_id: str
168
+ timestamp: float
169
+ duration: float | None = None
170
+ session_duration: float | None = None
171
+ ttft: float | None = None
172
+ cancelled: bool | None = None
173
+ input_tokens: int | None = None
174
+ output_tokens: int | None = None
175
+ total_tokens: int | None = None
176
+ tokens_per_second: float | None = None
177
+ input_token_details: RealtimeInputTokenDetails
178
+ output_token_details: RealtimeOutputTokenDetails
179
+ acquire_time: float | None = None
180
+ connection_reused: bool | None = None
181
+ metadata: Metadata | None = None
182
+
183
+
184
+ # No Pinecall channel has one today; carried because the library measures it.
185
+ class AvatarMetrics(WireModel):
186
+ """Timing of a video avatar worker, when one is in the chain."""
187
+
188
+ type: Literal["avatar_metrics"] = "avatar_metrics"
189
+ timestamp: float
190
+ playback_latency: float | None = None
191
+ session_started_time: float | None = None
192
+ avatar_joined_time: float | None = None
193
+ metadata: Metadata | None = None
194
+
195
+
196
+ # livekit's MetricsMetadata, as it sits on the ChatMessage.
197
+ class TurnMetadata(WireModel):
198
+ """Which model handled one leg of a turn."""
199
+
200
+ model_name: str | None = None
201
+ model_provider: str | None = None
202
+
203
+
204
+ # livekit stamps it on the user ChatMessage; every field is optional, since a text session has no
205
+ # speech to time.
206
+ class UserTurnMetrics(WireModel):
207
+ """What the session measured about the caller's turn."""
208
+
209
+ started_speaking_at: float | None = None
210
+ stopped_speaking_at: float | None = None
211
+ transcription_delay: float | None = None
212
+ end_of_turn_delay: float | None = None
213
+ on_user_turn_completed_delay: float | None = None
214
+ stt_metadata: TurnMetadata | None = None
215
+
216
+
217
+ # livekit stamps it on the assistant ChatMessage; every field is optional, since a text session has
218
+ # no audio to time.
219
+ class AgentTurnMetrics(WireModel):
220
+ """What the session measured about the agent's reply."""
221
+
222
+ started_speaking_at: float | None = None
223
+ stopped_speaking_at: float | None = None
224
+ llm_node_ttft: float | None = None
225
+ llm_node_tps: float | None = None
226
+ llm_node_ttfs: float | None = None
227
+ tts_node_ttfb: float | None = None
228
+ playback_latency: float | None = None
229
+ e2e_latency: float | None = None
230
+ provider_request_ids: list[str] | None = None
231
+ llm_metadata: TurnMetadata | None = None
232
+ tts_metadata: TurnMetadata | None = None
233
+
234
+
235
+ # One row per provider and model, as livekit's usage collector sums it.
236
+ class LLMModelUsage(WireModel):
237
+ """Everything one LLM consumed over the call."""
238
+
239
+ type: Literal["llm_usage"] = "llm_usage"
240
+ provider: str
241
+ model: str
242
+ input_tokens: int | None = None
243
+ input_cached_tokens: int | None = None
244
+ input_cache_creation_tokens: int | None = None
245
+ input_audio_tokens: int | None = None
246
+ input_cached_audio_tokens: int | None = None
247
+ input_text_tokens: int | None = None
248
+ input_cached_text_tokens: int | None = None
249
+ input_image_tokens: int | None = None
250
+ input_cached_image_tokens: int | None = None
251
+ output_tokens: int | None = None
252
+ output_audio_tokens: int | None = None
253
+ output_text_tokens: int | None = None
254
+ output_reasoning_tokens: int | None = None
255
+ session_duration: float | None = None
256
+
257
+
258
+ class TTSModelUsage(WireModel):
259
+ """Everything one TTS consumed over the call."""
260
+
261
+ type: Literal["tts_usage"] = "tts_usage"
262
+ provider: str
263
+ model: str
264
+ input_tokens: int | None = None
265
+ output_tokens: int | None = None
266
+ characters_count: int | None = None
267
+ audio_duration: float | None = None
268
+
269
+
270
+ class STTModelUsage(WireModel):
271
+ """Everything one STT consumed over the call."""
272
+
273
+ type: Literal["stt_usage"] = "stt_usage"
274
+ provider: str
275
+ model: str
276
+ input_tokens: int | None = None
277
+ output_tokens: int | None = None
278
+ audio_duration: float | None = None
279
+
280
+
281
+ class InterruptionModelUsage(WireModel):
282
+ """How often the interruption detector was asked over the call."""
283
+
284
+ type: Literal["interruption_usage"] = "interruption_usage"
285
+ provider: str
286
+ model: str
287
+ total_requests: int | None = None
288
+
289
+
290
+ class EOTModelUsage(WireModel):
291
+ """How often the end-of-turn model was asked over the call."""
292
+
293
+ type: Literal["eot_usage"] = "eot_usage"
294
+ provider: str
295
+ model: str
296
+ total_requests: int | None = None
297
+
298
+
299
+ # One usage row, told apart by its type tag.
300
+ type ModelUsage = Annotated[
301
+ LLMModelUsage | TTSModelUsage | STTModelUsage | InterruptionModelUsage | EOTModelUsage,
302
+ Field(discriminator="type"),
303
+ ]
File without changes
@@ -0,0 +1,341 @@
1
+ """Generated from schema: every event and command by its wire type."""
2
+
3
+ from typing import Literal
4
+
5
+ from pinecall_protocol._base import WireModel
6
+ from pinecall_protocol.commands import (
7
+ AgentConfigure,
8
+ AgentRegister,
9
+ AgentReply,
10
+ AgentSay,
11
+ CallDial,
12
+ CallDtmf,
13
+ CallEvent,
14
+ CallHangup,
15
+ CallHold,
16
+ CallLog,
17
+ CallMute,
18
+ CallTransfer,
19
+ CallUnhold,
20
+ CallUnmute,
21
+ DevAnswer,
22
+ ParticipantMute,
23
+ ParticipantRemove,
24
+ Ping,
25
+ PromptSet,
26
+ RoomInvite,
27
+ RoomSend,
28
+ SessionConfigure,
29
+ StateSet,
30
+ SupervisorVerb,
31
+ ToolsSet,
32
+ )
33
+ from pinecall_protocol.defs import ToolResult
34
+ from pinecall_protocol.events import (
35
+ AgentConfigured,
36
+ AgentDetached,
37
+ AgentRegistered,
38
+ AgentStateChanged,
39
+ AgentTranscript,
40
+ AgentTurnEnded,
41
+ CallbackRequested,
42
+ CallDialing,
43
+ CallEnded,
44
+ CallLine,
45
+ CallRinging,
46
+ CallScore,
47
+ CallStarted,
48
+ CallSummary,
49
+ CallTransferred,
50
+ ConfirmDeclined,
51
+ ConfirmGranted,
52
+ ConfirmRequest,
53
+ CreditsExhausted,
54
+ Custom,
55
+ DevRequest,
56
+ DocsSources,
57
+ ErrorEvent,
58
+ FleetFull,
59
+ LogCaughtUp,
60
+ LogGap,
61
+ MemoryOps,
62
+ Pong,
63
+ PromptChanged,
64
+ StateChanged,
65
+ SupervisorEnded,
66
+ SupervisorReleased,
67
+ SupervisorSaid,
68
+ SupervisorTookOver,
69
+ SupervisorTransferred,
70
+ SupervisorWhispered,
71
+ ToolCall,
72
+ ToolsChanged,
73
+ UserStateChanged,
74
+ UserTranscript,
75
+ UserTurnEnded,
76
+ )
77
+ from pinecall_protocol.metrics import (
78
+ AvatarMetrics,
79
+ EOTInferenceMetrics,
80
+ EOUMetrics,
81
+ InterruptionMetrics,
82
+ LLMMetrics,
83
+ RealtimeModelMetrics,
84
+ STTMetrics,
85
+ TTSMetrics,
86
+ VADMetrics,
87
+ )
88
+ from pinecall_protocol.room import (
89
+ EventReceived,
90
+ ParticipantJoined,
91
+ ParticipantLeft,
92
+ ParticipantSpeaking,
93
+ RoomOpened,
94
+ RoomSent,
95
+ TrackPublished,
96
+ TrackUnpublished,
97
+ )
98
+
99
+ # Every event by its wire type; codec looks the model up here.
100
+ EVENTS: dict[str, type[WireModel]] = {
101
+ "agent.configured": AgentConfigured,
102
+ "agent.detached": AgentDetached,
103
+ "agent.registered": AgentRegistered,
104
+ "agent.state": AgentStateChanged,
105
+ "agent.transcript": AgentTranscript,
106
+ "call.dialing": CallDialing,
107
+ "call.ended": CallEnded,
108
+ "call.line": CallLine,
109
+ "call.ringing": CallRinging,
110
+ "call.score": CallScore,
111
+ "call.started": CallStarted,
112
+ "call.summary": CallSummary,
113
+ "call.transferred": CallTransferred,
114
+ "callback.requested": CallbackRequested,
115
+ "confirm.declined": ConfirmDeclined,
116
+ "confirm.granted": ConfirmGranted,
117
+ "confirm.request": ConfirmRequest,
118
+ "credits.exhausted": CreditsExhausted,
119
+ "custom": Custom,
120
+ "dev.request": DevRequest,
121
+ "docs.sources": DocsSources,
122
+ "error": ErrorEvent,
123
+ "event.received": EventReceived,
124
+ "fleet.full": FleetFull,
125
+ "log.caught_up": LogCaughtUp,
126
+ "log.gap": LogGap,
127
+ "memory.ops": MemoryOps,
128
+ "metrics.avatar": AvatarMetrics,
129
+ "metrics.eot": EOTInferenceMetrics,
130
+ "metrics.eou": EOUMetrics,
131
+ "metrics.interruption": InterruptionMetrics,
132
+ "metrics.llm": LLMMetrics,
133
+ "metrics.realtime": RealtimeModelMetrics,
134
+ "metrics.stt": STTMetrics,
135
+ "metrics.tts": TTSMetrics,
136
+ "metrics.vad": VADMetrics,
137
+ "participant.joined": ParticipantJoined,
138
+ "participant.left": ParticipantLeft,
139
+ "participant.speaking": ParticipantSpeaking,
140
+ "pong": Pong,
141
+ "prompt.changed": PromptChanged,
142
+ "room.opened": RoomOpened,
143
+ "room.sent": RoomSent,
144
+ "state.changed": StateChanged,
145
+ "supervisor.ended": SupervisorEnded,
146
+ "supervisor.released": SupervisorReleased,
147
+ "supervisor.said": SupervisorSaid,
148
+ "supervisor.took_over": SupervisorTookOver,
149
+ "supervisor.transferred": SupervisorTransferred,
150
+ "supervisor.whispered": SupervisorWhispered,
151
+ "tool.call": ToolCall,
152
+ "tool.result": ToolResult,
153
+ "tools.changed": ToolsChanged,
154
+ "track.published": TrackPublished,
155
+ "track.unpublished": TrackUnpublished,
156
+ "turn.agent": AgentTurnEnded,
157
+ "turn.user": UserTurnEnded,
158
+ "user.state": UserStateChanged,
159
+ "user.transcript": UserTranscript,
160
+ }
161
+
162
+ type EventType = Literal[
163
+ "agent.configured",
164
+ "agent.detached",
165
+ "agent.registered",
166
+ "agent.state",
167
+ "agent.transcript",
168
+ "call.dialing",
169
+ "call.ended",
170
+ "call.line",
171
+ "call.ringing",
172
+ "call.score",
173
+ "call.started",
174
+ "call.summary",
175
+ "call.transferred",
176
+ "callback.requested",
177
+ "confirm.declined",
178
+ "confirm.granted",
179
+ "confirm.request",
180
+ "credits.exhausted",
181
+ "custom",
182
+ "dev.request",
183
+ "docs.sources",
184
+ "error",
185
+ "event.received",
186
+ "fleet.full",
187
+ "log.caught_up",
188
+ "log.gap",
189
+ "memory.ops",
190
+ "metrics.avatar",
191
+ "metrics.eot",
192
+ "metrics.eou",
193
+ "metrics.interruption",
194
+ "metrics.llm",
195
+ "metrics.realtime",
196
+ "metrics.stt",
197
+ "metrics.tts",
198
+ "metrics.vad",
199
+ "participant.joined",
200
+ "participant.left",
201
+ "participant.speaking",
202
+ "pong",
203
+ "prompt.changed",
204
+ "room.opened",
205
+ "room.sent",
206
+ "state.changed",
207
+ "supervisor.ended",
208
+ "supervisor.released",
209
+ "supervisor.said",
210
+ "supervisor.took_over",
211
+ "supervisor.transferred",
212
+ "supervisor.whispered",
213
+ "tool.call",
214
+ "tool.result",
215
+ "tools.changed",
216
+ "track.published",
217
+ "track.unpublished",
218
+ "turn.agent",
219
+ "turn.user",
220
+ "user.state",
221
+ "user.transcript",
222
+ ]
223
+
224
+
225
+ # Events a store may drop and a slow reader may miss: the entry's ephemeral flag defaults
226
+ # to this.
227
+ EPHEMERAL_EVENTS: frozenset[str] = frozenset(
228
+ {
229
+ "agent.transcript",
230
+ "dev.request",
231
+ "log.caught_up",
232
+ "log.gap",
233
+ "metrics.vad",
234
+ "participant.speaking",
235
+ "pong",
236
+ "room.sent",
237
+ "user.transcript",
238
+ }
239
+ )
240
+
241
+
242
+ # The one event that ends a call: after it nothing more is true and the log is sealed.
243
+ TERMINAL_EVENT: str = "call.score"
244
+
245
+
246
+ # Every command by its wire type; codec looks the model up here.
247
+ COMMANDS: dict[str, type[WireModel]] = {
248
+ "agent.configure": AgentConfigure,
249
+ "agent.register": AgentRegister,
250
+ "agent.reply": AgentReply,
251
+ "agent.say": AgentSay,
252
+ "call.dial": CallDial,
253
+ "call.dtmf": CallDtmf,
254
+ "call.event": CallEvent,
255
+ "call.hangup": CallHangup,
256
+ "call.hold": CallHold,
257
+ "call.log": CallLog,
258
+ "call.mute": CallMute,
259
+ "call.transfer": CallTransfer,
260
+ "call.unhold": CallUnhold,
261
+ "call.unmute": CallUnmute,
262
+ "dev.answer": DevAnswer,
263
+ "participant.mute": ParticipantMute,
264
+ "participant.remove": ParticipantRemove,
265
+ "ping": Ping,
266
+ "prompt.set": PromptSet,
267
+ "room.invite": RoomInvite,
268
+ "room.send": RoomSend,
269
+ "session.configure": SessionConfigure,
270
+ "state.set": StateSet,
271
+ "supervisor.verb": SupervisorVerb,
272
+ "tool.result": ToolResult,
273
+ "tools.set": ToolsSet,
274
+ }
275
+
276
+ type CommandType = Literal[
277
+ "agent.configure",
278
+ "agent.register",
279
+ "agent.reply",
280
+ "agent.say",
281
+ "call.dial",
282
+ "call.dtmf",
283
+ "call.event",
284
+ "call.hangup",
285
+ "call.hold",
286
+ "call.log",
287
+ "call.mute",
288
+ "call.transfer",
289
+ "call.unhold",
290
+ "call.unmute",
291
+ "dev.answer",
292
+ "participant.mute",
293
+ "participant.remove",
294
+ "ping",
295
+ "prompt.set",
296
+ "room.invite",
297
+ "room.send",
298
+ "session.configure",
299
+ "state.set",
300
+ "supervisor.verb",
301
+ "tool.result",
302
+ "tools.set",
303
+ ]
304
+
305
+
306
+ # Which events a command lands in the log as, so a caller knows what to wait for.
307
+ PRODUCES: dict[str, tuple[str, ...]] = {
308
+ "agent.configure": ("agent.configured",),
309
+ "agent.register": ("agent.registered",),
310
+ "agent.reply": ("turn.agent",),
311
+ "agent.say": ("turn.agent",),
312
+ "call.dial": ("call.dialing",),
313
+ "call.dtmf": (),
314
+ "call.event": ("event.received",),
315
+ "call.hangup": ("call.ended",),
316
+ "call.hold": ("call.line",),
317
+ "call.log": ("custom",),
318
+ "call.mute": ("call.line",),
319
+ "call.transfer": ("call.transferred",),
320
+ "call.unhold": ("call.line",),
321
+ "call.unmute": ("call.line",),
322
+ "dev.answer": (),
323
+ "participant.mute": ("track.unpublished",),
324
+ "participant.remove": ("participant.left",),
325
+ "ping": ("pong",),
326
+ "prompt.set": ("prompt.changed",),
327
+ "room.invite": ("participant.joined",),
328
+ "room.send": ("room.sent",),
329
+ "session.configure": ("state.changed", "agent.configured"),
330
+ "state.set": ("state.changed",),
331
+ "supervisor.verb": (
332
+ "supervisor.said",
333
+ "supervisor.whispered",
334
+ "supervisor.took_over",
335
+ "supervisor.released",
336
+ "supervisor.transferred",
337
+ "supervisor.ended",
338
+ ),
339
+ "tool.result": ("tool.result",),
340
+ "tools.set": ("tools.changed",),
341
+ }