alpha-avatar-core 0.6.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- alpha_avatar_core-0.6.4.dist-info/METADATA +61 -0
- alpha_avatar_core-0.6.4.dist-info/RECORD +17 -0
- alpha_avatar_core-0.6.4.dist-info/WHEEL +4 -0
- alphaavatar/core/__init__.py +13 -0
- alphaavatar/core/env/__init__.py +20 -0
- alphaavatar/core/env/annotation.py +62 -0
- alphaavatar/core/env/observation.py +205 -0
- alphaavatar/core/media/__init__.py +26 -0
- alphaavatar/core/media/formats.py +57 -0
- alphaavatar/core/media/payload.py +142 -0
- alphaavatar/core/media/video.py +148 -0
- alphaavatar/core/perception/__init__.py +38 -0
- alphaavatar/core/perception/runtime.py +162 -0
- alphaavatar/core/perception/stream.py +138 -0
- alphaavatar/core/perception/timeline.py +186 -0
- alphaavatar/core/perception/window.py +120 -0
- alphaavatar/core/version.py +14 -0
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: alpha-avatar-core
|
|
3
|
+
Version: 0.6.4
|
|
4
|
+
Summary: Core runtime schemas and transport primitives for AlphaAvatar.
|
|
5
|
+
Project-URL: Source, https://github.com/AlphaAvatar/AlphaAvatar
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Keywords: AI,audio,multimodal,omni,realtime,rtc,runtime,video
|
|
8
|
+
Classifier: Intended Audience :: Developers
|
|
9
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Topic :: Multimedia :: Sound/Audio
|
|
14
|
+
Classifier: Topic :: Multimedia :: Video
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
16
|
+
Classifier: Topic :: Software Development :: Libraries :: Application Frameworks
|
|
17
|
+
Requires-Python: <3.12,>=3.11
|
|
18
|
+
Provides-Extra: livekit
|
|
19
|
+
Requires-Dist: livekit-api<2,>=1.0.4; extra == 'livekit'
|
|
20
|
+
Requires-Dist: livekit-protocol~=1.0; extra == 'livekit'
|
|
21
|
+
Requires-Dist: livekit<2,>=1.0.12; extra == 'livekit'
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# AlphaAvatar Core
|
|
25
|
+
|
|
26
|
+
## π§© Introduction
|
|
27
|
+
|
|
28
|
+
`avatar-core` provides the transport-agnostic runtime primitives shared across AlphaAvatar.
|
|
29
|
+
|
|
30
|
+
It defines how multimodal observations are represented, enriched with annotations, published through typed perception streams, aligned on a shared timeline, and selected as consumer-specific windows. The package intentionally remains independent from LiveKit Agents, model providers, and plugin implementations so that RTC adapters, Persona, Memory, Vision, and future multimodal components can evolve without coupling the core runtime to a specific backend.
|
|
31
|
+
|
|
32
|
+
## π¦ Package Structure
|
|
33
|
+
|
|
34
|
+
```text
|
|
35
|
+
avatar-core/
|
|
36
|
+
βββ README.md
|
|
37
|
+
βββ alphaavatar/
|
|
38
|
+
β βββ core/
|
|
39
|
+
β βββ __init__.py
|
|
40
|
+
β βββ env/ # Environment observation envelopes and annotations.
|
|
41
|
+
β β βββ __init__.py
|
|
42
|
+
β β βββ annotation.py # Structured metadata attached to an observation.
|
|
43
|
+
β β βββ observation.py # Runtime observation envelope for video, audio, screen, and events.
|
|
44
|
+
β βββ media/ # Backend-independent multimodal payload representations.
|
|
45
|
+
β β βββ __init__.py
|
|
46
|
+
β β βββ formats.py # Payload formats and views such as raw, annotated, and derived.
|
|
47
|
+
β β βββ payload.py # Thread-safe multi-representation MediaPayload container.
|
|
48
|
+
β β βββ video.py # Generic video frame buffers and video payload helpers.
|
|
49
|
+
β βββ perception/ # Full-duplex perception transport, alignment, and windowing.
|
|
50
|
+
β β βββ __init__.py
|
|
51
|
+
β β βββ runtime.py # PerceptionRuntime entry point and typed stream orchestration.
|
|
52
|
+
β β βββ stream.py # Multi-consumer streams with independent cursors and backpressure.
|
|
53
|
+
β β βββ timeline.py # Observationβannotation alignment and renderer coordination.
|
|
54
|
+
β β βββ window.py # Ordered multimodal windows for Memory, Persona, Vision, and routers.
|
|
55
|
+
β βββ version.py # Package version metadata.
|
|
56
|
+
βββ pyproject.toml
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## π Workflow
|
|
60
|
+
|
|
61
|
+
External RTC or device adapters first normalize incoming media into AlphaAvatar-owned payloads. `EnvObservation` wraps each payload with identity, source, timestamp, and metadata before publishing it to `PerceptionRuntime`. Persona and other perception modules can attach annotations and produce annotated payload views without overwriting the raw representation. Consumer-specific windows then provide ordered observations to modules such as ENV Memory, Sampled Frame Vision, routers, and future audio or event processors.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
alphaavatar/core/__init__.py,sha256=RgNvd3aQnnX-fj5o7PpCR-rSvGFonqD6s6VN0cn031o,583
|
|
2
|
+
alphaavatar/core/version.py,sha256=E0TQJAIHA9qIzlJCRktdYBfRGAwiCdzN3hoMMmQHYb0,604
|
|
3
|
+
alphaavatar/core/env/__init__.py,sha256=uUj0yTnqWvdZ4nY_0XyAo2UoKUj3CFwdtJBBtfDqvyk,719
|
|
4
|
+
alphaavatar/core/env/annotation.py,sha256=ClRAr6FW3Y_EgRj30_mMYwLJivTqxIen7q-EjbQWOO0,1931
|
|
5
|
+
alphaavatar/core/env/observation.py,sha256=ucIv1L4bYqLt8n4_PZUVTwwiC-ZosbrSOfnLuwz0rk0,5737
|
|
6
|
+
alphaavatar/core/media/__init__.py,sha256=ALPmE_gDulwSfFZxRURfPfzKvw04phimP9bT5izVzfQ,922
|
|
7
|
+
alphaavatar/core/media/formats.py,sha256=M2uc536ome5v8b28Q7GZUJ8WROUvd_KSktENW_j9EhM,1640
|
|
8
|
+
alphaavatar/core/media/payload.py,sha256=YJ4odXWfhfAf9TQY6YKy5Hhu9X6rIUpVCO1ITuQEnNI,3927
|
|
9
|
+
alphaavatar/core/media/video.py,sha256=HkdOyPrexmlJVd6KdPHw3BrviPm9bepUGl8nFElebc8,3989
|
|
10
|
+
alphaavatar/core/perception/__init__.py,sha256=dzgnBuTL7MKplD8c0Q-zCporXYSfgp1JoS18PAuwBSU,1066
|
|
11
|
+
alphaavatar/core/perception/runtime.py,sha256=PR_IGuidxxZnxLPTvVmz2JXXPPmHV4VhYfYtbuEXIyo,4423
|
|
12
|
+
alphaavatar/core/perception/stream.py,sha256=XlQSHBTqBshXFy5O7F6tMotFJ-KB3gHeHlKWSQEFETk,3670
|
|
13
|
+
alphaavatar/core/perception/timeline.py,sha256=aRnop-f-5LbaOpWC4Kck6oXUdsJPmlCymqB5pP_Aask,5462
|
|
14
|
+
alphaavatar/core/perception/window.py,sha256=AGzsl_NfN2x_xpc98QOn4MvtgEe13qrozpoAzuCM_vI,3490
|
|
15
|
+
alpha_avatar_core-0.6.4.dist-info/METADATA,sha256=YK5rEACRedpn3jEJUyJb5oawDlrzdvy8JsKZiyOdn-0,3894
|
|
16
|
+
alpha_avatar_core-0.6.4.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
17
|
+
alpha_avatar_core-0.6.4.dist-info/RECORD,,
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
from .annotation import EnvAnnotation
|
|
15
|
+
from .observation import EnvObservation
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"EnvAnnotation",
|
|
19
|
+
"EnvObservation",
|
|
20
|
+
]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
import uuid
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(slots=True)
|
|
20
|
+
class EnvAnnotation:
|
|
21
|
+
"""
|
|
22
|
+
Lightweight environment annotation attached to an observation or frame.
|
|
23
|
+
|
|
24
|
+
Examples:
|
|
25
|
+
- face boxes from persona.face_stream
|
|
26
|
+
- object boxes from interaction_router
|
|
27
|
+
- gaze/focus regions from avatar vision
|
|
28
|
+
- OCR/screen regions
|
|
29
|
+
- speaker/voice alignment metadata
|
|
30
|
+
|
|
31
|
+
This object must stay LiveKit-independent.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
source: str
|
|
35
|
+
annotation_type: str
|
|
36
|
+
data: dict[str, Any]
|
|
37
|
+
|
|
38
|
+
# Usually one of these two is enough.
|
|
39
|
+
frame_id: str | None = None
|
|
40
|
+
observation_id: str | None = None
|
|
41
|
+
|
|
42
|
+
timestamp: str | None = None
|
|
43
|
+
annotation_id: str = field(default_factory=lambda: str(uuid.uuid4()))
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def target_key(self) -> str | None:
|
|
47
|
+
if self.observation_id:
|
|
48
|
+
return f"observation:{self.observation_id}"
|
|
49
|
+
if self.frame_id:
|
|
50
|
+
return f"frame:{self.frame_id}"
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
def to_dict(self) -> dict[str, Any]:
|
|
54
|
+
return {
|
|
55
|
+
"annotation_id": self.annotation_id,
|
|
56
|
+
"source": self.source,
|
|
57
|
+
"type": self.annotation_type,
|
|
58
|
+
"timestamp": self.timestamp,
|
|
59
|
+
"frame_id": self.frame_id,
|
|
60
|
+
"observation_id": self.observation_id,
|
|
61
|
+
"data": self.data,
|
|
62
|
+
}
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import uuid
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from alphaavatar.core.media import MediaPayload
|
|
21
|
+
|
|
22
|
+
from .annotation import EnvAnnotation
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(slots=True)
|
|
26
|
+
class EnvObservation:
|
|
27
|
+
"""
|
|
28
|
+
Runtime environment observation envelope.
|
|
29
|
+
|
|
30
|
+
payload:
|
|
31
|
+
AlphaAvatar-owned MediaPayload.
|
|
32
|
+
|
|
33
|
+
It must not directly contain:
|
|
34
|
+
- LiveKit rtc.VideoFrame
|
|
35
|
+
- provider-specific content blocks
|
|
36
|
+
- LangChain message objects
|
|
37
|
+
|
|
38
|
+
path:
|
|
39
|
+
Optional persisted evidence path.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
kind: str
|
|
43
|
+
timestamp: str
|
|
44
|
+
source_id: str
|
|
45
|
+
|
|
46
|
+
observation_id: str = field(default_factory=lambda: str(uuid.uuid4()))
|
|
47
|
+
|
|
48
|
+
path: str | None = None
|
|
49
|
+
mime_type: str | None = None
|
|
50
|
+
|
|
51
|
+
payload: MediaPayload | None = field(
|
|
52
|
+
default=None,
|
|
53
|
+
repr=False,
|
|
54
|
+
compare=False,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
58
|
+
annotations: list[EnvAnnotation] = field(default_factory=list)
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def frame_id(self) -> str | None:
|
|
62
|
+
metadata_frame_id = self.metadata.get("frame_id")
|
|
63
|
+
if metadata_frame_id:
|
|
64
|
+
return str(metadata_frame_id)
|
|
65
|
+
|
|
66
|
+
payload_frame_id = getattr(self.payload, "frame_id", None)
|
|
67
|
+
if payload_frame_id:
|
|
68
|
+
return str(payload_frame_id)
|
|
69
|
+
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def has_payload(self) -> bool:
|
|
74
|
+
return self.payload is not None and self.payload.has_any
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def has_persisted_evidence(self) -> bool:
|
|
78
|
+
return bool(self.path)
|
|
79
|
+
|
|
80
|
+
def add_annotation(self, annotation: EnvAnnotation) -> bool:
|
|
81
|
+
"""
|
|
82
|
+
Add annotation once.
|
|
83
|
+
|
|
84
|
+
Returns True when added and False when it already existed.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
for current in self.annotations:
|
|
88
|
+
if current.annotation_id == annotation.annotation_id:
|
|
89
|
+
return False
|
|
90
|
+
|
|
91
|
+
self.annotations.append(annotation)
|
|
92
|
+
return True
|
|
93
|
+
|
|
94
|
+
def clear_payload(self) -> None:
|
|
95
|
+
if self.payload is not None:
|
|
96
|
+
self.payload.clear()
|
|
97
|
+
|
|
98
|
+
self.payload = None
|
|
99
|
+
|
|
100
|
+
def to_evidence_dict(self) -> dict[str, Any]:
|
|
101
|
+
if self.path:
|
|
102
|
+
data: dict[str, Any] = {
|
|
103
|
+
"observation_id": self.observation_id,
|
|
104
|
+
"kind": self.kind,
|
|
105
|
+
"timestamp": self.timestamp,
|
|
106
|
+
"source_id": self.source_id,
|
|
107
|
+
"mime_type": self.mime_type,
|
|
108
|
+
"metadata": self.metadata,
|
|
109
|
+
"annotations": [annotation.to_dict() for annotation in self.annotations],
|
|
110
|
+
"path": self.path,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
return data
|
|
114
|
+
else:
|
|
115
|
+
return {}
|
|
116
|
+
|
|
117
|
+
@classmethod
|
|
118
|
+
def video_frame(
|
|
119
|
+
cls,
|
|
120
|
+
*,
|
|
121
|
+
timestamp: str,
|
|
122
|
+
source_id: str,
|
|
123
|
+
payload: MediaPayload,
|
|
124
|
+
path: str | None = None,
|
|
125
|
+
metadata: dict[str, Any] | None = None,
|
|
126
|
+
annotations: list[EnvAnnotation] | None = None,
|
|
127
|
+
) -> EnvObservation:
|
|
128
|
+
return cls(
|
|
129
|
+
kind="video_frame",
|
|
130
|
+
timestamp=timestamp,
|
|
131
|
+
source_id=source_id,
|
|
132
|
+
path=path,
|
|
133
|
+
mime_type="image/jpeg",
|
|
134
|
+
payload=payload,
|
|
135
|
+
metadata=metadata or {},
|
|
136
|
+
annotations=annotations or [],
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
@classmethod
|
|
140
|
+
def screen_frame(
|
|
141
|
+
cls,
|
|
142
|
+
*,
|
|
143
|
+
timestamp: str,
|
|
144
|
+
source_id: str,
|
|
145
|
+
payload: MediaPayload,
|
|
146
|
+
path: str | None = None,
|
|
147
|
+
metadata: dict[str, Any] | None = None,
|
|
148
|
+
annotations: list[EnvAnnotation] | None = None,
|
|
149
|
+
) -> EnvObservation:
|
|
150
|
+
return cls(
|
|
151
|
+
kind="screen_frame",
|
|
152
|
+
timestamp=timestamp,
|
|
153
|
+
source_id=source_id,
|
|
154
|
+
path=path,
|
|
155
|
+
mime_type="image/jpeg",
|
|
156
|
+
payload=payload,
|
|
157
|
+
metadata=metadata or {},
|
|
158
|
+
annotations=annotations or [],
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
@classmethod
|
|
162
|
+
def video_clip(
|
|
163
|
+
cls,
|
|
164
|
+
*,
|
|
165
|
+
timestamp: str,
|
|
166
|
+
source_id: str,
|
|
167
|
+
payload: MediaPayload | None = None,
|
|
168
|
+
path: str | None = None,
|
|
169
|
+
mime_type: str = "video/mp4",
|
|
170
|
+
metadata: dict[str, Any] | None = None,
|
|
171
|
+
annotations: list[EnvAnnotation] | None = None,
|
|
172
|
+
) -> EnvObservation:
|
|
173
|
+
return cls(
|
|
174
|
+
kind="video_clip",
|
|
175
|
+
timestamp=timestamp,
|
|
176
|
+
source_id=source_id,
|
|
177
|
+
path=path,
|
|
178
|
+
mime_type=mime_type,
|
|
179
|
+
payload=payload,
|
|
180
|
+
metadata=metadata or {},
|
|
181
|
+
annotations=annotations or [],
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
@classmethod
|
|
185
|
+
def audio_segment(
|
|
186
|
+
cls,
|
|
187
|
+
*,
|
|
188
|
+
timestamp: str,
|
|
189
|
+
source_id: str,
|
|
190
|
+
payload: MediaPayload | None = None,
|
|
191
|
+
path: str | None = None,
|
|
192
|
+
mime_type: str = "audio/wav",
|
|
193
|
+
metadata: dict[str, Any] | None = None,
|
|
194
|
+
annotations: list[EnvAnnotation] | None = None,
|
|
195
|
+
) -> EnvObservation:
|
|
196
|
+
return cls(
|
|
197
|
+
kind="audio_segment",
|
|
198
|
+
timestamp=timestamp,
|
|
199
|
+
source_id=source_id,
|
|
200
|
+
path=path,
|
|
201
|
+
mime_type=mime_type,
|
|
202
|
+
payload=payload,
|
|
203
|
+
metadata=metadata or {},
|
|
204
|
+
annotations=annotations or [],
|
|
205
|
+
)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
from .formats import PayloadFormat, PayloadView
|
|
15
|
+
from .payload import MediaPayload, PayloadFormatUnavailable
|
|
16
|
+
from .video import PixelFormat, VideoFrame, VideoFramePayload
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"MediaPayload",
|
|
20
|
+
"PayloadFormat",
|
|
21
|
+
"PayloadFormatUnavailable",
|
|
22
|
+
"PayloadView",
|
|
23
|
+
"PixelFormat",
|
|
24
|
+
"VideoFrame",
|
|
25
|
+
"VideoFramePayload",
|
|
26
|
+
]
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from enum import StrEnum
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class PayloadView(StrEnum):
|
|
20
|
+
"""
|
|
21
|
+
Logical representation view.
|
|
22
|
+
|
|
23
|
+
RAW:
|
|
24
|
+
Original input representation produced by an input adapter.
|
|
25
|
+
|
|
26
|
+
ANNOTATED:
|
|
27
|
+
Representation containing perception annotations, such as face boxes.
|
|
28
|
+
|
|
29
|
+
DERIVED:
|
|
30
|
+
Other derived representations, such as thumbnails or provider-ready blocks.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
RAW = "raw"
|
|
34
|
+
ANNOTATED = "annotated"
|
|
35
|
+
DERIVED = "derived"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class PayloadFormat(StrEnum):
|
|
39
|
+
"""
|
|
40
|
+
AlphaAvatar-owned payload formats.
|
|
41
|
+
|
|
42
|
+
Do not add provider-specific or RTC-specific classes here.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
# Generic video frame owned by AlphaAvatar.
|
|
46
|
+
VIDEO_FRAME = "video.frame"
|
|
47
|
+
|
|
48
|
+
# Encoded visual representations.
|
|
49
|
+
IMAGE_JPEG_BYTES = "image/jpeg.bytes"
|
|
50
|
+
IMAGE_PNG_BYTES = "image/png.bytes"
|
|
51
|
+
|
|
52
|
+
# Reserved for later audio implementation.
|
|
53
|
+
AUDIO_PCM16_BYTES = "audio/pcm16.bytes"
|
|
54
|
+
AUDIO_WAV_BYTES = "audio/wav.bytes"
|
|
55
|
+
|
|
56
|
+
# Reserved for provider adapters.
|
|
57
|
+
PROVIDER_CONTENT_BLOCK = "provider.content_block"
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# Copyright 2026 AlphaAvatar project
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from threading import RLock
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from .formats import PayloadFormat, PayloadView
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class PayloadFormatUnavailable(LookupError):
|
|
23
|
+
"""Requested payload representation is unavailable."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class MediaPayload:
|
|
27
|
+
"""
|
|
28
|
+
Multi-representation runtime media payload.
|
|
29
|
+
|
|
30
|
+
One logical media item can expose several representations:
|
|
31
|
+
|
|
32
|
+
RAW + VIDEO_FRAME
|
|
33
|
+
RAW + IMAGE_JPEG_BYTES
|
|
34
|
+
ANNOTATED + VIDEO_FRAME
|
|
35
|
+
ANNOTATED + IMAGE_JPEG_BYTES
|
|
36
|
+
|
|
37
|
+
Consumers explicitly request the representation they need.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def __init__(
|
|
41
|
+
self,
|
|
42
|
+
*,
|
|
43
|
+
modality: str,
|
|
44
|
+
metadata: dict[str, Any] | None = None,
|
|
45
|
+
) -> None:
|
|
46
|
+
self.modality = modality
|
|
47
|
+
self.metadata = metadata or {}
|
|
48
|
+
|
|
49
|
+
self._representations: dict[
|
|
50
|
+
tuple[PayloadView, PayloadFormat],
|
|
51
|
+
Any,
|
|
52
|
+
] = {}
|
|
53
|
+
|
|
54
|
+
self._lock = RLock()
|
|
55
|
+
self._revision = 0
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def revision(self) -> int:
|
|
59
|
+
with self._lock:
|
|
60
|
+
return self._revision
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def has_any(self) -> bool:
|
|
64
|
+
with self._lock:
|
|
65
|
+
return bool(self._representations)
|
|
66
|
+
|
|
67
|
+
def put(
|
|
68
|
+
self,
|
|
69
|
+
fmt: PayloadFormat,
|
|
70
|
+
value: Any,
|
|
71
|
+
*,
|
|
72
|
+
view: PayloadView = PayloadView.RAW,
|
|
73
|
+
) -> None:
|
|
74
|
+
if value is None:
|
|
75
|
+
raise ValueError(f"Cannot store None payload representation: view={view}, format={fmt}")
|
|
76
|
+
|
|
77
|
+
with self._lock:
|
|
78
|
+
self._representations[(view, fmt)] = value
|
|
79
|
+
self._revision += 1
|
|
80
|
+
|
|
81
|
+
def has(
|
|
82
|
+
self,
|
|
83
|
+
fmt: PayloadFormat,
|
|
84
|
+
*,
|
|
85
|
+
view: PayloadView = PayloadView.RAW,
|
|
86
|
+
fallback_to_raw: bool = True,
|
|
87
|
+
) -> bool:
|
|
88
|
+
with self._lock:
|
|
89
|
+
if (view, fmt) in self._representations:
|
|
90
|
+
return True
|
|
91
|
+
|
|
92
|
+
if fallback_to_raw and view != PayloadView.RAW:
|
|
93
|
+
return (PayloadView.RAW, fmt) in self._representations
|
|
94
|
+
|
|
95
|
+
return False
|
|
96
|
+
|
|
97
|
+
def get(
|
|
98
|
+
self,
|
|
99
|
+
fmt: PayloadFormat,
|
|
100
|
+
*,
|
|
101
|
+
view: PayloadView = PayloadView.RAW,
|
|
102
|
+
fallback_to_raw: bool = True,
|
|
103
|
+
) -> Any:
|
|
104
|
+
with self._lock:
|
|
105
|
+
key = (view, fmt)
|
|
106
|
+
|
|
107
|
+
if key in self._representations:
|
|
108
|
+
return self._representations[key]
|
|
109
|
+
|
|
110
|
+
if fallback_to_raw and view != PayloadView.RAW:
|
|
111
|
+
raw_key = (PayloadView.RAW, fmt)
|
|
112
|
+
if raw_key in self._representations:
|
|
113
|
+
return self._representations[raw_key]
|
|
114
|
+
|
|
115
|
+
raise PayloadFormatUnavailable(
|
|
116
|
+
"Payload representation unavailable: "
|
|
117
|
+
f"modality={self.modality!r}, "
|
|
118
|
+
f"view={view.value!r}, "
|
|
119
|
+
f"format={fmt.value!r}"
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
def remove(
|
|
123
|
+
self,
|
|
124
|
+
fmt: PayloadFormat,
|
|
125
|
+
*,
|
|
126
|
+
view: PayloadView,
|
|
127
|
+
) -> None:
|
|
128
|
+
with self._lock:
|
|
129
|
+
removed = self._representations.pop((view, fmt), None)
|
|
130
|
+
if removed is not None:
|
|
131
|
+
self._revision += 1
|
|
132
|
+
|
|
133
|
+
def clear(self) -> None:
|
|
134
|
+
with self._lock:
|
|
135
|
+
self._representations.clear()
|
|
136
|
+
self._revision += 1
|
|
137
|
+
|
|
138
|
+
def available_representations(
|
|
139
|
+
self,
|
|
140
|
+
) -> tuple[tuple[PayloadView, PayloadFormat], ...]:
|
|
141
|
+
with self._lock:
|
|
142
|
+
return tuple(self._representations.keys())
|