haystack-ml-stack 0.4.12__tar.gz → 0.4.13__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/PKG-INFO +2 -1
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/pyproject.toml +3 -3
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/__init__.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/_kafka.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/_serializers.py +76 -0
- haystack_ml_stack-0.4.13/src/haystack_ml_stack/_version.py +1 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/app.py +12 -1
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/cache.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/dynamo.py +60 -23
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/generated/v1/features_pb2.py +7 -1
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/generated/v1/features_pb2.pyi +30 -0
- haystack_ml_stack-0.4.13/src/haystack_ml_stack/history.py +38 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/settings.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/PKG-INFO +2 -1
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/SOURCES.txt +2 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/requires.txt +1 -0
- haystack_ml_stack-0.4.13/tests/test_history_fetch.py +226 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/tests/test_serializers.py +92 -0
- haystack_ml_stack-0.4.12/src/haystack_ml_stack/_version.py +0 -1
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/README.md +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/setup.cfg +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/exceptions.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/generated/__init__.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/generated/v1/__init__.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/model_store.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/utils.py +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/dependency_links.txt +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/top_level.txt +0 -0
- {haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/tests/test_utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: haystack-ml-stack
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.13
|
|
4
4
|
Summary: Functions related to Haystack ML
|
|
5
5
|
Author-email: Oscar Vega <oscar@haystack.tv>
|
|
6
6
|
License: MIT
|
|
@@ -8,6 +8,7 @@ Requires-Python: >=3.11
|
|
|
8
8
|
Description-Content-Type: text/markdown
|
|
9
9
|
Requires-Dist: protobuf==6.33.2
|
|
10
10
|
Requires-Dist: orjson==3.11.7
|
|
11
|
+
Requires-Dist: numpy>=1.26
|
|
11
12
|
Provides-Extra: server
|
|
12
13
|
Requires-Dist: pydantic==2.5.0; extra == "server"
|
|
13
14
|
Requires-Dist: cachetools==5.5.2; extra == "server"
|
|
@@ -5,13 +5,13 @@ build-backend = "setuptools.build_meta"
|
|
|
5
5
|
|
|
6
6
|
[project]
|
|
7
7
|
name = "haystack-ml-stack"
|
|
8
|
-
version = "0.4.
|
|
8
|
+
version = "0.4.13"
|
|
9
9
|
description = "Functions related to Haystack ML"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
authors = [{ name = "Oscar Vega", email = "oscar@haystack.tv" }]
|
|
12
12
|
requires-python = ">=3.11"
|
|
13
13
|
dependencies = [
|
|
14
|
-
"protobuf==6.33.2", "orjson==3.11.7"
|
|
14
|
+
"protobuf==6.33.2", "orjson==3.11.7", "numpy>=1.26"
|
|
15
15
|
]
|
|
16
16
|
license = { text = "MIT" }
|
|
17
17
|
|
|
@@ -25,4 +25,4 @@ server = [
|
|
|
25
25
|
"pydantic-settings==2.2",
|
|
26
26
|
"newrelic==11.1.0",
|
|
27
27
|
"confluent-kafka==2.13.0"
|
|
28
|
-
]
|
|
28
|
+
]
|
|
File without changes
|
|
File without changes
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
from .generated.v1 import features_pb2 as features_pb2_v1
|
|
2
|
+
from .history import CONTEXT_CODES, STREAM_URL_PREFIX
|
|
2
3
|
from google.protobuf.message import Message
|
|
3
4
|
from google.protobuf.json_format import ParseDict as ProtoParseDict
|
|
5
|
+
import numpy as np
|
|
4
6
|
import typing as _t
|
|
5
7
|
from abc import ABC, abstractmethod
|
|
6
8
|
|
|
@@ -305,6 +307,64 @@ class GlobalChannelsSerializerV1(SimpleSerializer):
|
|
|
305
307
|
return root_msg
|
|
306
308
|
|
|
307
309
|
|
|
310
|
+
class StreamEmbeddingSerializerV1(SimpleSerializer):
|
|
311
|
+
"""The raw 512-coordinate prefix, normalized in float32 then stored as float16."""
|
|
312
|
+
|
|
313
|
+
def __init__(self):
|
|
314
|
+
super().__init__(msg_class=features_pb2_v1.StreamEmbedding)
|
|
315
|
+
|
|
316
|
+
def serialize(self, value) -> bytes:
|
|
317
|
+
return self.build_msg(value).SerializeToString()
|
|
318
|
+
|
|
319
|
+
def build_msg(self, value) -> features_pb2_v1.StreamEmbedding:
|
|
320
|
+
assert value["version"] == 1, "Wrong version given!"
|
|
321
|
+
data = np.asarray(value["data"], dtype=np.float32)
|
|
322
|
+
assert data.shape == (512,), "Expected the first 512 coordinates"
|
|
323
|
+
norm = np.linalg.norm(data)
|
|
324
|
+
if not np.isfinite(norm) or norm <= 0:
|
|
325
|
+
raise ValueError("Expected a finite, nonzero 512-coordinate vector norm")
|
|
326
|
+
message = self.msg_class()
|
|
327
|
+
message.version = 1
|
|
328
|
+
message.data = (data / norm).astype("<f2").tobytes()
|
|
329
|
+
return message
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
class UserHistorySerializerV1(SimpleSerializer):
|
|
333
|
+
"""Pack the SQL's selected states."""
|
|
334
|
+
|
|
335
|
+
def __init__(self):
|
|
336
|
+
super().__init__(msg_class=features_pb2_v1.UserHistory)
|
|
337
|
+
|
|
338
|
+
def serialize(self, value) -> bytes:
|
|
339
|
+
return self.build_msg(value).SerializeToString()
|
|
340
|
+
|
|
341
|
+
def build_msg(self, value) -> features_pb2_v1.UserHistory:
|
|
342
|
+
assert value["version"] == 1, "Wrong version given!"
|
|
343
|
+
data = value["data"]
|
|
344
|
+
columns = ["stream_urls", "event_at_ms", "seconds_watched", "contexts", "browsed"]
|
|
345
|
+
lengths = {len(data[column]) for column in columns}
|
|
346
|
+
assert len(lengths) == 1, "History columns must have equal lengths"
|
|
347
|
+
message = self.msg_class()
|
|
348
|
+
message.version = 1
|
|
349
|
+
message.data.snapshot_at_ms = data["snapshot_at_ms"]
|
|
350
|
+
count = len(data["stream_urls"])
|
|
351
|
+
assert count <= 256, "Expected at most 256 history states"
|
|
352
|
+
flags = bytearray()
|
|
353
|
+
for url, at_ms, context, browsed in zip(
|
|
354
|
+
data["stream_urls"], data["event_at_ms"], data["contexts"], data["browsed"]
|
|
355
|
+
):
|
|
356
|
+
suffix = url.removeprefix(STREAM_URL_PREFIX)
|
|
357
|
+
message.data.stream_ids.append(
|
|
358
|
+
suffix if url.startswith(STREAM_URL_PREFIX) and not suffix.startswith(":")
|
|
359
|
+
else ":" + url
|
|
360
|
+
)
|
|
361
|
+
message.data.event_age_ms.append(data["snapshot_at_ms"] - at_ms)
|
|
362
|
+
flags.append(CONTEXT_CODES.get(context, 1) | (128 if browsed else 0))
|
|
363
|
+
message.data.seconds_watched.extend(data["seconds_watched"])
|
|
364
|
+
message.data.context_browsed = bytes(flags)
|
|
365
|
+
return message
|
|
366
|
+
|
|
367
|
+
|
|
308
368
|
class PassThroughSerializer(Serializer):
|
|
309
369
|
def serialize(self, value):
|
|
310
370
|
return value
|
|
@@ -326,6 +386,8 @@ stream_similarity_scores_serializer_v1 = StreamSimilaritySerializerV1()
|
|
|
326
386
|
global_playlist_stats_serializer_v1 = GlobalPlaylistStatsSerializerV1()
|
|
327
387
|
user_playlist_stats_serializer_v1 = UserPlaylistStatsSerializerV1()
|
|
328
388
|
global_channels_serializer_v1 = GlobalChannelsSerializerV1()
|
|
389
|
+
stream_embedding_serializer_v1 = StreamEmbeddingSerializerV1()
|
|
390
|
+
user_history_serializer_v1 = UserHistorySerializerV1()
|
|
329
391
|
|
|
330
392
|
|
|
331
393
|
class FeatureRegistryId(_t.NamedTuple):
|
|
@@ -468,6 +530,18 @@ global_channels_v1_features: list[FeatureRegistryId] = [
|
|
|
468
530
|
),
|
|
469
531
|
]
|
|
470
532
|
|
|
533
|
+
gemini_title_transcript_512_v1_features: list[FeatureRegistryId] = [
|
|
534
|
+
FeatureRegistryId(
|
|
535
|
+
entity_type="STREAM",
|
|
536
|
+
feature_id="EMBEDDING#GEMINI_2_TITLE_TRANSCRIPT#512",
|
|
537
|
+
version="v1",
|
|
538
|
+
),
|
|
539
|
+
]
|
|
540
|
+
|
|
541
|
+
user_history_v1_features: list[FeatureRegistryId] = [
|
|
542
|
+
FeatureRegistryId(entity_type="USER", feature_id="HISTORY#5M#256", version="v1"),
|
|
543
|
+
]
|
|
544
|
+
|
|
471
545
|
features_serializer_tuples: list[tuple[list[FeatureRegistryId], Serializer]] = [
|
|
472
546
|
(stream_pwatched_v0_features, stream_pwatched_serializer_v0),
|
|
473
547
|
(stream_pwatched_v1_features, stream_pwatched_serializer_v1),
|
|
@@ -494,6 +568,8 @@ features_serializer_tuples: list[tuple[list[FeatureRegistryId], Serializer]] = [
|
|
|
494
568
|
global_channels_v1_features,
|
|
495
569
|
global_channels_serializer_v1,
|
|
496
570
|
),
|
|
571
|
+
(gemini_title_transcript_512_v1_features, stream_embedding_serializer_v1),
|
|
572
|
+
(user_history_v1_features, user_history_serializer_v1),
|
|
497
573
|
]
|
|
498
574
|
|
|
499
575
|
SerializerRegistry: dict[FeatureRegistryId, Serializer] = {
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.4.13"
|
|
@@ -70,9 +70,14 @@ async def load_model(state, cfg: Settings) -> None:
|
|
|
70
70
|
(entity_type, feature_id)
|
|
71
71
|
for entity_type, feature_id, _ in SerializerRegistry.keys()
|
|
72
72
|
)
|
|
73
|
+
# Optional {"user_feature": ..., "stream_features": [...]}: the streams
|
|
74
|
+
# in that history record get these stream features.
|
|
75
|
+
history = state["model"].get("history")
|
|
76
|
+
history_features = history["stream_features"] if history else []
|
|
73
77
|
all_features = set(
|
|
74
78
|
[("STREAM", feature_name) for feature_name in state["stream_features"]]
|
|
75
79
|
+ [("USER", feature_name) for feature_name in state["user_features"]]
|
|
80
|
+
+ [("STREAM", feature_name) for feature_name in history_features]
|
|
76
81
|
)
|
|
77
82
|
invalid_features = all_features.difference(valid_features)
|
|
78
83
|
if invalid_features:
|
|
@@ -230,6 +235,8 @@ def create_app(
|
|
|
230
235
|
model = state["model"]
|
|
231
236
|
stream_features = model.get("stream_features", []) or []
|
|
232
237
|
user_features = model.get("user_features", []) or []
|
|
238
|
+
history = model.get("history")
|
|
239
|
+
history_streams: List[Dict[str, Any]] = []
|
|
233
240
|
retrieval_meta = FeatureRetrievalMeta(
|
|
234
241
|
cache_misses=0,
|
|
235
242
|
stream_cache_misses=0,
|
|
@@ -240,7 +247,7 @@ def create_app(
|
|
|
240
247
|
dynamo_ms=0,
|
|
241
248
|
parsing_ms=0,
|
|
242
249
|
)
|
|
243
|
-
if stream_features:
|
|
250
|
+
if stream_features or user_features:
|
|
244
251
|
try:
|
|
245
252
|
retrieval_meta = await set_all_features(
|
|
246
253
|
dynamo_client=state["dynamo_client"],
|
|
@@ -252,6 +259,8 @@ def create_app(
|
|
|
252
259
|
user_features_cache=user_features_cache,
|
|
253
260
|
features_table=cfg.features_table,
|
|
254
261
|
cache_sep=cfg.cache_separator,
|
|
262
|
+
history=history,
|
|
263
|
+
history_streams=history_streams,
|
|
255
264
|
)
|
|
256
265
|
except exceptions.InvalidFeaturesException as e:
|
|
257
266
|
logger.error(
|
|
@@ -280,6 +289,8 @@ def create_app(
|
|
|
280
289
|
try:
|
|
281
290
|
preprocess_start = time.perf_counter_ns()
|
|
282
291
|
model["params"]["query_params"] = query_params
|
|
292
|
+
# Outside `user`, so that the history vectors are not logged.
|
|
293
|
+
model["params"]["history_streams"] = history_streams
|
|
283
294
|
model_input = model["preprocess"](
|
|
284
295
|
user,
|
|
285
296
|
streams,
|
|
File without changes
|
|
@@ -9,6 +9,7 @@ from boto3.dynamodb.types import TypeDeserializer
|
|
|
9
9
|
|
|
10
10
|
from . import exceptions
|
|
11
11
|
from ._serializers import FeatureRegistryId, SerializerRegistry
|
|
12
|
+
from .history import history_stream_urls
|
|
12
13
|
from .utils import _complete_features_for_channels, DEFAULT_CHANNELS
|
|
13
14
|
|
|
14
15
|
logger = logging.getLogger(__name__)
|
|
@@ -106,7 +107,13 @@ async def set_all_features(
|
|
|
106
107
|
features_table: str,
|
|
107
108
|
cache_sep: str,
|
|
108
109
|
dynamo_client,
|
|
110
|
+
history: Dict[str, Any] | None = None,
|
|
111
|
+
history_streams: List[Dict[str, Any]] | None = None,
|
|
109
112
|
) -> FeatureRetrievalMeta:
|
|
113
|
+
"""Fetch features in parallel, filling history_streams when history is configured.
|
|
114
|
+
|
|
115
|
+
Cached history vectors join the primary fetch; cold history uses a second stage.
|
|
116
|
+
"""
|
|
110
117
|
time_start = time.perf_counter_ns()
|
|
111
118
|
if not streams or (not stream_features and not user_features):
|
|
112
119
|
return FeatureRetrievalMeta(
|
|
@@ -119,24 +126,13 @@ async def set_all_features(
|
|
|
119
126
|
dynamo_ms=0,
|
|
120
127
|
parsing_ms=0,
|
|
121
128
|
)
|
|
122
|
-
cache_miss: Dict[str, Dict[str, Any]] = {}
|
|
123
|
-
|
|
129
|
+
cache_miss: Dict[str, List[Dict[str, Any]]] = {}
|
|
130
|
+
history_features = history["stream_features"] if history else []
|
|
131
|
+
if history_streams is None:
|
|
132
|
+
history_streams = []
|
|
133
|
+
all_feature_keys = [*stream_features, *user_features, *history_features]
|
|
124
134
|
cache_delay_obj: dict[str, float] = {f: 0 for f in all_feature_keys}
|
|
125
135
|
now = datetime.datetime.utcnow()
|
|
126
|
-
for f in stream_features:
|
|
127
|
-
for s in streams:
|
|
128
|
-
cache_miss, cache_delay_obj = _check_cache(
|
|
129
|
-
obj=s,
|
|
130
|
-
id_type="STREAM",
|
|
131
|
-
id_key=s["streamUrl"],
|
|
132
|
-
feature_key=f,
|
|
133
|
-
cache_sep=cache_sep,
|
|
134
|
-
features_cache=stream_features_cache,
|
|
135
|
-
cache_miss=cache_miss,
|
|
136
|
-
cache_delay=cache_delay_obj,
|
|
137
|
-
now=now,
|
|
138
|
-
)
|
|
139
|
-
stream_cache_misses = len(cache_miss)
|
|
140
136
|
for f in user_features:
|
|
141
137
|
cache_miss, cache_delay_obj = _check_cache(
|
|
142
138
|
obj=user,
|
|
@@ -149,7 +145,26 @@ async def set_all_features(
|
|
|
149
145
|
cache_delay=cache_delay_obj,
|
|
150
146
|
now=now,
|
|
151
147
|
)
|
|
152
|
-
user_cache_misses = len(cache_miss)
|
|
148
|
+
user_cache_misses = len(cache_miss)
|
|
149
|
+
cached_history = user.get(history["user_feature"]) if history else None
|
|
150
|
+
if history:
|
|
151
|
+
history_streams[:] = [{"streamUrl": url} for url in history_stream_urls(cached_history)]
|
|
152
|
+
# A cached history reveals its vector keys before the first parallel fetch.
|
|
153
|
+
for targets, features in ((streams, stream_features), (history_streams, history_features)):
|
|
154
|
+
for f in features:
|
|
155
|
+
for s in targets:
|
|
156
|
+
cache_miss, cache_delay_obj = _check_cache(
|
|
157
|
+
obj=s,
|
|
158
|
+
id_type="STREAM",
|
|
159
|
+
id_key=s["streamUrl"],
|
|
160
|
+
feature_key=f,
|
|
161
|
+
cache_sep=cache_sep,
|
|
162
|
+
features_cache=stream_features_cache,
|
|
163
|
+
cache_miss=cache_miss,
|
|
164
|
+
cache_delay=cache_delay_obj,
|
|
165
|
+
now=now,
|
|
166
|
+
)
|
|
167
|
+
stream_cache_misses = len(cache_miss) - user_cache_misses
|
|
153
168
|
valid_cache_delays = list(v for v in cache_delay_obj.values() if v > 0)
|
|
154
169
|
cache_delay = min(valid_cache_delays) if valid_cache_delays else 0
|
|
155
170
|
if not cache_miss:
|
|
@@ -198,7 +213,7 @@ async def set_all_features(
|
|
|
198
213
|
updated_keys = set()
|
|
199
214
|
for item in items:
|
|
200
215
|
full_id = item["pk"]["S"]
|
|
201
|
-
id_type, id_key = full_id.split("#")
|
|
216
|
+
id_type, id_key = full_id.split("#", 1)
|
|
202
217
|
feature_name = item["sk"]["S"]
|
|
203
218
|
if id_type == "STREAM":
|
|
204
219
|
cache_to_use = stream_features_cache
|
|
@@ -242,7 +257,8 @@ async def set_all_features(
|
|
|
242
257
|
}
|
|
243
258
|
|
|
244
259
|
if cache_key in cache_miss:
|
|
245
|
-
cache_miss[cache_key]
|
|
260
|
+
for obj in cache_miss[cache_key]:
|
|
261
|
+
obj[feature_name] = value
|
|
246
262
|
updated_keys.add(cache_key)
|
|
247
263
|
parsing_end = time.perf_counter_ns()
|
|
248
264
|
# Save keys that were not found in DynamoDB with None value
|
|
@@ -258,7 +274,7 @@ async def set_all_features(
|
|
|
258
274
|
"cache_ttl_in_seconds": 6 * 3600,
|
|
259
275
|
}
|
|
260
276
|
end_time = time.perf_counter_ns()
|
|
261
|
-
|
|
277
|
+
meta = FeatureRetrievalMeta(
|
|
262
278
|
cache_misses=user_cache_misses + stream_cache_misses,
|
|
263
279
|
user_cache_misses=user_cache_misses,
|
|
264
280
|
stream_cache_misses=stream_cache_misses,
|
|
@@ -268,6 +284,26 @@ async def set_all_features(
|
|
|
268
284
|
dynamo_ms=_perf_counter_ns_delta_in_ms(dynamo_start, dynamo_end),
|
|
269
285
|
parsing_ms=_perf_counter_ns_delta_in_ms(dynamo_end, parsing_end),
|
|
270
286
|
)
|
|
287
|
+
if history and cached_history is None:
|
|
288
|
+
record = user.get(history["user_feature"])
|
|
289
|
+
history_streams[:] = [{"streamUrl": url} for url in history_stream_urls(record)]
|
|
290
|
+
# An uncached history needs a second stage; its batches are still parallel.
|
|
291
|
+
history_meta = await set_all_features(
|
|
292
|
+
user=user, streams=history_streams, stream_features=history_features,
|
|
293
|
+
user_features=[], stream_features_cache=stream_features_cache,
|
|
294
|
+
user_features_cache=user_features_cache, features_table=features_table,
|
|
295
|
+
cache_sep=cache_sep, dynamo_client=dynamo_client,
|
|
296
|
+
)
|
|
297
|
+
meta = meta._replace(
|
|
298
|
+
cache_misses=meta.cache_misses + history_meta.cache_misses,
|
|
299
|
+
stream_cache_misses=meta.stream_cache_misses + history_meta.stream_cache_misses,
|
|
300
|
+
retrieval_ms=_perf_counter_ns_delta_in_ms(time_start, time.perf_counter_ns()),
|
|
301
|
+
success=meta.success and history_meta.success,
|
|
302
|
+
cache_delay_minutes=max(meta.cache_delay_minutes, history_meta.cache_delay_minutes),
|
|
303
|
+
dynamo_ms=meta.dynamo_ms + history_meta.dynamo_ms,
|
|
304
|
+
parsing_ms=meta.parsing_ms + history_meta.parsing_ms,
|
|
305
|
+
)
|
|
306
|
+
return meta
|
|
271
307
|
|
|
272
308
|
|
|
273
309
|
_MOBILE_OS = {"ios", "android", "iphone", "galaxy"}
|
|
@@ -296,7 +332,7 @@ async def create_channel_candidates(
|
|
|
296
332
|
os_cat = _get_os_cat(user.get("clientOs", ""))
|
|
297
333
|
|
|
298
334
|
global_holder: Dict[str, Any] = {}
|
|
299
|
-
cache_miss: Dict[str, Dict[str, Any]] = {}
|
|
335
|
+
cache_miss: Dict[str, List[Dict[str, Any]]] = {}
|
|
300
336
|
all_feature_keys = [*global_features, *user_features]
|
|
301
337
|
cache_delay_obj: dict[str, float] = {f: 0 for f in all_feature_keys}
|
|
302
338
|
now = datetime.datetime.utcnow()
|
|
@@ -437,7 +473,8 @@ async def create_channel_candidates(
|
|
|
437
473
|
}
|
|
438
474
|
|
|
439
475
|
if cache_key in cache_miss:
|
|
440
|
-
cache_miss[cache_key]
|
|
476
|
+
for obj in cache_miss[cache_key]:
|
|
477
|
+
obj[feature_name] = value
|
|
441
478
|
updated_keys.add(cache_key)
|
|
442
479
|
|
|
443
480
|
parsing_end = time.perf_counter_ns()
|
|
@@ -594,7 +631,7 @@ def _check_cache(
|
|
|
594
631
|
(now - cached["inserted_at"]).total_seconds(),
|
|
595
632
|
)
|
|
596
633
|
else:
|
|
597
|
-
cache_miss[
|
|
634
|
+
cache_miss.setdefault(key, []).append(obj)
|
|
598
635
|
return cache_miss, cache_delay
|
|
599
636
|
|
|
600
637
|
|
|
@@ -24,7 +24,7 @@ _sym_db = _symbol_database.Default()
|
|
|
24
24
|
|
|
25
25
|
|
|
26
26
|
|
|
27
|
-
DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(b'\n\x0e\x66\x65\x61tures.proto\x12\x1ahaystack_ml_stack.features\"7\n\x12\x45ntryContextCounts\x12\x10\n\x08\x61ttempts\x18\x01 \x01(\x05\x12\x0f\n\x07watched\x18\x02 \x01(\x05\"_\n\x0cSelectCounts\x12\x15\n\rtotal_selects\x18\x01 \x01(\x05\x12!\n\x19total_selects_and_watched\x18\x02 \x01(\x05\x12\x15\n\rtotal_browsed\x18\x03 \x01(\x05\"\xf3\x02\n\x14\x45ntryContextPWatched\x12@\n\x08\x61utoplay\x18\x01 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12\x41\n\tsel_thumb\x18\x02 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12\x43\n\x0b\x63hoose_next\x18\x03 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12@\n\x08\x63h_swtch\x18\x04 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12O\n\x17launch_first_in_session\x18\x05 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\"\x85\x02\n\x0fPositionPSelect\x12;\n\tfirst_pos\x18\x01 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\x12<\n\nsecond_pos\x18\x02 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\x12;\n\tthird_pos\x18\x03 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\x12:\n\x08rest_pos\x18\x04 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\"\xa9\x01\n\x1f\x42rowsedDebiasedPositionPSelects\x12\x44\n\x0fup_to_4_browsed\x18\x01 \x01(\x0b\x32+.haystack_ml_stack.features.PositionPSelect\x12@\n\x0b\x61ll_browsed\x18\x02 \x01(\x0b\x32+.haystack_ml_stack.features.PositionPSelect\"\xb8\x01\n\x16PlaylistStatsForGlobal\x12\x15\n\rwatched_count\x18\x01 \x01(\x05\x12\x19\n\x11not_watched_count\x18\x02 \x01(\x05\x12\x1b\n\x13\x63\x61pped_watched_secs\x18\x03 \x01(\x02\x12\x1f\n\x17\x63\x61pped_not_watched_secs\x18\x04 \x01(\x02\x12\x14\n\x0cwatched_secs\x18\x05 \x01(\x02\x12\x18\n\x10not_watched_secs\x18\x06 \x01(\x02\"\xc5\x01\n\x14PlaylistStatsForUser\x12\x12\n\ntotal_days\x18\x01 \x01(\x05\x12\x12\n\nstart_days\x18\x02 \x01(\x05\x12\x13\n\x0b\x61\x63tive_days\x18\x03 \x01(\x05\x12\x15\n\rtotal_watched\x18\x04 \x01(\x02\x12\x1c\n\x14\x63\x61pped_total_watched\x18\x05 \x01(\x02\x12 \n\x18\x63\x61pped_total_log_watched\x18\x06 \x01(\x02\x12\x19\n\x11\x66irst_active_date\x18\x07 \x01(\t\"^\n\x07\x43hannel\x12\x0c\n\x04name\x18\x01 \x01(\t\x12\x16\n\x0e\x63\x61tegory_group\x18\x02 \x01(\t\x12\x12\n\nstart_date\x18\x03 \x01(\x05\x12\x19\n\x11required_features\x18\x04 \x03(\t\"k\n\rStreamPSelect\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12I\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32;.haystack_ml_stack.features.BrowsedDebiasedPositionPSelects\"a\n\x0eStreamPWatched\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12>\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.EntryContextPWatched\"_\n\x0cUserPWatched\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12>\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.EntryContextPWatched\"\xda\x01\n\x19UserPersonalizingPWatched\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12M\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32?.haystack_ml_stack.features.UserPersonalizingPWatched.DataEntry\x1a]\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12?\n\x05value\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.EntryContextPWatched:\x02\x38\x01\"i\n\x0bUserPSelect\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12I\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32;.haystack_ml_stack.features.BrowsedDebiasedPositionPSelects\"\xe3\x01\n\x18UserPersonalizingPSelect\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12L\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32>.haystack_ml_stack.features.UserPersonalizingPSelect.DataEntry\x1ah\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12J\n\x05value\x18\x02 \x01(\x0b\x32;.haystack_ml_stack.features.BrowsedDebiasedPositionPSelects:\x02\x38\x01\"\xa2\x01\n\x16StreamSimilarityScores\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12J\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32<.haystack_ml_stack.features.StreamSimilarityScores.DataEntry\x1a+\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12\r\n\x05value\x18\x02 \x01(\x01:\x02\x38\x01\"\xd0\x01\n\x13GlobalPlaylistStats\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12G\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32\x39.haystack_ml_stack.features.GlobalPlaylistStats.DataEntry\x1a_\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12\x41\n\x05value\x18\x02 \x01(\x0b\x32\x32.haystack_ml_stack.features.PlaylistStatsForGlobal:\x02\x38\x01\"\xca\x01\n\x11UserPlaylistStats\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12\x45\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32\x37.haystack_ml_stack.features.UserPlaylistStats.DataEntry\x1a]\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12?\n\x05value\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.PlaylistStatsForUser:\x02\x38\x01\"T\n\x0eGlobalChannels\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12\x31\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32#.haystack_ml_stack.features.
|
|
27
|
+
DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(b'\n\x0e\x66\x65\x61tures.proto\x12\x1ahaystack_ml_stack.features\"7\n\x12\x45ntryContextCounts\x12\x10\n\x08\x61ttempts\x18\x01 \x01(\x05\x12\x0f\n\x07watched\x18\x02 \x01(\x05\"_\n\x0cSelectCounts\x12\x15\n\rtotal_selects\x18\x01 \x01(\x05\x12!\n\x19total_selects_and_watched\x18\x02 \x01(\x05\x12\x15\n\rtotal_browsed\x18\x03 \x01(\x05\"\xf3\x02\n\x14\x45ntryContextPWatched\x12@\n\x08\x61utoplay\x18\x01 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12\x41\n\tsel_thumb\x18\x02 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12\x43\n\x0b\x63hoose_next\x18\x03 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12@\n\x08\x63h_swtch\x18\x04 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\x12O\n\x17launch_first_in_session\x18\x05 \x01(\x0b\x32..haystack_ml_stack.features.EntryContextCounts\"\x85\x02\n\x0fPositionPSelect\x12;\n\tfirst_pos\x18\x01 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\x12<\n\nsecond_pos\x18\x02 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\x12;\n\tthird_pos\x18\x03 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\x12:\n\x08rest_pos\x18\x04 \x01(\x0b\x32(.haystack_ml_stack.features.SelectCounts\"\xa9\x01\n\x1f\x42rowsedDebiasedPositionPSelects\x12\x44\n\x0fup_to_4_browsed\x18\x01 \x01(\x0b\x32+.haystack_ml_stack.features.PositionPSelect\x12@\n\x0b\x61ll_browsed\x18\x02 \x01(\x0b\x32+.haystack_ml_stack.features.PositionPSelect\"\xb8\x01\n\x16PlaylistStatsForGlobal\x12\x15\n\rwatched_count\x18\x01 \x01(\x05\x12\x19\n\x11not_watched_count\x18\x02 \x01(\x05\x12\x1b\n\x13\x63\x61pped_watched_secs\x18\x03 \x01(\x02\x12\x1f\n\x17\x63\x61pped_not_watched_secs\x18\x04 \x01(\x02\x12\x14\n\x0cwatched_secs\x18\x05 \x01(\x02\x12\x18\n\x10not_watched_secs\x18\x06 \x01(\x02\"\xc5\x01\n\x14PlaylistStatsForUser\x12\x12\n\ntotal_days\x18\x01 \x01(\x05\x12\x12\n\nstart_days\x18\x02 \x01(\x05\x12\x13\n\x0b\x61\x63tive_days\x18\x03 \x01(\x05\x12\x15\n\rtotal_watched\x18\x04 \x01(\x02\x12\x1c\n\x14\x63\x61pped_total_watched\x18\x05 \x01(\x02\x12 \n\x18\x63\x61pped_total_log_watched\x18\x06 \x01(\x02\x12\x19\n\x11\x66irst_active_date\x18\x07 \x01(\t\"^\n\x07\x43hannel\x12\x0c\n\x04name\x18\x01 \x01(\t\x12\x16\n\x0e\x63\x61tegory_group\x18\x02 \x01(\t\x12\x12\n\nstart_date\x18\x03 \x01(\x05\x12\x19\n\x11required_features\x18\x04 \x03(\t\"k\n\rStreamPSelect\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12I\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32;.haystack_ml_stack.features.BrowsedDebiasedPositionPSelects\"a\n\x0eStreamPWatched\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12>\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.EntryContextPWatched\"_\n\x0cUserPWatched\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12>\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.EntryContextPWatched\"\xda\x01\n\x19UserPersonalizingPWatched\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12M\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32?.haystack_ml_stack.features.UserPersonalizingPWatched.DataEntry\x1a]\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12?\n\x05value\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.EntryContextPWatched:\x02\x38\x01\"i\n\x0bUserPSelect\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12I\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32;.haystack_ml_stack.features.BrowsedDebiasedPositionPSelects\"\xe3\x01\n\x18UserPersonalizingPSelect\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12L\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32>.haystack_ml_stack.features.UserPersonalizingPSelect.DataEntry\x1ah\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12J\n\x05value\x18\x02 \x01(\x0b\x32;.haystack_ml_stack.features.BrowsedDebiasedPositionPSelects:\x02\x38\x01\"\xa2\x01\n\x16StreamSimilarityScores\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12J\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32<.haystack_ml_stack.features.StreamSimilarityScores.DataEntry\x1a+\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12\r\n\x05value\x18\x02 \x01(\x01:\x02\x38\x01\"\xd0\x01\n\x13GlobalPlaylistStats\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12G\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32\x39.haystack_ml_stack.features.GlobalPlaylistStats.DataEntry\x1a_\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12\x41\n\x05value\x18\x02 \x01(\x0b\x32\x32.haystack_ml_stack.features.PlaylistStatsForGlobal:\x02\x38\x01\"\xca\x01\n\x11UserPlaylistStats\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12\x45\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32\x37.haystack_ml_stack.features.UserPlaylistStats.DataEntry\x1a]\n\tDataEntry\x12\x0b\n\x03key\x18\x01 \x01(\t\x12?\n\x05value\x18\x02 \x01(\x0b\x32\x30.haystack_ml_stack.features.PlaylistStatsForUser:\x02\x38\x01\"T\n\x0eGlobalChannels\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12\x31\n\x04\x64\x61ta\x18\x02 \x03(\x0b\x32#.haystack_ml_stack.features.Channel\"0\n\x0fStreamEmbedding\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12\x0c\n\x04\x64\x61ta\x18\x02 \x01(\x0c\"\x83\x01\n\rHistoryStates\x12\x12\n\nstream_ids\x18\x01 \x03(\t\x12\x14\n\x0c\x65vent_age_ms\x18\x02 \x03(\x04\x12\x17\n\x0fseconds_watched\x18\x03 \x03(\x02\x12\x17\n\x0f\x63ontext_browsed\x18\x04 \x01(\x0c\x12\x16\n\x0esnapshot_at_ms\x18\x06 \x01(\x03\"]\n\x0bUserHistory\x12\x0f\n\x07version\x18\x01 \x01(\x05\x12\x37\n\x04\x64\x61ta\x18\x02 \x01(\x0b\x32).haystack_ml_stack.features.HistoryStatesJ\x04\x08\x03\x10\x04\x62\x06proto3')
|
|
28
28
|
|
|
29
29
|
_globals = globals()
|
|
30
30
|
_builder.BuildMessageAndEnumDescriptors(DESCRIPTOR, _globals)
|
|
@@ -87,4 +87,10 @@ if not _descriptor._USE_C_DESCRIPTORS:
|
|
|
87
87
|
_globals['_USERPLAYLISTSTATS_DATAENTRY']._serialized_end=2935
|
|
88
88
|
_globals['_GLOBALCHANNELS']._serialized_start=2937
|
|
89
89
|
_globals['_GLOBALCHANNELS']._serialized_end=3021
|
|
90
|
+
_globals['_STREAMEMBEDDING']._serialized_start=3023
|
|
91
|
+
_globals['_STREAMEMBEDDING']._serialized_end=3071
|
|
92
|
+
_globals['_HISTORYSTATES']._serialized_start=3074
|
|
93
|
+
_globals['_HISTORYSTATES']._serialized_end=3205
|
|
94
|
+
_globals['_USERHISTORY']._serialized_start=3207
|
|
95
|
+
_globals['_USERHISTORY']._serialized_end=3300
|
|
90
96
|
# @@protoc_insertion_point(module_scope)
|
|
@@ -218,3 +218,33 @@ class GlobalChannels(_message.Message):
|
|
|
218
218
|
version: int
|
|
219
219
|
data: _containers.RepeatedCompositeFieldContainer[Channel]
|
|
220
220
|
def __init__(self, version: _Optional[int] = ..., data: _Optional[_Iterable[_Union[Channel, _Mapping]]] = ...) -> None: ...
|
|
221
|
+
|
|
222
|
+
class StreamEmbedding(_message.Message):
|
|
223
|
+
__slots__ = ()
|
|
224
|
+
VERSION_FIELD_NUMBER: _ClassVar[int]
|
|
225
|
+
DATA_FIELD_NUMBER: _ClassVar[int]
|
|
226
|
+
version: int
|
|
227
|
+
data: bytes
|
|
228
|
+
def __init__(self, version: _Optional[int] = ..., data: _Optional[bytes] = ...) -> None: ...
|
|
229
|
+
|
|
230
|
+
class HistoryStates(_message.Message):
|
|
231
|
+
__slots__ = ()
|
|
232
|
+
STREAM_IDS_FIELD_NUMBER: _ClassVar[int]
|
|
233
|
+
EVENT_AGE_MS_FIELD_NUMBER: _ClassVar[int]
|
|
234
|
+
SECONDS_WATCHED_FIELD_NUMBER: _ClassVar[int]
|
|
235
|
+
CONTEXT_BROWSED_FIELD_NUMBER: _ClassVar[int]
|
|
236
|
+
SNAPSHOT_AT_MS_FIELD_NUMBER: _ClassVar[int]
|
|
237
|
+
stream_ids: _containers.RepeatedScalarFieldContainer[str]
|
|
238
|
+
event_age_ms: _containers.RepeatedScalarFieldContainer[int]
|
|
239
|
+
seconds_watched: _containers.RepeatedScalarFieldContainer[float]
|
|
240
|
+
context_browsed: bytes
|
|
241
|
+
snapshot_at_ms: int
|
|
242
|
+
def __init__(self, stream_ids: _Optional[_Iterable[str]] = ..., event_age_ms: _Optional[_Iterable[int]] = ..., seconds_watched: _Optional[_Iterable[float]] = ..., context_browsed: _Optional[bytes] = ..., snapshot_at_ms: _Optional[int] = ...) -> None: ...
|
|
243
|
+
|
|
244
|
+
class UserHistory(_message.Message):
|
|
245
|
+
__slots__ = ()
|
|
246
|
+
VERSION_FIELD_NUMBER: _ClassVar[int]
|
|
247
|
+
DATA_FIELD_NUMBER: _ClassVar[int]
|
|
248
|
+
version: int
|
|
249
|
+
data: HistoryStates
|
|
250
|
+
def __init__(self, version: _Optional[int] = ..., data: _Optional[_Union[HistoryStates, _Mapping]] = ...) -> None: ...
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Compact history v1: shared URL and state decoding for serving and replay."""
|
|
2
|
+
|
|
3
|
+
STREAM_URL_PREFIX = "http://haystack.tv/id/"
|
|
4
|
+
# Wire codes are append-only. Missing and unsupported contexts are distinct.
|
|
5
|
+
CONTEXTS = ("", "__unsupported__", "autoplay", "sel thumb", "pres nxt",
|
|
6
|
+
"swpe nxt", "ch swtch", "launch", "pl refrsh")
|
|
7
|
+
CONTEXT_CODES = {context: code for code, context in enumerate(CONTEXTS)}
|
|
8
|
+
CONTEXT_CODES.update({"null": 0, "__missing__": 0})
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def history_stream_urls(record):
|
|
12
|
+
"""Rebuild exact vector keys; ':' explicitly escapes a full URL."""
|
|
13
|
+
if record is None:
|
|
14
|
+
return []
|
|
15
|
+
return [value[1:] if value.startswith(":") else STREAM_URL_PREFIX + value
|
|
16
|
+
for value in record.data.stream_ids]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def history_entries(record):
|
|
20
|
+
"""Yield URL, timestamp ms, seconds, context, browsed, watched (over 25 s, not browsed)."""
|
|
21
|
+
if record is None:
|
|
22
|
+
return
|
|
23
|
+
data = record.data
|
|
24
|
+
count = len(data.stream_ids)
|
|
25
|
+
if not (record.version == 1 and count <= 256
|
|
26
|
+
and len(data.event_age_ms) == len(data.seconds_watched)
|
|
27
|
+
== len(data.context_browsed) == count):
|
|
28
|
+
raise ValueError("Invalid compact history version or column lengths")
|
|
29
|
+
for url, age, seconds, flags in zip(
|
|
30
|
+
history_stream_urls(record), data.event_age_ms,
|
|
31
|
+
data.seconds_watched, data.context_browsed,
|
|
32
|
+
):
|
|
33
|
+
context = flags & 127
|
|
34
|
+
if context >= len(CONTEXTS):
|
|
35
|
+
raise ValueError("Unknown compact history context code")
|
|
36
|
+
browsed = bool(flags & 128)
|
|
37
|
+
yield (url, data.snapshot_at_ms - age, seconds, CONTEXTS[context],
|
|
38
|
+
browsed, seconds > 25 and not browsed)
|
|
File without changes
|
{haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: haystack-ml-stack
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.13
|
|
4
4
|
Summary: Functions related to Haystack ML
|
|
5
5
|
Author-email: Oscar Vega <oscar@haystack.tv>
|
|
6
6
|
License: MIT
|
|
@@ -8,6 +8,7 @@ Requires-Python: >=3.11
|
|
|
8
8
|
Description-Content-Type: text/markdown
|
|
9
9
|
Requires-Dist: protobuf==6.33.2
|
|
10
10
|
Requires-Dist: orjson==3.11.7
|
|
11
|
+
Requires-Dist: numpy>=1.26
|
|
11
12
|
Provides-Extra: server
|
|
12
13
|
Requires-Dist: pydantic==2.5.0; extra == "server"
|
|
13
14
|
Requires-Dist: cachetools==5.5.2; extra == "server"
|
{haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/SOURCES.txt
RENAMED
|
@@ -8,6 +8,7 @@ src/haystack_ml_stack/app.py
|
|
|
8
8
|
src/haystack_ml_stack/cache.py
|
|
9
9
|
src/haystack_ml_stack/dynamo.py
|
|
10
10
|
src/haystack_ml_stack/exceptions.py
|
|
11
|
+
src/haystack_ml_stack/history.py
|
|
11
12
|
src/haystack_ml_stack/model_store.py
|
|
12
13
|
src/haystack_ml_stack/settings.py
|
|
13
14
|
src/haystack_ml_stack/utils.py
|
|
@@ -20,5 +21,6 @@ src/haystack_ml_stack/generated/__init__.py
|
|
|
20
21
|
src/haystack_ml_stack/generated/v1/__init__.py
|
|
21
22
|
src/haystack_ml_stack/generated/v1/features_pb2.py
|
|
22
23
|
src/haystack_ml_stack/generated/v1/features_pb2.pyi
|
|
24
|
+
tests/test_history_fetch.py
|
|
23
25
|
tests/test_serializers.py
|
|
24
26
|
tests/test_utils.py
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import base64
|
|
3
|
+
import datetime
|
|
4
|
+
import json
|
|
5
|
+
|
|
6
|
+
import aiobotocore.session
|
|
7
|
+
import newrelic.agent
|
|
8
|
+
import pytest
|
|
9
|
+
from fastapi import BackgroundTasks, Response
|
|
10
|
+
from starlette.requests import Request
|
|
11
|
+
|
|
12
|
+
from haystack_ml_stack import FeatureRegistryId, SerializerRegistry
|
|
13
|
+
from haystack_ml_stack.app import create_app
|
|
14
|
+
from haystack_ml_stack.cache import make_features_cache
|
|
15
|
+
from haystack_ml_stack.dynamo import _build_cache_key, set_all_features
|
|
16
|
+
from haystack_ml_stack._kafka import default_serialization
|
|
17
|
+
from haystack_ml_stack.settings import Settings
|
|
18
|
+
|
|
19
|
+
EMBEDDING = "EMBEDDING#GEMINI_2_TITLE_TRANSCRIPT#512"
|
|
20
|
+
HISTORY = "HISTORY#5M#256"
|
|
21
|
+
CANDIDATE = "http://haystack.tv/id/candidate"
|
|
22
|
+
WATCHED = "https://example.com/watch,video#clip"
|
|
23
|
+
NO_VECTOR = "http://haystack.tv/id/no-vector"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class FakeDynamo:
|
|
27
|
+
"""batch_get_item over in-memory items."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, items=()):
|
|
30
|
+
self.items = {(i["pk"]["S"], i["sk"]["S"]): i for i in items}
|
|
31
|
+
self.calls = 0
|
|
32
|
+
self.batches = []
|
|
33
|
+
self.active = 0
|
|
34
|
+
self.max_active = 0
|
|
35
|
+
|
|
36
|
+
async def batch_get_item(self, RequestItems):
|
|
37
|
+
self.calls += 1
|
|
38
|
+
((table, request),) = RequestItems.items()
|
|
39
|
+
keys = [(k["pk"]["S"], k["sk"]["S"]) for k in request["Keys"]]
|
|
40
|
+
self.batches.append(keys)
|
|
41
|
+
self.active += 1
|
|
42
|
+
self.max_active = max(self.max_active, self.active)
|
|
43
|
+
try:
|
|
44
|
+
await asyncio.sleep(0)
|
|
45
|
+
return {"Responses": {table: [self.items[k] for k in keys if k in self.items]}}
|
|
46
|
+
finally:
|
|
47
|
+
self.active -= 1
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class FakeSession:
|
|
51
|
+
def __init__(self, client):
|
|
52
|
+
self.client = client
|
|
53
|
+
|
|
54
|
+
def create_client(self, *args, **kwargs):
|
|
55
|
+
client = self.client
|
|
56
|
+
|
|
57
|
+
class Context:
|
|
58
|
+
async def __aenter__(self):
|
|
59
|
+
return client
|
|
60
|
+
|
|
61
|
+
async def __aexit__(self, *exc):
|
|
62
|
+
return False
|
|
63
|
+
|
|
64
|
+
return Context()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def item(pk, sk, value):
|
|
68
|
+
return {
|
|
69
|
+
"pk": {"S": pk},
|
|
70
|
+
"sk": {"S": sk},
|
|
71
|
+
"value": {"B": value},
|
|
72
|
+
"version": {"S": "v1"},
|
|
73
|
+
"cache_ttl_in_seconds": {"N": "3600"},
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@pytest.mark.parametrize("candidate_features", [[EMBEDDING], []])
|
|
78
|
+
def test_history_vectors_reach_the_model_outside_the_logged_user(candidate_features):
|
|
79
|
+
history = SerializerRegistry[FeatureRegistryId("USER", HISTORY, "v1")].serialize(
|
|
80
|
+
{
|
|
81
|
+
"version": 1,
|
|
82
|
+
"data": {
|
|
83
|
+
"snapshot_at_ms": 1_758_499_200_000,
|
|
84
|
+
"stream_urls": [WATCHED, NO_VECTOR],
|
|
85
|
+
"event_at_ms": [1_758_400_000_000, 1_758_450_000_000],
|
|
86
|
+
"seconds_watched": [120.0, 0.0],
|
|
87
|
+
"contexts": ["sel thumb", "sel thumb"],
|
|
88
|
+
"browsed": [False, True],
|
|
89
|
+
},
|
|
90
|
+
}
|
|
91
|
+
)
|
|
92
|
+
vector = SerializerRegistry[FeatureRegistryId("STREAM", EMBEDDING, "v1")].serialize(
|
|
93
|
+
{"version": 1, "data": [0.01] * 512}
|
|
94
|
+
)
|
|
95
|
+
client = FakeDynamo(
|
|
96
|
+
[
|
|
97
|
+
item("USER#u1", HISTORY, history),
|
|
98
|
+
item(f"STREAM#{CANDIDATE}", EMBEDDING, vector),
|
|
99
|
+
item(f"STREAM#{WATCHED}", EMBEDDING, vector),
|
|
100
|
+
]
|
|
101
|
+
)
|
|
102
|
+
seen = {}
|
|
103
|
+
|
|
104
|
+
def preprocess(user, streams, playlist, params):
|
|
105
|
+
seen["user"] = dict(user)
|
|
106
|
+
seen["history_streams"] = params["history_streams"]
|
|
107
|
+
return streams
|
|
108
|
+
|
|
109
|
+
model = {
|
|
110
|
+
"preprocess": preprocess,
|
|
111
|
+
"predict": lambda streams, params: {s["streamUrl"]: 0.5 for s in streams},
|
|
112
|
+
"params": {},
|
|
113
|
+
"stream_features": candidate_features,
|
|
114
|
+
"user_features": [HISTORY],
|
|
115
|
+
"history": {"user_feature": HISTORY, "stream_features": [EMBEDDING]},
|
|
116
|
+
}
|
|
117
|
+
events = []
|
|
118
|
+
original_session = aiobotocore.session.get_session
|
|
119
|
+
original_record = newrelic.agent.record_custom_event
|
|
120
|
+
aiobotocore.session.get_session = lambda: FakeSession(client)
|
|
121
|
+
newrelic.agent.record_custom_event = lambda name, meta: events.append(meta)
|
|
122
|
+
try:
|
|
123
|
+
app = create_app(Settings(), preloaded_model=model)
|
|
124
|
+
score = next(r for r in app.routes if getattr(r, "path", "") == "/score")
|
|
125
|
+
body = json.dumps({"user": {"userid": "u1"}, "streams": [{"streamUrl": CANDIDATE}]}).encode()
|
|
126
|
+
|
|
127
|
+
async def receive():
|
|
128
|
+
return {"type": "http.request", "body": body, "more_body": False}
|
|
129
|
+
|
|
130
|
+
async def call():
|
|
131
|
+
scope = {"type": "http", "method": "POST", "path": "/score",
|
|
132
|
+
"headers": [], "query_string": b""}
|
|
133
|
+
async with app.router.lifespan_context(app):
|
|
134
|
+
return await score.endpoint(Request(scope, receive), Response(), BackgroundTasks())
|
|
135
|
+
|
|
136
|
+
output = asyncio.run(call())
|
|
137
|
+
finally:
|
|
138
|
+
aiobotocore.session.get_session = original_session
|
|
139
|
+
newrelic.agent.record_custom_event = original_record
|
|
140
|
+
|
|
141
|
+
assert output == {CANDIDATE: 0.5}
|
|
142
|
+
assert [s["streamUrl"] for s in seen["history_streams"]] == [WATCHED, NO_VECTOR]
|
|
143
|
+
assert len(seen["history_streams"][0][EMBEDDING].data) == 1024 # 512 float16
|
|
144
|
+
assert EMBEDDING not in seen["history_streams"][1]
|
|
145
|
+
assert set(seen["user"]) == {"userid", HISTORY}
|
|
146
|
+
# The retrieval metrics count both reads: the candidate's vector and the history
|
|
147
|
+
# record, then the vectors of the history's two streams.
|
|
148
|
+
(event,) = events
|
|
149
|
+
expected = (4, 3, 1) if candidate_features else (3, 2, 1)
|
|
150
|
+
assert (event["cache_misses"], event["stream_cache_misses"], event["user_cache_misses"]) == expected
|
|
151
|
+
assert event["retrieval_success"] == 1
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@pytest.mark.parametrize("cache_state", ["cold", "history", "all"])
|
|
155
|
+
def test_parallel_history_fetch_and_shared_vector_assignment(cache_state):
|
|
156
|
+
import orjson
|
|
157
|
+
|
|
158
|
+
candidates = [f"http://haystack.tv/id/c{i}" for i in range(100)]
|
|
159
|
+
urls = candidates[:50] + [f"http://haystack.tv/id/h{i}" for i in range(206)]
|
|
160
|
+
history_serializer = SerializerRegistry[FeatureRegistryId("USER", HISTORY, "v1")]
|
|
161
|
+
vector_serializer = SerializerRegistry[FeatureRegistryId("STREAM", EMBEDDING, "v1")]
|
|
162
|
+
history = history_serializer.serialize({
|
|
163
|
+
"version": 1,
|
|
164
|
+
"data": {"snapshot_at_ms": 1_758_499_200_000,
|
|
165
|
+
"stream_urls": urls, "event_at_ms": [1_758_400_000_000] * 256,
|
|
166
|
+
"seconds_watched": [30.] * 256, "contexts": ["pres nxt"] * 256,
|
|
167
|
+
"browsed": [False] * 256},
|
|
168
|
+
})
|
|
169
|
+
vector = vector_serializer.serialize({"version": 1, "data": [.01] * 512})
|
|
170
|
+
rows = [item("USER#u1", HISTORY, history)] + [
|
|
171
|
+
item("STREAM#" + url, EMBEDDING, vector) for url in dict.fromkeys(candidates + urls)
|
|
172
|
+
]
|
|
173
|
+
client = FakeDynamo(rows)
|
|
174
|
+
user_cache, stream_cache = make_features_cache(1000), make_features_cache(1000)
|
|
175
|
+
def cache_value(cache, kind, entity, feature, value):
|
|
176
|
+
cache[_build_cache_key(kind, entity, feature, "--")] = {
|
|
177
|
+
"value": value, "cache_ttl_in_seconds": 3600,
|
|
178
|
+
"inserted_at": datetime.datetime.utcnow(),
|
|
179
|
+
}
|
|
180
|
+
if cache_state != "cold":
|
|
181
|
+
cache_value(user_cache, "USER", "u1", HISTORY, history_serializer.deserialize(history))
|
|
182
|
+
if cache_state == "all":
|
|
183
|
+
for url in dict.fromkeys(candidates + urls):
|
|
184
|
+
cache_value(stream_cache, "STREAM", url, EMBEDDING, vector_serializer.deserialize(vector))
|
|
185
|
+
user, streams, history_streams = {"userid": "u1"}, [{"streamUrl": u} for u in candidates], []
|
|
186
|
+
meta = asyncio.run(set_all_features(
|
|
187
|
+
user=user, streams=streams, stream_features=[EMBEDDING], user_features=[HISTORY],
|
|
188
|
+
history={"user_feature": HISTORY, "stream_features": [EMBEDDING]},
|
|
189
|
+
history_streams=history_streams, stream_features_cache=stream_cache,
|
|
190
|
+
user_features_cache=user_cache, features_table="features", cache_sep="--", dynamo_client=client,
|
|
191
|
+
))
|
|
192
|
+
assert meta.success
|
|
193
|
+
assert [s["streamUrl"] for s in history_streams] == urls
|
|
194
|
+
assert all(s[EMBEDDING].data == vector_serializer.deserialize(vector).data for s in streams + history_streams)
|
|
195
|
+
assert set(user) == {"userid", HISTORY}
|
|
196
|
+
# Both a candidate and its history reference receive the one fetched object.
|
|
197
|
+
assert streams[0][EMBEDDING] is history_streams[0][EMBEDDING]
|
|
198
|
+
batches = [len(batch) for batch in client.batches]
|
|
199
|
+
assert batches == {"cold": [100, 1, 100, 100, 6], "history": [100, 100, 100, 6], "all": []}[cache_state]
|
|
200
|
+
assert client.max_active == {"cold": 3, "history": 4, "all": 0}[cache_state]
|
|
201
|
+
keys = [key for batch in client.batches for key in batch]
|
|
202
|
+
assert len(keys) == len(set(keys)) # no repeat lookups across either stage
|
|
203
|
+
assert meta.stream_cache_misses == (0 if cache_state == "all" else 306)
|
|
204
|
+
# Exercise the real Kafka encoder; vectors of history stay outside user/streams.
|
|
205
|
+
logged = orjson.loads(orjson.dumps({"user": user, "streams": streams}, default=default_serialization))
|
|
206
|
+
assert set(logged["user"]) == {"userid", HISTORY}
|
|
207
|
+
restored = history_serializer.deserialize(base64.b64decode(logged["user"][HISTORY]["proto"]))
|
|
208
|
+
assert restored == user[HISTORY]
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def test_missing_history_is_cached_only_after_a_successful_read():
|
|
212
|
+
client = FakeDynamo()
|
|
213
|
+
user_cache, stream_cache = make_features_cache(10), make_features_cache(10)
|
|
214
|
+
async def run():
|
|
215
|
+
for _ in range(2):
|
|
216
|
+
history_streams = []
|
|
217
|
+
meta = await set_all_features(
|
|
218
|
+
user={"userid": "u1"}, streams=[{"streamUrl": CANDIDATE}],
|
|
219
|
+
stream_features=[], user_features=[HISTORY],
|
|
220
|
+
history={"user_feature": HISTORY, "stream_features": [EMBEDDING]},
|
|
221
|
+
history_streams=history_streams, stream_features_cache=stream_cache,
|
|
222
|
+
user_features_cache=user_cache, features_table="features", cache_sep="--", dynamo_client=client,
|
|
223
|
+
)
|
|
224
|
+
assert meta.success and history_streams == []
|
|
225
|
+
asyncio.run(run())
|
|
226
|
+
assert client.calls == 1
|
|
@@ -1,6 +1,12 @@
|
|
|
1
|
+
import math
|
|
2
|
+
import struct
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
1
6
|
from google.protobuf.json_format import ParseDict as ProtoParseDict
|
|
2
7
|
from haystack_ml_stack import SerializerRegistry
|
|
3
8
|
from haystack_ml_stack.generated.v1 import features_pb2 as features_pb2_v1
|
|
9
|
+
from haystack_ml_stack.history import history_entries, history_stream_urls
|
|
4
10
|
|
|
5
11
|
|
|
6
12
|
def test_v0_serializers():
|
|
@@ -211,3 +217,89 @@ def test_user_playlist_stats_serializer_backward_compatibility():
|
|
|
211
217
|
assert result.data["sports"].total_days == 30
|
|
212
218
|
assert result.data["sports"].capped_total_log_watched == 0.0
|
|
213
219
|
assert result.data["sports"].first_active_date == ""
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def test_stream_embedding_serializer():
|
|
223
|
+
values = [math.sin(0.37 * (i + 1)) for i in range(512)]
|
|
224
|
+
serializer = SerializerRegistry[
|
|
225
|
+
("STREAM", "EMBEDDING#GEMINI_2_TITLE_TRANSCRIPT#512", "v1")
|
|
226
|
+
]
|
|
227
|
+
message = serializer.deserialize(serializer.serialize({"version": 1, "data": values}))
|
|
228
|
+
vector = struct.unpack("<512e", message.data)
|
|
229
|
+
norm = math.sqrt(sum(x * x for x in values))
|
|
230
|
+
assert message.version == 1
|
|
231
|
+
assert all(abs(a - b / norm) < 1e-3 for a, b in zip(vector, values))
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def test_stream_embedding_normalizes_before_float16_storage():
|
|
235
|
+
import numpy as np
|
|
236
|
+
values = [math.sin(0.37 * (i + 1)) * (1 + i % 7) for i in range(512)]
|
|
237
|
+
first = np.asarray(values, dtype=np.float32)
|
|
238
|
+
expected = (first / np.linalg.norm(first)).astype("<f2")
|
|
239
|
+
serializer = SerializerRegistry[
|
|
240
|
+
("STREAM", "EMBEDDING#GEMINI_2_TITLE_TRANSCRIPT#512", "v1")
|
|
241
|
+
]
|
|
242
|
+
message = serializer.deserialize(serializer.serialize({"version": 1, "data": values}))
|
|
243
|
+
assert message.data == expected.tobytes()
|
|
244
|
+
assert len(message.data) == 1024
|
|
245
|
+
rounded = first.astype(np.float16).astype(np.float32)
|
|
246
|
+
assert message.data != (rounded / np.linalg.norm(rounded)).astype("<f2").tobytes()
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
@pytest.mark.parametrize("count", [511, 513])
|
|
250
|
+
def test_stream_embedding_requires_512_coordinates(count):
|
|
251
|
+
serializer = SerializerRegistry[("STREAM", "EMBEDDING#GEMINI_2_TITLE_TRANSCRIPT#512", "v1")]
|
|
252
|
+
with pytest.raises(AssertionError, match="512 coordinates"):
|
|
253
|
+
serializer.serialize({"version": 1, "data": [1.] * count})
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def test_user_history_serializer():
|
|
257
|
+
data = {
|
|
258
|
+
"snapshot_at_ms": 1758499200000,
|
|
259
|
+
"stream_urls": ["http://haystack.tv/id/a", "https://example.com/a,b#c",
|
|
260
|
+
"http://haystack.tv/id/:escape", "relative/path", "http://haystack.tv/id/e"],
|
|
261
|
+
"event_at_ms": [1758400000001 + i for i in range(5)],
|
|
262
|
+
"seconds_watched": [25.7, 3600.0, 30.0, 0.0, 25.0],
|
|
263
|
+
"contexts": ["pres nxt", "null", "sel thumb", "unsupported", "__missing__"],
|
|
264
|
+
"browsed": [False, False, True, False, False],
|
|
265
|
+
}
|
|
266
|
+
serializer = SerializerRegistry[("USER", "HISTORY#5M#256", "v1")]
|
|
267
|
+
message = serializer.deserialize(
|
|
268
|
+
serializer.serialize({"version": 1, "data": data})
|
|
269
|
+
)
|
|
270
|
+
assert message.version == 1
|
|
271
|
+
assert message.data.snapshot_at_ms == data["snapshot_at_ms"]
|
|
272
|
+
assert list(message.data.stream_ids) == ["a", ":https://example.com/a,b#c",
|
|
273
|
+
":http://haystack.tv/id/:escape", ":relative/path", "e"]
|
|
274
|
+
assert history_stream_urls(message) == data["stream_urls"]
|
|
275
|
+
rows = list(history_entries(message))
|
|
276
|
+
assert [r[1] for r in rows] == data["event_at_ms"]
|
|
277
|
+
assert [r[2] for r in rows] == pytest.approx(data["seconds_watched"])
|
|
278
|
+
assert [r[3] for r in rows] == ["pres nxt", "", "sel thumb", "__unsupported__", ""]
|
|
279
|
+
assert [r[4] for r in rows] == data["browsed"]
|
|
280
|
+
assert [r[5] for r in rows] == [True, True, False, False, False]
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def test_empty_history_keeps_snapshot_inside_data():
|
|
284
|
+
serializer = SerializerRegistry[("USER", "HISTORY#5M#256", "v1")]
|
|
285
|
+
data = dict(snapshot_at_ms=1758499200000, stream_urls=[], event_at_ms=[],
|
|
286
|
+
seconds_watched=[], contexts=[], browsed=[])
|
|
287
|
+
message = serializer.deserialize(serializer.serialize({"version": 1, "data": data}))
|
|
288
|
+
assert message.data.snapshot_at_ms == data["snapshot_at_ms"]
|
|
289
|
+
assert history_stream_urls(message) == []
|
|
290
|
+
assert list(history_entries(message)) == []
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
@pytest.mark.parametrize("value", [0.0, float("nan"), float("inf")])
|
|
294
|
+
def test_invalid_vector_norm_has_a_clear_error(value):
|
|
295
|
+
serializer = SerializerRegistry[("STREAM", "EMBEDDING#GEMINI_2_TITLE_TRANSCRIPT#512", "v1")]
|
|
296
|
+
with pytest.raises(ValueError, match="finite, nonzero"):
|
|
297
|
+
serializer.serialize({"version": 1, "data": [value] * 512})
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def test_history_decoder_rejects_misaligned_columns():
|
|
301
|
+
message = features_pb2_v1.UserHistory(version=1)
|
|
302
|
+
message.data.snapshot_at_ms = 1000
|
|
303
|
+
message.data.stream_ids.append("a")
|
|
304
|
+
with pytest.raises(ValueError, match="column lengths"):
|
|
305
|
+
list(history_entries(message))
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "0.4.12"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/generated/__init__.py
RENAMED
|
File without changes
|
{haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack/generated/v1/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{haystack_ml_stack-0.4.12 → haystack_ml_stack-0.4.13}/src/haystack_ml_stack.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|