in-layers-data 0.4.11__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
in_layers/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """Namespace package for in_layers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import pkgutil
6
+
7
+ __path__ = pkgutil.extend_path(__path__, __name__) # type: ignore[name-defined]
@@ -0,0 +1,21 @@
1
+ from . import services
2
+ from .protocols import (
3
+ BackendConfig,
4
+ DataNamespace,
5
+ DynamoDBBackendConfig,
6
+ InLayersDataConfig,
7
+ MongoBackendConfig,
8
+ SupportedBackend,
9
+ )
10
+
11
+ name = DataNamespace.root.value
12
+
13
+ __all__ = [
14
+ "BackendConfig",
15
+ "DynamoDBBackendConfig",
16
+ "InLayersDataConfig",
17
+ "MongoBackendConfig",
18
+ "SupportedBackend",
19
+ "name",
20
+ "services",
21
+ ]
File without changes
@@ -0,0 +1,5 @@
1
+ """DynamoDB backend for InLayers models."""
2
+
3
+ from .services import DynamoDBBackend
4
+
5
+ __all__ = ["DynamoDBBackend"]
@@ -0,0 +1,86 @@
1
+ """Pure functional utilities for DynamoDB query conversion and data formatting."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Mapping
7
+ from datetime import datetime
8
+ from typing import Any
9
+
10
+ from in_layers.core.models.protocols import ModelDefinition
11
+
12
+
13
+ def get_table_name_for_model(
14
+ environment: str, model_definition: ModelDefinition
15
+ ) -> str:
16
+ """Generate a DynamoDB table name from a model definition."""
17
+ name = model_definition.plural_name.replace("@", "").replace("/", "-")
18
+ # Convert to kebab-case: insert hyphens before uppercase letters (except first)
19
+ # and handle sequences of uppercase letters
20
+ name = re.sub(r"([a-z0-9])([A-Z])", r"\1-\2", name)
21
+ name = re.sub(r"([A-Z]+)([A-Z][a-z])", r"\1-\2", name)
22
+ return f"{name}-{environment}".lower()
23
+
24
+
25
+ def format_for_dynamodb(data: Mapping[str, Any]) -> dict[str, Any]:
26
+ """Format data for DynamoDB storage.
27
+
28
+ DynamoDB natively handles:
29
+ - Strings
30
+ - Numbers (int, float, Decimal)
31
+ - Binary data
32
+ - Boolean
33
+ - Null
34
+ - Lists
35
+ - Maps (dicts)
36
+ - Sets (string sets, number sets, binary sets)
37
+
38
+ The boto3 DynamoDBDocumentClient will handle conversion automatically,
39
+ but we ensure datetime objects are converted to ISO format strings.
40
+ """
41
+ result = dict(data)
42
+ # Convert datetime objects to ISO format strings for DynamoDB
43
+
44
+ for key, value in result.items():
45
+ if isinstance(value, datetime):
46
+ result[key] = value.isoformat()
47
+ return result
48
+
49
+
50
+ def from_dynamodb(item: dict[str, Any] | None) -> dict[str, Any]:
51
+ """Convert a DynamoDB item to a plain dictionary.
52
+
53
+ The boto3 DynamoDBDocumentClient already converts AttributeValue format
54
+ to native Python types, so this is mainly for consistency and future-proofing.
55
+ """
56
+ if item is None:
57
+ return {}
58
+ return dict(item)
59
+
60
+
61
+ def build_scan_params(
62
+ table_name: str, exclusive_start_key: dict[str, Any] | None = None
63
+ ) -> dict[str, Any]:
64
+ """Build parameters for a DynamoDB Scan operation."""
65
+ params: dict[str, Any] = {
66
+ "TableName": table_name,
67
+ }
68
+ if exclusive_start_key:
69
+ params["ExclusiveStartKey"] = exclusive_start_key
70
+ return params
71
+
72
+
73
+ def split_array_into_batches(array: list[Any], max_batch_size: int) -> list[list[Any]]:
74
+ """Split an array into batches of maximum size.
75
+
76
+ DynamoDB has limits on batch operations (e.g., BatchWriteItem max 25 items).
77
+ """
78
+ if not isinstance(array, list):
79
+ raise ValueError("Input must be a list")
80
+ if max_batch_size < 1:
81
+ raise ValueError("max_batch_size must be at least 1")
82
+
83
+ batches = []
84
+ for i in range(0, len(array), max_batch_size):
85
+ batches.append(array[i : i + max_batch_size])
86
+ return batches
@@ -0,0 +1,387 @@
1
+ """DynamoDB backend implementation for InLayers models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from typing import Any
7
+ from uuid import uuid4
8
+
9
+ from box import Box
10
+ from in_layers.core.models.backends import (
11
+ _apply_sort,
12
+ _apply_take,
13
+ _matches_query_tokens,
14
+ )
15
+ from in_layers.core.models.protocols import (
16
+ InLayersModel,
17
+ ModelSearch,
18
+ ModelSearchResult,
19
+ PrimaryKeyType,
20
+ )
21
+
22
+ from in_layers.data.protocols import DynamoDBBackendConfig
23
+
24
+ from .libs import (
25
+ format_for_dynamodb,
26
+ from_dynamodb,
27
+ get_table_name_for_model,
28
+ split_array_into_batches,
29
+ )
30
+
31
+ # DynamoDB batch operation limits
32
+ MAX_BATCH_WRITE_SIZE = 25
33
+ SCAN_RETURN_THRESHOLD = 1000
34
+
35
+
36
+ class DynamoDBBackend:
37
+ """DynamoDB backend implementation."""
38
+
39
+ def __init__(self, context, config: DynamoDBBackendConfig):
40
+ self.__context = context
41
+ self.__config = config
42
+ self.__client: Any = None
43
+ self.__table_client: Any = None
44
+
45
+ @staticmethod
46
+ def create_unique_connection_string(config: DynamoDBBackendConfig) -> str:
47
+ """Create a unique connection string from config."""
48
+ region = config.region or "us-east-1"
49
+ endpoint_url = config.endpoint_url or ""
50
+
51
+ parts = [f"region={region}"]
52
+ if endpoint_url:
53
+ parts.append(f"endpoint={endpoint_url}")
54
+
55
+ return "|".join(parts)
56
+
57
+ def __get_config_value(self, key: str, default_value: Any = None) -> Any:
58
+ if key in self.__config:
59
+ return self.__config[key]
60
+ return default_value
61
+
62
+ def __connect(self) -> None:
63
+ """Connect to DynamoDB (private method)."""
64
+ # Use boto3 from config if provided (for testing), otherwise import it
65
+ if self.__get_config_value("boto3") is not None:
66
+ boto3 = self.__get_config_value("boto3")
67
+ else:
68
+ import boto3 # pragma: no cover # noqa: PLC0415
69
+
70
+ region = self.__get_config_value("region")
71
+ endpoint_url = self.__get_config_value("endpoint_url")
72
+ aws_access_key_id = self.__get_config_value("aws_access_key_id")
73
+ aws_secret_access_key = self.__get_config_value("aws_secret_access_key")
74
+ client_kwargs: dict[str, Any] = {}
75
+ if region:
76
+ client_kwargs["region_name"] = region
77
+ if endpoint_url:
78
+ client_kwargs["endpoint_url"] = endpoint_url
79
+ if aws_access_key_id:
80
+ client_kwargs["aws_access_key_id"] = aws_access_key_id
81
+ if aws_secret_access_key:
82
+ client_kwargs["aws_secret_access_key"] = aws_secret_access_key
83
+
84
+ # Create DynamoDB client
85
+ self.__client = boto3.client("dynamodb", **client_kwargs)
86
+
87
+ # Create DynamoDB resource for easier table operations
88
+ dynamodb_resource = boto3.resource("dynamodb", **client_kwargs)
89
+ self.__table_client = dynamodb_resource
90
+
91
+ def __disconnect(self) -> None:
92
+ """Disconnect from DynamoDB (private method)."""
93
+ # boto3 clients don't need explicit closing, but we can clear references
94
+ self.__client = None
95
+ self.__table_client = None
96
+
97
+ def __ensure_connected(self) -> None:
98
+ """Ensure DynamoDB connection is established."""
99
+ if self.__client is None:
100
+ self.__connect()
101
+
102
+ def create(self, model: InLayersModel, data: Mapping) -> Mapping:
103
+ """Create a new item in DynamoDB."""
104
+ self.__ensure_connected()
105
+
106
+ table_name = get_table_name_for_model(
107
+ self.__context.config.environment, model.get_model_definition()
108
+ )
109
+ table = self.__table_client.Table(table_name)
110
+
111
+ payload = dict(data)
112
+ formatted = format_for_dynamodb(payload)
113
+
114
+ pk_name = model.get_primary_key_name()
115
+ pk_value = formatted.get(pk_name)
116
+ if pk_value is None:
117
+ pk_value = str(uuid4())
118
+ formatted[pk_name] = pk_value
119
+
120
+ # Ensure primary key is a string (DynamoDB requirement)
121
+ formatted[pk_name] = str(pk_value)
122
+
123
+ # Put item in DynamoDB
124
+ table.put_item(Item=formatted)
125
+ return formatted
126
+
127
+ def retrieve(self, model: InLayersModel, id: PrimaryKeyType) -> Mapping | None:
128
+ """Retrieve an item by ID."""
129
+ self.__ensure_connected()
130
+
131
+ table_name = get_table_name_for_model(
132
+ self.__context.config.environment, model.get_model_definition()
133
+ )
134
+ table = self.__table_client.Table(table_name)
135
+
136
+ pk_name = model.get_primary_key_name()
137
+ key = {pk_name: str(id)}
138
+
139
+ response = table.get_item(Key=key)
140
+ item = response.get("Item")
141
+ if not item:
142
+ return None
143
+
144
+ return from_dynamodb(item)
145
+
146
+ def update(
147
+ self, model: InLayersModel, id: PrimaryKeyType, data: Mapping
148
+ ) -> Mapping:
149
+ """Update an item by ID.
150
+
151
+ Supports partial updates - only the fields provided in data will be updated.
152
+ Returns the full merged object (original + updates).
153
+ """
154
+ self.__ensure_connected()
155
+
156
+ table_name = get_table_name_for_model(
157
+ self.__context.config.environment, model.get_model_definition()
158
+ )
159
+ table = self.__table_client.Table(table_name)
160
+
161
+ pk_name = model.get_primary_key_name()
162
+ key = {pk_name: str(id)}
163
+
164
+ # Check if item exists
165
+ existing = table.get_item(Key=key)
166
+ if "Item" not in existing:
167
+ raise KeyError(f"Instance with id {id!r} not found")
168
+
169
+ payload = dict(data)
170
+ formatted = format_for_dynamodb(payload)
171
+
172
+ # Ensure primary key field remains consistent
173
+ formatted[pk_name] = str(id)
174
+
175
+ # Build UpdateExpression for partial update
176
+ # DynamoDB requires SET expressions for each attribute
177
+ update_expressions = []
178
+ expression_attribute_names = {}
179
+ expression_attribute_values = {}
180
+
181
+ for attr_name, attr_value in formatted.items():
182
+ # Skip the primary key as it's in the Key parameter
183
+ if attr_name == pk_name:
184
+ continue
185
+
186
+ # Use attribute name placeholders to handle reserved words
187
+ name_placeholder = f"#attr_{attr_name}"
188
+ value_placeholder = f":val_{attr_name}"
189
+
190
+ expression_attribute_names[name_placeholder] = attr_name
191
+ expression_attribute_values[value_placeholder] = attr_value
192
+ update_expressions.append(f"{name_placeholder} = {value_placeholder}")
193
+
194
+ if not update_expressions:
195
+ # No fields to update (only primary key was provided)
196
+ # Just return the existing item
197
+ return from_dynamodb(existing["Item"])
198
+
199
+ # Build the update expression
200
+ update_expression = f"SET {', '.join(update_expressions)}"
201
+
202
+ # Perform partial update using update_item
203
+ table.update_item(
204
+ Key=key,
205
+ UpdateExpression=update_expression,
206
+ ExpressionAttributeNames=expression_attribute_names,
207
+ ExpressionAttributeValues=expression_attribute_values,
208
+ )
209
+
210
+ # Retrieve the updated item to return the full merged object
211
+ updated = table.get_item(Key=key)
212
+ if "Item" not in updated:
213
+ raise KeyError(f"Instance with id {id!r} not found after update")
214
+
215
+ return from_dynamodb(updated["Item"])
216
+
217
+ def delete(self, model: InLayersModel, id: PrimaryKeyType) -> None:
218
+ """Delete an item by ID."""
219
+ self.__ensure_connected()
220
+
221
+ table_name = get_table_name_for_model(
222
+ self.__context.config.environment, model.get_model_definition()
223
+ )
224
+ table = self.__table_client.Table(table_name)
225
+
226
+ pk_name = model.get_primary_key_name()
227
+ key = {pk_name: str(id)}
228
+
229
+ table.delete_item(Key=key)
230
+
231
+ def search(self, model: InLayersModel, query: ModelSearch) -> ModelSearchResult:
232
+ """Search for items matching the query.
233
+
234
+ Note: DynamoDB Scan operations are expensive and should be used sparingly.
235
+ For production use, consider using Query operations with Global Secondary Indexes (GSI)
236
+ or Local Secondary Indexes (LSI) for better performance.
237
+
238
+ This implementation continues scanning across pages until:
239
+ - Threshold is met (SCAN_RETURN_THRESHOLD if no take specified)
240
+ - No more keys (LastEvaluatedKey is null)
241
+ - Take limit is reached (if take is specified)
242
+ """
243
+ self.__ensure_connected()
244
+
245
+ table_name = get_table_name_for_model(
246
+ self.__context.config.environment, model.get_model_definition()
247
+ )
248
+ table = self.__table_client.Table(table_name)
249
+
250
+ # Start recursive scanning
251
+ result = self._do_search_until_threshold_or_no_last_evaluated_key(
252
+ table, query, []
253
+ )
254
+
255
+ return Box(instances=result["instances"], page=result["page"])
256
+
257
+ def _do_search_until_threshold_or_no_last_evaluated_key(
258
+ self,
259
+ table: Any,
260
+ search: ModelSearch,
261
+ old_instances_found: list[dict[str, Any]],
262
+ ) -> dict[str, Any]:
263
+ """Recursively scan DynamoDB until threshold is met or no more keys.
264
+
265
+ This matches the TypeScript implementation's behavior of continuing
266
+ to scan across pages until enough filtered results are found.
267
+ """
268
+ # Build scan parameters
269
+ scan_kwargs: dict[str, Any] = {}
270
+ query = getattr(search, "query", [])
271
+ take = getattr(search, "take", None)
272
+ sort = getattr(search, "sort", None)
273
+ if getattr(search, "page", None):
274
+ scan_kwargs["ExclusiveStartKey"] = search.page
275
+
276
+ # Execute scan
277
+ response = table.scan(**scan_kwargs)
278
+ items = response.get("Items", [])
279
+
280
+ # Convert DynamoDB items to plain dicts
281
+ unfiltered = [from_dynamodb(item) for item in items]
282
+
283
+ # Apply filtering using the same logic as MemoryBackend
284
+ filtered = [r for r in unfiltered if _matches_query_tokens(r, query)]
285
+
286
+ # Combine with previously found instances
287
+ all_filtered = filtered + old_instances_found
288
+
289
+ # Determine threshold
290
+ using_take = take is not None and take > 0
291
+ take = take if using_take else SCAN_RETURN_THRESHOLD
292
+
293
+ # Get pagination key
294
+ last_evaluated_key = response.get("LastEvaluatedKey")
295
+
296
+ # Check stopping conditions:
297
+ # 1. We have enough results (more than threshold)
298
+ # 2. No more keys to evaluate
299
+ # 3. If using take, we've hit our max
300
+ # Note: TypeScript uses > (strictly greater), meaning we continue if we have exactly 'take' items
301
+ stop_for_threshold = len(all_filtered) > take
302
+ stop_for_no_more = last_evaluated_key is None
303
+
304
+ if stop_for_threshold or stop_for_no_more:
305
+ # Apply sorting and take limit
306
+ sorted_instances = _apply_sort(all_filtered, sort)
307
+ limited_instances = _apply_take(sorted_instances, search.take)
308
+
309
+ # Return page: null when using take, otherwise return LastEvaluatedKey
310
+ page = None if using_take else last_evaluated_key
311
+
312
+ return {
313
+ "instances": [dict(x) for x in limited_instances],
314
+ "page": page,
315
+ }
316
+
317
+ # Continue scanning with the new page key
318
+ # Create a new ModelSearch with updated page (frozen dataclass requires new instance)
319
+ new_query = ModelSearch(
320
+ query=query,
321
+ take=search.take,
322
+ sort=search.sort,
323
+ page=last_evaluated_key,
324
+ )
325
+ return self._do_search_until_threshold_or_no_last_evaluated_key(
326
+ table, new_query, all_filtered
327
+ )
328
+
329
+ def bulk_insert(self, model: InLayersModel, data: list[Mapping]) -> None:
330
+ """Bulk insert items."""
331
+ self.__ensure_connected()
332
+
333
+ table_name = get_table_name_for_model(
334
+ self.__context.config.environment, model.get_model_definition()
335
+ )
336
+ table = self.__table_client.Table(table_name)
337
+ pk_name = model.get_primary_key_name()
338
+
339
+ # Prepare items
340
+ items = []
341
+ for item in data:
342
+ payload = dict(item)
343
+ formatted = format_for_dynamodb(payload)
344
+
345
+ pk_value = formatted.get(pk_name)
346
+ if pk_value is None:
347
+
348
+ pk_value = str(uuid4())
349
+ formatted[pk_name] = pk_value
350
+
351
+ formatted[pk_name] = str(pk_value)
352
+ items.append(formatted)
353
+
354
+ # Split into batches (DynamoDB BatchWriteItem limit is 25)
355
+ batches = split_array_into_batches(items, MAX_BATCH_WRITE_SIZE)
356
+
357
+ # Write batches
358
+ for batch in batches:
359
+ with table.batch_writer() as writer:
360
+ for item in batch:
361
+ writer.put_item(Item=item)
362
+
363
+ def bulk_delete(self, model: InLayersModel, ids: list[PrimaryKeyType]) -> None:
364
+ """Bulk delete items by IDs."""
365
+ self.__ensure_connected()
366
+
367
+ table_name = get_table_name_for_model(
368
+ self.__context.config.environment, model.get_model_definition()
369
+ )
370
+ table = self.__table_client.Table(table_name)
371
+ pk_name = model.get_primary_key_name()
372
+
373
+ # Prepare keys
374
+ keys = [{pk_name: str(id)} for id in ids]
375
+
376
+ # Split into batches
377
+ batches = split_array_into_batches(keys, MAX_BATCH_WRITE_SIZE)
378
+
379
+ # Delete batches
380
+ for batch in batches:
381
+ with table.batch_writer() as writer:
382
+ for key in batch:
383
+ writer.delete_item(Key=key)
384
+
385
+ def dispose(self) -> None:
386
+ """Clean up resources."""
387
+ self.__disconnect()
File without changes
@@ -0,0 +1,229 @@
1
+ """Pure functional utilities for MongoDB query conversion and data formatting."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Mapping
7
+ from datetime import datetime
8
+ from typing import Any
9
+
10
+ from in_layers.core.models.protocols import (
11
+ BooleanQuery,
12
+ DatastoreValueType,
13
+ EqualitySymbol,
14
+ ModelDefinition,
15
+ PropertyQuery,
16
+ QueryTokens,
17
+ )
18
+
19
+
20
+ def get_collection_name_for_model(model_definition: ModelDefinition) -> str:
21
+ """Generate a MongoDB collection name from a model definition."""
22
+ name = model_definition.plural_name.replace("@", "").replace("/", "-")
23
+ # Convert to kebab-case: insert hyphens before uppercase letters (except first)
24
+ # and handle sequences of uppercase letters
25
+ name = re.sub(r"([a-z0-9])([A-Z])", r"\1-\2", name)
26
+ name = re.sub(r"([A-Z]+)([A-Z][a-z])", r"\1-\2", name)
27
+ return f"{name}".lower()
28
+
29
+
30
+ def escape_regex(s: str) -> str:
31
+ """Escape special regex characters in a string."""
32
+ return re.escape(s)
33
+
34
+
35
+ def build_string_pattern(
36
+ raw: str, starts_with: bool, ends_with: bool, includes: bool
37
+ ) -> str:
38
+ """Build a regex pattern for string matching."""
39
+ escaped = escape_regex(raw)
40
+ if starts_with:
41
+ return f"^{escaped}"
42
+ if ends_with:
43
+ return f"{escaped}$"
44
+ if includes:
45
+ return escaped
46
+ # default: exact match
47
+ return f"^{escaped}$"
48
+
49
+
50
+ def regex_object_for_pattern(pattern: str, case_sensitive: bool) -> dict[str, Any]:
51
+ """Create a MongoDB regex object from a pattern."""
52
+ if case_sensitive:
53
+ return {"$regex": pattern}
54
+ return {"$regex": pattern, "$options": "i"}
55
+
56
+
57
+ def build_mongo_find_value(query: PropertyQuery) -> dict[str, Any]: # noqa: PLR0911
58
+ """Convert a PropertyQuery to a MongoDB query value."""
59
+ value = query.value
60
+ if value is None:
61
+ return {query.key: None}
62
+
63
+ # Handle date objects
64
+ if hasattr(value, "isoformat"): # datetime objects
65
+ return {query.key: value}
66
+
67
+ if query.value_type == DatastoreValueType.string:
68
+ case_sensitive = bool(query.options.case_sensitive)
69
+ starts_with = bool(query.options.starts_with)
70
+ ends_with = bool(query.options.ends_with)
71
+ includes = bool(query.options.includes)
72
+
73
+ if query.equality_symbol not in (EqualitySymbol.eq, EqualitySymbol.ne):
74
+ raise ValueError(
75
+ f"Symbol {query.equality_symbol} is unhandled for string type"
76
+ )
77
+
78
+ raw = str(value)
79
+ pattern = build_string_pattern(raw, starts_with, ends_with, includes)
80
+ regex_obj = regex_object_for_pattern(pattern, case_sensitive)
81
+
82
+ if query.equality_symbol == EqualitySymbol.ne:
83
+ is_plain_exact = (
84
+ not starts_with and not ends_with and not includes and case_sensitive
85
+ )
86
+ if is_plain_exact:
87
+ return {query.key: {"$ne": raw}}
88
+ return {query.key: {"$not": regex_obj}}
89
+
90
+ use_regex = starts_with or ends_with or includes or not case_sensitive
91
+ if use_regex:
92
+ return {query.key: regex_obj}
93
+ return {query.key: raw}
94
+
95
+ if query.value_type == DatastoreValueType.number:
96
+ equality_symbol_to_mongo = {
97
+ EqualitySymbol.eq: "$eq",
98
+ EqualitySymbol.gt: "$gt",
99
+ EqualitySymbol.gte: "$gte",
100
+ EqualitySymbol.lt: "$lt",
101
+ EqualitySymbol.lte: "$lte",
102
+ EqualitySymbol.ne: "$ne",
103
+ }
104
+ mongo_symbol = equality_symbol_to_mongo.get(query.equality_symbol)
105
+ if not mongo_symbol:
106
+ raise ValueError(f"Symbol {query.equality_symbol} is unhandled")
107
+ return {query.key: {mongo_symbol: query.value}}
108
+
109
+ return {query.key: value}
110
+
111
+
112
+ def threeitize(
113
+ data: list[QueryTokens],
114
+ ) -> list[tuple[QueryTokens, BooleanQuery, QueryTokens]]:
115
+ """Convert a list of tokens into three-tuples of (left, link, right)."""
116
+ if len(data) in (0, 1):
117
+ return []
118
+ if len(data) % 2 == 0:
119
+ raise ValueError("Must be an odd number of 3 or greater.")
120
+ three = (data[0], _as_link(data[1]), data[2])
121
+ rest = data[2:]
122
+ more = threeitize(rest)
123
+ return [three, *more]
124
+
125
+
126
+ def _as_link(value: QueryTokens) -> BooleanQuery:
127
+ """Convert a token to a BooleanQuery link."""
128
+ if value in {"AND", "OR"}:
129
+ return value
130
+ raise ValueError("Must have AND/OR between statements")
131
+
132
+
133
+ def process_mongo_array(
134
+ tokens: list[QueryTokens],
135
+ ) -> dict[str, Any]:
136
+ """Process an array of query tokens into a MongoDB query."""
137
+ # If we don't have any AND/OR, it's all an AND
138
+ if all(t != "AND" and t != "OR" for t in tokens): # noqa: PLR1714
139
+ return {"$and": [handle_mongo_query(t) for t in tokens]}
140
+
141
+ # Process with threeitize
142
+ threes = threeitize(tokens)
143
+ threes.reverse()
144
+ result: dict[str, Any] = {}
145
+ for a, link, b in threes:
146
+ a_query = handle_mongo_query(a)
147
+ if result:
148
+ result = {f"${link.lower()}": [a_query, result]}
149
+ else:
150
+ b_query = handle_mongo_query(b)
151
+ result = {f"${link.lower()}": [a_query, b_query]}
152
+ return result
153
+
154
+
155
+ def handle_mongo_query(token: QueryTokens | list[QueryTokens]) -> dict[str, Any]:
156
+ """Convert a query token to a MongoDB query object."""
157
+ # Handle list of tokens (the main query)
158
+ if isinstance(token, list):
159
+ return process_mongo_array(token)
160
+
161
+ # Check by shape (duck typing) rather than isinstance
162
+ # PropertyQuery has type="property" and value attribute
163
+ if (
164
+ hasattr(token, "type")
165
+ and getattr(token, "type", None) == "property"
166
+ and hasattr(token, "value")
167
+ ):
168
+ return build_mongo_find_value(token)
169
+
170
+ # DatesBeforeQuery has type="datesBefore" and date attribute
171
+ if (
172
+ hasattr(token, "type")
173
+ and getattr(token, "type", None) == "datesBefore"
174
+ and hasattr(token, "date")
175
+ ):
176
+ date_value = token.date
177
+ if (
178
+ hasattr(token, "value_type") and token.value_type == DatastoreValueType.date
179
+ ) and isinstance(date_value, str):
180
+ # Convert string to datetime if needed
181
+ date_value = datetime.fromisoformat(date_value.replace("Z", "+00:00"))
182
+ operator = (
183
+ "$lte"
184
+ if (hasattr(token, "options") and token.options.equal_to_and_before)
185
+ else "$lt"
186
+ )
187
+ return {token.key: {operator: date_value}}
188
+
189
+ # DatesAfterQuery has type="datesAfter" and date attribute
190
+ if (
191
+ hasattr(token, "type")
192
+ and getattr(token, "type", None) == "datesAfter"
193
+ and hasattr(token, "date")
194
+ ):
195
+ date_value = token.date
196
+ if (
197
+ hasattr(token, "value_type") and token.value_type == DatastoreValueType.date
198
+ ) and isinstance(date_value, str):
199
+ date_value = datetime.fromisoformat(date_value.replace("Z", "+00:00"))
200
+ operator = (
201
+ "$gte"
202
+ if (hasattr(token, "options") and token.options.equal_to_and_after)
203
+ else "$gt"
204
+ )
205
+ return {token.key: {operator: date_value}}
206
+
207
+ raise ValueError(f"Unhandled query token {token}")
208
+
209
+
210
+ def to_mongo(query: list[QueryTokens]) -> list[dict[str, Any]]:
211
+ """Convert a query list to MongoDB aggregation pipeline stages."""
212
+ if not query:
213
+ return [{"$match": {}}]
214
+ # Pass the list directly to handle_mongo_query, which will process it as an array
215
+ match_query = handle_mongo_query(query)
216
+ return [{"$match": match_query}]
217
+
218
+
219
+ def format_for_mongo(data: Mapping[str, Any]) -> dict[str, Any]:
220
+ """Format data for MongoDB storage, converting dates and other types."""
221
+ result = dict(data)
222
+ # Convert datetime objects (they're already in the right format for MongoDB)
223
+ # In the TypeScript version, this iterates over model properties to find Datetime types
224
+ # For Python, we'll preserve datetime objects as-is (MongoDB handles them natively)
225
+ # and convert string ISO dates if they're clearly datetime values
226
+ for key, value in result.items():
227
+ if isinstance(value, datetime):
228
+ result[key] = value
229
+ return result
@@ -0,0 +1,237 @@
1
+ """MongoDB backend implementation for InLayers models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from typing import Any
7
+ from uuid import uuid4
8
+
9
+ from box import Box
10
+ from in_layers.core.models.protocols import (
11
+ InLayersModel,
12
+ ModelSearch,
13
+ ModelSearchResult,
14
+ PrimaryKeyType,
15
+ SortOrder,
16
+ )
17
+
18
+ from in_layers.data.protocols import MongoBackendConfig
19
+
20
+ from .libs import (
21
+ format_for_mongo,
22
+ get_collection_name_for_model,
23
+ to_mongo,
24
+ )
25
+
26
+
27
+ class MongoBackend:
28
+ """MongoDB backend implementation."""
29
+
30
+ def __init__(self, context, config: MongoBackendConfig):
31
+ self.__context = context
32
+ self.__config = config
33
+ self.__client: Any = None
34
+ self.__db: Any = None
35
+
36
+ @staticmethod
37
+ def create_unique_connection_string(config: MongoBackendConfig) -> str:
38
+ """Create a unique connection string from config."""
39
+ host = config.host
40
+ port = config.port
41
+ username = config.username
42
+ password = config.password
43
+ database = config.database
44
+
45
+ if username and password:
46
+ return f"mongodb://{username}:{password}@{host}:{port}/{database}"
47
+ return f"mongodb://{host}:{port}/{database}"
48
+
49
+ def __connect(self) -> None:
50
+ """Connect to MongoDB (private method)."""
51
+ from pymongo import MongoClient # noqa: PLC0415
52
+
53
+ connection_string = self.create_unique_connection_string(
54
+ {
55
+ "host": self.__config.host,
56
+ "port": self.__config.port,
57
+ "username": self.__config.username,
58
+ "password": self.__config.password,
59
+ "database": self.__config.database,
60
+ }
61
+ )
62
+ self.__client = MongoClient(connection_string)
63
+ self.__db = self.__client[self.__config.database]
64
+
65
+ def __disconnect(self) -> None:
66
+ """Disconnect from MongoDB (private method)."""
67
+ if self.__client:
68
+ self.__client.close()
69
+ self.__client = None
70
+ self.__db = None
71
+
72
+ def __ensure_connected(self) -> None:
73
+ """Ensure MongoDB connection is established."""
74
+ if self.__db is None:
75
+ self.__connect()
76
+
77
+ def create(self, model: InLayersModel, data: Mapping) -> Mapping:
78
+ """Create a new document in MongoDB."""
79
+ self.__ensure_connected()
80
+
81
+ collection_name = get_collection_name_for_model(model.get_model_definition())
82
+ collection = self.__db[collection_name]
83
+ payload = dict(data)
84
+ formatted = format_for_mongo(payload)
85
+
86
+ pk_name = model.get_primary_key_name()
87
+ pk_value = formatted.get(pk_name)
88
+ if pk_value is None:
89
+ # Generate a simple ID - in production you might want UUID
90
+
91
+ pk_value = str(uuid4())
92
+ formatted[pk_name] = pk_value
93
+
94
+ # Use _id as MongoDB's primary key, mapping from model's primary key
95
+ insert_data = {**formatted, "_id": pk_value}
96
+ collection.insert_one(insert_data)
97
+ # Return without _id
98
+ result = {k: v for k, v in insert_data.items() if k != "_id"}
99
+ return result
100
+
101
+ def retrieve(self, model: InLayersModel, id: PrimaryKeyType) -> Mapping | None:
102
+ """Retrieve a document by ID."""
103
+ self.__ensure_connected()
104
+
105
+ collection_name = get_collection_name_for_model(model.get_model_definition())
106
+ collection = self.__db[collection_name]
107
+ doc = collection.find_one({"_id": id})
108
+ if not doc:
109
+ return None
110
+ # Remove _id and return
111
+ result = {k: v for k, v in doc.items() if k != "_id"}
112
+ return result
113
+
114
+ def update(
115
+ self, model: InLayersModel, id: PrimaryKeyType, data: Mapping
116
+ ) -> Mapping:
117
+ """Update a document by ID.
118
+
119
+ Supports partial updates - only the fields provided in data will be updated.
120
+ Returns the full merged object (original + updates).
121
+ """
122
+ self.__ensure_connected()
123
+
124
+ collection_name = get_collection_name_for_model(model.get_model_definition())
125
+ collection = self.__db[collection_name]
126
+
127
+ # Check if document exists
128
+ existing = collection.find_one({"_id": id})
129
+ if not existing:
130
+ raise KeyError(f"Instance with id {id!r} not found")
131
+
132
+ payload = dict(data)
133
+ formatted = format_for_mongo(payload)
134
+
135
+ # Ensure primary key field remains consistent
136
+ pk_name = model.get_primary_key_name()
137
+ formatted[pk_name] = id
138
+
139
+ # Update with _id mapping - use $set for partial update
140
+ update_data = {**formatted, "_id": id}
141
+ collection.update_one({"_id": id}, {"$set": update_data})
142
+
143
+ # Retrieve the updated document to return the full merged object
144
+ updated = collection.find_one({"_id": id})
145
+ if not updated:
146
+ raise KeyError(f"Instance with id {id!r} not found after update")
147
+
148
+ # Return without _id
149
+ result = {k: v for k, v in updated.items() if k != "_id"}
150
+ return result
151
+
152
+ def delete(self, model: InLayersModel, id: PrimaryKeyType) -> None:
153
+ """Delete a document by ID."""
154
+ self.__ensure_connected()
155
+
156
+ collection_name = get_collection_name_for_model(model.get_model_definition())
157
+ collection = self.__db[collection_name]
158
+ collection.delete_one({"_id": id})
159
+
160
+ def search(self, model: InLayersModel, query: ModelSearch) -> ModelSearchResult:
161
+ """Search for documents matching the query."""
162
+ self.__ensure_connected()
163
+
164
+ collection_name = get_collection_name_for_model(
165
+ self.__context.environment, model.get_model_definition()
166
+ )
167
+ collection = self.__db[collection_name]
168
+
169
+ # Build aggregation pipeline
170
+ pipeline = []
171
+
172
+ # Add match stage if there's a query
173
+ if query.query:
174
+ mongo_query = to_mongo(query.query)
175
+ pipeline.extend(mongo_query)
176
+ else:
177
+ pipeline.append({"$match": {}})
178
+
179
+ # Add sort stage if needed
180
+ if query.sort:
181
+ sort_direction = 1 if query.sort.order == SortOrder.asc else -1
182
+ pipeline.append({"$sort": {query.sort.key: sort_direction}})
183
+
184
+ # Add limit stage if needed
185
+ if query.take:
186
+ pipeline.append({"$limit": query.take})
187
+
188
+ # Execute aggregation
189
+ results = list(collection.aggregate(pipeline))
190
+ instances = [{k: v for k, v in doc.items() if k != "_id"} for doc in results]
191
+
192
+ return Box(instances=instances, page=query.page)
193
+
194
+ def bulk_insert(self, model: InLayersModel, data: list[Mapping]) -> None:
195
+ """Bulk insert documents."""
196
+ from pymongo.operations import UpdateOne # noqa: PLC0415
197
+
198
+ self.__ensure_connected()
199
+
200
+ collection_name = get_collection_name_for_model(model.get_model_definition())
201
+ collection = self.__db[collection_name]
202
+ pk_name = model.get_primary_key_name()
203
+
204
+ # Prepare bulk write operations
205
+ operations = []
206
+ for item in data:
207
+ payload = dict(item)
208
+ formatted = format_for_mongo(payload)
209
+
210
+ pk_value = formatted.get(pk_name)
211
+ if pk_value is None:
212
+ pk_value = str(uuid4())
213
+ formatted[pk_name] = pk_value
214
+
215
+ doc = {**formatted, "_id": pk_value}
216
+ operations.append(
217
+ UpdateOne(
218
+ {"_id": pk_value},
219
+ {"$set": doc},
220
+ upsert=True,
221
+ )
222
+ )
223
+
224
+ if operations:
225
+ collection.bulk_write(operations)
226
+
227
+ def bulk_delete(self, model: InLayersModel, ids: list[PrimaryKeyType]) -> None:
228
+ """Bulk delete documents by IDs."""
229
+ self.__ensure_connected()
230
+
231
+ collection_name = get_collection_name_for_model(model.get_model_definition())
232
+ collection = self.__db[collection_name]
233
+ collection.delete_many({"_id": {"$in": ids}})
234
+
235
+ def dispose(self) -> None:
236
+ """Clean up resources."""
237
+ self.__disconnect()
@@ -0,0 +1,43 @@
1
+ from collections.abc import Mapping
2
+ from enum import Enum
3
+ from typing import Any, Literal, Protocol
4
+
5
+
6
+ class SupportedBackend(Enum):
7
+ MongoDB = "mongodb"
8
+ DynamoDB = "dynamodb"
9
+
10
+
11
+ class MongoBackendConfig(Protocol):
12
+ type: Literal[SupportedBackend.MongoDB]
13
+ host: str
14
+ port: int | None
15
+ username: str | None
16
+ password: str | None
17
+ database: str | None
18
+
19
+
20
+ class DynamoDBBackendConfig(Protocol):
21
+ type: Literal[SupportedBackend.DynamoDB]
22
+ region: str | None
23
+ endpoint_url: str | None
24
+ aws_access_key_id: str | None
25
+ aws_secret_access_key: str | None
26
+ boto3: Any | None
27
+
28
+
29
+ BackendConfig = MongoBackendConfig | DynamoDBBackendConfig
30
+
31
+
32
+ class DataNamespace(Enum):
33
+ root = "in_layers_data"
34
+ backends = "in_layers_data_backends"
35
+
36
+
37
+ class InLayersDataConfig(Protocol):
38
+ default: BackendConfig
39
+ model_to_backend: Mapping[str, BackendConfig] | None
40
+
41
+
42
+ class WithInLayersDataConfig(Protocol):
43
+ config: InLayersDataConfig
@@ -0,0 +1,91 @@
1
+ from typing import Protocol
2
+
3
+ from in_layers.core import create_error_object
4
+ from in_layers.core.models.protocols import BackendProtocol, ModelDefinition
5
+
6
+ from .backends.dynamodb.services import DynamoDBBackend
7
+ from .backends.mongodb.services import MongoBackend
8
+ from .protocols import (
9
+ BackendConfig,
10
+ SupportedBackend,
11
+ WithInLayersDataConfig,
12
+ )
13
+
14
+
15
+ class _ExpectedContext(Protocol):
16
+ config: WithInLayersDataConfig
17
+
18
+
19
+ class InLayersDataServices:
20
+ def __init__(self, context: _ExpectedContext):
21
+ self.__context = context
22
+ self.__backend_by_unique_key = {}
23
+ self.__backends = None
24
+
25
+ def __initialize_backend(self, config: BackendConfig) -> BackendProtocol:
26
+ if config.type in [SupportedBackend.MongoDB, SupportedBackend.MongoDB.value]:
27
+ unique = MongoBackend.create_unique_connection_string(config)
28
+ if unique in self.__backend_by_unique_key:
29
+ return self.__backend_by_unique_key[unique]
30
+ backend = MongoBackend(self.__context, config)
31
+ self.__backend_by_unique_key[unique] = backend
32
+ return backend
33
+ elif config.type in [
34
+ SupportedBackend.DynamoDB,
35
+ SupportedBackend.DynamoDB.value,
36
+ ]:
37
+ unique = DynamoDBBackend.create_unique_connection_string(config)
38
+ if unique in self.__backend_by_unique_key:
39
+ return self.__backend_by_unique_key[unique]
40
+ backend = DynamoDBBackend(self.__context, config)
41
+ self.__backend_by_unique_key[unique] = backend
42
+ return backend
43
+ else:
44
+ raise ValueError(f"Unsupported backend type: {config.type}")
45
+
46
+ def __initialize_backends(self):
47
+ if self.__backends is not None:
48
+ return
49
+ config = self.__context.config.in_layers_data
50
+ default_backend_config = config.default
51
+ model_to_backend_config = getattr(config, "model_to_backend", {}) or {}
52
+
53
+ self.__backends = {"default": self.__initialize_backend(default_backend_config)}
54
+ for model_key, backend_config in model_to_backend_config.items():
55
+ backend_instance = self.__initialize_backend(backend_config)
56
+ self.__backends[model_key] = backend_instance
57
+
58
+ def __get_backend_for_model(self, meta: ModelDefinition) -> BackendProtocol:
59
+ self.__initialize_backends()
60
+ model_key_full = f"{meta.domain}.{meta.plural_name}"
61
+ if model_key_full in self.__backends:
62
+ return self.__backends[model_key_full]
63
+ if meta.domain in self.__backends:
64
+ return self.__backends[meta.domain]
65
+ return self.__backends["default"]
66
+
67
+ def get_model_backend(self, model_definition: ModelDefinition) -> BackendProtocol:
68
+ backend = self.__get_backend_for_model(model_definition)
69
+ if not backend:
70
+ raise ValueError(
71
+ f"No backend found for model {model_definition.domain}.{model_definition.plural_name}"
72
+ )
73
+ return backend
74
+
75
+ def dispose(self):
76
+ for backend in self.__backends.values():
77
+ try:
78
+ backend.dispose()
79
+ except Exception as e:
80
+ error_obj = create_error_object(
81
+ "DATA_SERVICES_DISPOSE_ERROR", "Error disposing backend", e
82
+ )
83
+ log = self.__context.log.get_inner_logger("dispose")
84
+ log.warn("Error disposing backend", error_obj)
85
+
86
+ self.__backends = None
87
+ self.__backend_by_unique_key = {}
88
+
89
+
90
+ def create(context: _ExpectedContext) -> InLayersDataServices:
91
+ return InLayersDataServices(context)
@@ -0,0 +1,79 @@
1
+ Metadata-Version: 2.4
2
+ Name: in-layers-data
3
+ Version: 0.4.11
4
+ Summary: A data layer for In Layers Core. (in-layers-core). Part of the Node in Layers ecosystem
5
+ License: GPLv3
6
+ Keywords: layers,python,in-layers,node,domains,data,models,orm
7
+ Author: Mike Cornwell
8
+ Author-email: mike@mikecornwell.com
9
+ Requires-Python: >=3.13,<4.0
10
+ Classifier: License :: Other/Proprietary License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Programming Language :: Python :: 3.14
14
+ Requires-Dist: httpx (>=0.28.1,<0.29.0)
15
+ Requires-Dist: in-layers-core (>=0.4.2,<0.5.0)
16
+ Requires-Dist: pydantic (>=2.12.5,<3.0.0)
17
+ Requires-Dist: python-box (>=7,<9)
18
+ Description-Content-Type: text/markdown
19
+
20
+ # In Layers Data
21
+ A data layer for the In Layers Core framework.
22
+
23
+ NOTE: There are no explicit dependencies on any database. To use a specific database you must install it in your own system. These databases are "imported in" as the databases are actually used at runtime.
24
+
25
+ ## How To Use
26
+ 1. Install in-layers-data
27
+ 1. Set `"in_layers_data"` to the `in_layers_core.models.model_backend` property
28
+ 1. Add `in_layers_data` configuration to your config
29
+ 1. Install database libraries to use. Example: `pymongo` or `boto3`
30
+
31
+ ### Configuration Example
32
+ ```python
33
+ # config_base.py
34
+ from box import Box
35
+ def get_base_config():
36
+ return Box(
37
+ ...,
38
+ in_layers_core=Box(
39
+ ...
40
+ models=Box(
41
+ model_backend="in_layers_data",
42
+ )
43
+ ),
44
+ in_layers_data=Box(
45
+ default=Box(
46
+ type="mongodb"
47
+ # Connection information here
48
+ ),
49
+ # Optional: Set "domain" or "domain.ModelPluralNames" to a specific database configuration.
50
+ # model_to_backend=Box(
51
+ # "domain.ModelPluralNames"=Box(
52
+ # type="mongodb",
53
+ # host="different-host"
54
+ # )
55
+ # )
56
+ )
57
+ )
58
+
59
+
60
+ ```
61
+
62
+ ## Key Features
63
+ - Drop in, Swappable Databases
64
+ - Multi-database support
65
+ - Low dependencies
66
+
67
+ ## Databases Supported
68
+ - Mongodb
69
+ - Dynamodb
70
+
71
+ ## Database Info
72
+ ### Mongo
73
+ Mongodb requires `pymongo`
74
+
75
+ ### Dynamodb
76
+ Dynamodb requires `boto3`
77
+
78
+ #### Important
79
+ Dynamodb is very poor at performing search queries. While this is implemented, it is not-recommended for use. Instead use the retrieve.
@@ -0,0 +1,14 @@
1
+ in_layers/__init__.py,sha256=VyrcAibWxTcBorfOipl5a1Vg9ZY8hOQt2T-lsWOGFps,173
2
+ in_layers/data/__init__.py,sha256=eDl_a3x2QVMUauUSuTw0if180sdSER6DHYAip546d5M,387
3
+ in_layers/data/backends/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ in_layers/data/backends/dynamodb/__init__.py,sha256=q5OQ4ygMLYHGV_L30K4HSrVk10dROgNTrjUg32Ki2XM,114
5
+ in_layers/data/backends/dynamodb/libs.py,sha256=-c_gV_i_r2uy-HxjK0sRUW0N3ihjGtLGN-5fCBXM7jM,2762
6
+ in_layers/data/backends/dynamodb/services.py,sha256=F1wGdehucZ-tS4XzaAsXVRTQ_EZneSOgIzWbL-gBcsQ,13580
7
+ in_layers/data/backends/mongodb/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
8
+ in_layers/data/backends/mongodb/libs.py,sha256=ukv18zwR3amLh_rGmgIrGzc-DKoST4aZRs3OzPHLRSw,8095
9
+ in_layers/data/backends/mongodb/services.py,sha256=Xct2HElasiUGU2ninyFWs91tN24LfGaQ1e4FofmqZkY,8026
10
+ in_layers/data/protocols.py,sha256=yIH2Bu6K11cTBws-IrnopKB4-L_ltf6yLWDZAWec844,960
11
+ in_layers/data/services.py,sha256=qPJCt4KzL7qfXWzXNzi8pnoeD7QSoZY-NpCB6KGKi74,3616
12
+ in_layers_data-0.4.11.dist-info/METADATA,sha256=stkddU5tZ7jsbwuC_OjggApyG4CjbkLaCS4zQedUxew,2367
13
+ in_layers_data-0.4.11.dist-info/WHEEL,sha256=zp0Cn7JsFoX2ATtOhtaFYIiE2rmFAD4OcMhtUki8W3U,88
14
+ in_layers_data-0.4.11.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.2.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any