tts-data-utils 0.7.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. tts_data_utils/__init__.py +0 -0
  2. tts_data_utils/core/__init__.py +0 -0
  3. tts_data_utils/core/container_history.py +79 -0
  4. tts_data_utils/core/data_container.py +2102 -0
  5. tts_data_utils/core/data_item.py +487 -0
  6. tts_data_utils/core/diff.py +155 -0
  7. tts_data_utils/core/generic.py +78 -0
  8. tts_data_utils/core/lorem_utils.py +206 -0
  9. tts_data_utils/core/visual_diff.py +148 -0
  10. tts_data_utils/invulnerable_data_manager/__init__.py +0 -0
  11. tts_data_utils/invulnerable_data_manager/batch.py +176 -0
  12. tts_data_utils/invulnerable_data_manager/invulnerable_data_manager.py +171 -0
  13. tts_data_utils/invulnerable_data_manager/utilities.py +151 -0
  14. tts_data_utils/multimission/__init__.py +0 -0
  15. tts_data_utils/multimission/alarms.py +109 -0
  16. tts_data_utils/multimission/eha.py +253 -0
  17. tts_data_utils/multimission/evr.py +340 -0
  18. tts_data_utils/multimission/evr_gaps.py +148 -0
  19. tts_data_utils/multimission/expected_lad.py +161 -0
  20. tts_data_utils/multimission/planning_rule.py +106 -0
  21. tts_data_utils/multimission/taco.py +53 -0
  22. tts_data_utils/test/__init__.py +0 -0
  23. tts_data_utils/test/core/__init__.py +0 -0
  24. tts_data_utils/test/core/test_data_container.py +403 -0
  25. tts_data_utils/test/core/test_data_item.py +92 -0
  26. tts_data_utils/test/core/test_diff.py +173 -0
  27. tts_data_utils/test/core/test_files/diff/diff_dict/actual_diff.csv +5135 -0
  28. tts_data_utils/test/core/test_files/diff/diff_dict/expected_diff.csv +5135 -0
  29. tts_data_utils/test/core/test_files/diff/diff_dict/left.csv +637 -0
  30. tts_data_utils/test/core/test_files/diff/diff_dict/right.csv +637 -0
  31. tts_data_utils/test/core/test_files/diff/diff_list/actual_diff.csv +5135 -0
  32. tts_data_utils/test/core/test_files/diff/diff_list/expected_diff.csv +5135 -0
  33. tts_data_utils/test/core/test_files/diff/diff_list/left.csv +637 -0
  34. tts_data_utils/test/core/test_files/diff/diff_list/right.csv +637 -0
  35. tts_data_utils/test/core/test_files/diff/diff_list_size/actual_diff.csv +33 -0
  36. tts_data_utils/test/core/test_files/diff/diff_list_size/expected_diff.csv +33 -0
  37. tts_data_utils/test/core/test_files/diff/diff_list_size/left.csv +637 -0
  38. tts_data_utils/test/core/test_files/diff/diff_list_size/right.csv +637 -0
  39. tts_data_utils/test/core/test_files/diff/diff_simple_types/actual_diff.csv +17894 -0
  40. tts_data_utils/test/core/test_files/diff/diff_simple_types/expected_diff.csv +17894 -0
  41. tts_data_utils/test/core/test_files/diff/diff_simple_types/left.csv +637 -0
  42. tts_data_utils/test/core/test_files/diff/diff_simple_types/right.csv +637 -0
  43. tts_data_utils/test/core/test_files/diff/diff_type/actual_diff.csv +5135 -0
  44. tts_data_utils/test/core/test_files/diff/diff_type/expected_diff.csv +5135 -0
  45. tts_data_utils/test/core/test_files/diff/diff_type/left.csv +637 -0
  46. tts_data_utils/test/core/test_files/diff/diff_type/right.csv +637 -0
  47. tts_data_utils/test/core/test_files/diff/same_not_same/actual_diff.csv +5135 -0
  48. tts_data_utils/test/core/test_files/visdiff/adjacent_insert_delete.html +1581 -0
  49. tts_data_utils/test/core/test_files/visdiff/insert_and_delete.html +1581 -0
  50. tts_data_utils/test/core/test_files/visdiff/no_diff.html +1581 -0
  51. tts_data_utils/test/core/test_files/visdiff/overlapping_insert_delete.html +1541 -0
  52. tts_data_utils/test/core/test_files/visdiff/replace.html +1581 -0
  53. tts_data_utils/test/core/test_files/visdiff/replace_with_delete.html +1581 -0
  54. tts_data_utils/test/core/test_files/visdiff/simple_delete.html +1581 -0
  55. tts_data_utils/test/core/test_files/visdiff/simple_insert.html +1581 -0
  56. tts_data_utils/test/core/test_files/visdiff/stressing_case.html +2061 -0
  57. tts_data_utils/test/core/test_filters.py +319 -0
  58. tts_data_utils/test/core/test_generic_container.py +22 -0
  59. tts_data_utils/test/core/test_history.py +44 -0
  60. tts_data_utils/test/core/test_visdiff.py +193 -0
  61. tts_data_utils/test/invulnerable_data_manager/test_invulnerable_data_manager.py +150 -0
  62. tts_data_utils/test/multimission/__init__.py +0 -0
  63. tts_data_utils/test/multimission/evr/__init__.py +0 -0
  64. tts_data_utils/test/multimission/evr/test_evr.py +29 -0
  65. tts_data_utils/test/multimission/evr/test_files/actual_evr_gaps.csv +12 -0
  66. tts_data_utils/test/multimission/evr/test_files/evrs_with_gaps.csv +30 -0
  67. tts_data_utils/test/multimission/evr/test_files/expected_evr_gaps.csv +12 -0
  68. tts_data_utils/util.py +104 -0
  69. tts_data_utils-0.7.3.dist-info/METADATA +81 -0
  70. tts_data_utils-0.7.3.dist-info/RECORD +72 -0
  71. tts_data_utils-0.7.3.dist-info/WHEEL +5 -0
  72. tts_data_utils-0.7.3.dist-info/top_level.txt +1 -0
@@ -0,0 +1,2102 @@
1
+ #Python Imports
2
+ from abc import ABC, abstractmethod
3
+ from copy import copy, deepcopy
4
+ from datetime import datetime, timedelta
5
+ import hashlib
6
+ import inspect
7
+ import json
8
+ from math import isnan
9
+ import os
10
+ import pandas as pd
11
+ from difflib import SequenceMatcher
12
+ import pdb
13
+ import re
14
+ import sys
15
+ from tabulate import tabulate
16
+ from itertools import product
17
+
18
+ #JPL Imports
19
+ from tts_html_utils.core.components.table import PowerTable
20
+ from tts_utilities.logger import create_logger
21
+
22
+ #This Library Imports
23
+ from tts_data_utils.core.data_item import DataItem
24
+
25
+
26
+ log = create_logger(__name__)
27
+
28
+ #TO DO: Move this to utils
29
+ def find_bad_utf8_characters(filepath):
30
+ """
31
+ Helper function to identify non-UTF-8 characters in CSV files, common when
32
+ interacting with Windows-generated Microsoft Excel files.
33
+
34
+ **The Problem:**
35
+ When CSVs are saved via Excel on Windows, non-UTF-8 characters—like curled
36
+ quotation marks, single-character arrows, and degree symbols—are often added.
37
+ Attempting to read these into Pandas causes a `UnicodeDecodeError`.
38
+
39
+ **The Solution:**
40
+ This script allows developers to catch that exception, read the file in binary
41
+ mode, and report the exact line and byte offset of the first error to
42
+ facilitate cleaning.
43
+
44
+ **Future Improvements:**
45
+ * Amend to repair the file automatically (referencing the M20 dictionary
46
+ input management logic).
47
+ * Report all encoding errors instead of just the first.
48
+
49
+ See ticket #31 ( TO DO: Migrate out of JPL-internal issues))
50
+
51
+ :param filepath: Path to the CSV file to be checked.
52
+ :type filepath: str or pathlib.Path
53
+ """
54
+ with open(filepath, 'rb') as file:
55
+ for line_num, line_bytes in enumerate(file, start=1):
56
+ try:
57
+ line_bytes.decode('utf-8')
58
+ except UnicodeDecodeError as e:
59
+ print(f"UnicodeDecodeError on line {line_num}")
60
+ print(f" Problem at byte offset: {e.start}")
61
+ print(f" Invalid byte: {line_bytes[e.start]:#04x}")
62
+
63
+ # Print context (optional)
64
+ context = line_bytes[max(0, e.start-10):e.start+10]
65
+ print(f" Context (raw bytes): {context}")
66
+ print(f" Full Line (raw bytes): {line_bytes}")
67
+
68
+ # TO DO: Make it so instead of just reporting and bailing, this
69
+ # function repairs the bad byte. They're essentially never due to
70
+ # file corruption. Only due to Microsoft thinking they're more clever
71
+ # than everyone else and using curly quotes or similar.
72
+ # In the M20 dictionary input management code, we handled this more
73
+ # gracefully, and this should be built into a general soltion to do the same.
74
+ sys.exit()
75
+
76
+ class DataContainer(ABC):
77
+ """
78
+ Primary (abstract) class for this library. Provides representation of 2D data with
79
+ extension hooks for easy definition of quality-of-life features for any bespoke
80
+ data type across projects.
81
+
82
+ **Concept:**
83
+ Allows for easy tabular representation in terminals and HTML, playing nicely with
84
+ `html_utils` to provide easy reporting of tabular data and nested tabular data.
85
+
86
+ When defining an extension of this class, a `DataItem` class is also provided,
87
+ which controls the expected columns in each row.
88
+
89
+ Each row of the 2D data is represented by an instance of the associated `DataItem`
90
+ class, stored in `self.records`. Most dunder methods have been defined such that
91
+ this class behaves like a list (mapping to `self.records`), but carries the
92
+ container's metadata and history along with it.
93
+
94
+ **TO DO:** Provide gallery of examples of outputs (see ticket #34 TO DO: Migrate out of JPL-internal issues)
95
+
96
+ :param raw_data: 2D data to be transformed into DataContainer.
97
+ :type raw_data: list[dict], optional
98
+ :param subcontainers: List of dictionaries where key is a label and value is a
99
+ DataContainer. Must match length of raw_data.
100
+ :type subcontainers: list[dict[str, DataContainer]], optional
101
+ :param csv_path: Path for CSV to be transformed into DataContainer.
102
+ :type csv_path: Path | str, optional
103
+ :param xlsx_path: Path for XLSX to be transformed into DataContainer.
104
+ :type xlsx_path: Path | str, optional
105
+ :param django_records: Django object containing data to be transformed.
106
+ :type django_records: QuerySet, optional
107
+ :param metadata: Arbitrary user information to be carried with the container.
108
+ :type metadata: dict, optional
109
+ :param name: Name of the DataContainer instance.
110
+ :type name: str, optional
111
+ :param cast_fields: If True, attempts to force data into types defined in DataItem.
112
+ :type cast_fields: bool
113
+ :param validate: If True, validates inputs against DataItem's valid keys/types.
114
+ :type validate: bool
115
+ :param lorem: If provided as an integer, generates that many rows of dummy data.
116
+ :type lorem: int, optional
117
+ """
118
+ DATA_ITEM_CLS = None
119
+ """Associated DataItem that must be defined alongside a DataContainer."""
120
+
121
+ DO_NOT_DIFF = []
122
+ """Keys to ignore when running self.diff."""
123
+
124
+ def __init__(self, raw_data=None, subcontainers=None, csv_path=None, xlsx_path=None, django_records=None, metadata=None, name=None, cast_fields=False, validate=True, lorem=None, **kwargs):
125
+ self.name = self.NAME if name is None else name
126
+
127
+ if metadata is not None:
128
+ metadata = {k:v for k, v in metadata.items() if k[0] != '_' and k != 'dictionary'}
129
+
130
+ if '_repr_cols' not in self.__dict__.keys():
131
+ self._repr_cols = [x for x, _ in self.DATA_ITEM_CLS.DICT_VALID_KEYS]
132
+ if '_csv_cols' not in self.__dict__.keys():
133
+ self._csv_cols = [x for x, _ in self.DATA_ITEM_CLS.DICT_VALID_KEYS]
134
+ if '_repr_filters' not in self.__dict__.keys():
135
+ self._repr_filters = []
136
+ if 'name' not in self.__dict__.keys():
137
+ self.name = self.NAME
138
+
139
+ mutually_exclusive_kwargs = [raw_data, csv_path, xlsx_path, django_records, lorem]
140
+ if sum([m is not None for m in mutually_exclusive_kwargs]) > 1:
141
+ raise Exception('Cannot have more than one data source.')
142
+
143
+ is_django = False
144
+ if csv_path is not None:
145
+ # did you know that DataFrame.fillna(None) doens't work???
146
+ try:
147
+ raw_data = self.read_csv(csv_path)
148
+ except UnicodeDecodeError:
149
+ find_bad_utf8_characters(csv_path)
150
+ for row in raw_data:
151
+ for k, v in row.items():
152
+ if not isinstance(v, (int, float)): continue
153
+ if isnan(v): row[k] = None
154
+ elif xlsx_path is not None:
155
+ raw_data = self.read_xlsx(xlsx_path)
156
+ elif raw_data is not None:
157
+ pass
158
+ elif django_records is not None:
159
+ raw_data = django_records
160
+ is_django = True
161
+ elif lorem is not None:
162
+ # Import here to avoid circular imports
163
+ from tts_data_utils.core.lorem_utils import generate_lorem_data_for_item
164
+
165
+ # Generate lorem ipsum data based on the DATA_ITEM_CLS
166
+ if not isinstance(lorem, int) or lorem <= 0:
167
+ lorem = 10 # Default to 10 records if not specified correctly
168
+ raw_data = generate_lorem_data_for_item(self.DATA_ITEM_CLS, num_records=lorem)
169
+ else:
170
+ raw_data = []
171
+
172
+ if subcontainers is None:
173
+ self.records = [self.DATA_ITEM_CLS(r, cast_fields=cast_fields, validate=validate, is_django=is_django) for r in raw_data]
174
+ elif len(raw_data) != len(subcontainers):
175
+ raise Exception('"subcontainers" must be the same length as data that forms DataItems (e.g raw_data, data at csv_path, data coming from django object)')
176
+ else:
177
+ self.records = [self.DATA_ITEM_CLS(r, cast_fields=cast_fields, validate=validate, is_django=is_django, subcontainers=s) for r, s in zip(raw_data, subcontainers)]
178
+
179
+ self.metadata = metadata
180
+ # Avoids circular dependency. Probably a better way to do it
181
+ # but here we are...
182
+ from tts_data_utils.core.container_history import DataContainerHistoryContainer
183
+ if not isinstance(self, DataContainerHistoryContainer):
184
+ self.history = DataContainerHistoryContainer(self.name, self.metadata)
185
+ self.history._add_record({
186
+ 'Action': 'Initialized',
187
+ 'Description': str(self.metadata),
188
+ 'Ending Count': len(self.records),
189
+ 'Starting Count': '0',
190
+ 'Percent Remaining': 'NA'
191
+ })
192
+ else:
193
+ # Avoid infinite recursion.
194
+ # I tried once, but never got to the bottom of it.
195
+ self.history = None
196
+
197
+
198
+ ##################################################################################
199
+ # Dexter-specific attributes, consider reorganizing this so not every DataItem gets this
200
+ ##################################################################################
201
+ self._bypass_validation = False
202
+ self._sub_container = False
203
+
204
+ def _impl_init(self):
205
+ """Internal setup hook for subclasses."""
206
+ return
207
+
208
+ def _impl_populate(self):
209
+ """Internal data population hook for subclasses."""
210
+ return
211
+
212
+ @classmethod
213
+ @property
214
+ @abstractmethod
215
+ def NAME(cls):
216
+ """Name of the data type being contained, i.e. 'evr', 'transpire_commands'."""
217
+ raise NotImplementedError
218
+
219
+
220
+ @property
221
+ def repr_cols(self):
222
+ """Columns to be used for representations (terminal, HTML, etc.)."""
223
+ if '_repr_cols' in self.__dict__.keys():
224
+ return self._repr_cols
225
+ else:
226
+ return [x for x, _ in self.DATA_ITEM_CLS.DICT_VALID_KEYS]
227
+
228
+ def docx_table(self, template=None):
229
+ """
230
+ Produces a Microsoft Word table representation.
231
+
232
+ :param template: Path to an optional template docx for styling.
233
+ :return: Rendered DocxTable object.
234
+ """
235
+
236
+ #TO DO: Fix this without a circular dependency. Data utils can't require
237
+ #Papertrail because Papertrail already depends on data_utils
238
+ table_builder = DocxTable(template, self.records, headers=self._repr_cols, row_styles=[r.default_rich_text_row_style for r in self.records])
239
+ return table_builder.render()
240
+
241
+ def power_table(self, superheader=None, columns=None, bypass_styles=False, row_styles=None, cell_styles=None, **kwargs):
242
+ """
243
+ Produce a rich, interactive HTML table representation of this DataContainer.
244
+
245
+ **Concept:**
246
+ This method integrates with `html_utils` to translate the 2D records into a
247
+ `PowerTable`. It handles complex nesting by recursively calling `power_table`
248
+ on any subcontainers linked to specific rows.
249
+
250
+ :param superheader: Title row spanning the full width of the table.
251
+ :type superheader: str
252
+ :param columns: Labels to include. Defaults to `self.repr_cols`.
253
+ :type columns: list[str]
254
+ :param bypass_styles: If True, default CSS and row-level styles are ignored.
255
+ :type bypass_styles: bool
256
+ :param row_styles: Custom CSS for each row. Must match `self.records` length.
257
+ :type row_styles: list[dict[str, str]]
258
+ :param cell_styles: Custom CSS for each cell. Must match `self.records` length.
259
+ :type cell_styles: list[list[dict[str, str]]]
260
+ :param kwargs: Passthrough arguments for PowerTable (e.g., `id`, `add_filters`).
261
+ :return: A rendered PowerTable component.
262
+ """
263
+
264
+ # TO DO: Rethink how we handle repr_cols here when you're not so braindead
265
+ row_data = [(r.values, [subcontainer_obj.power_table(subcontainer_name) for subcontainer_name, subcontainer_obj in r.subcontainers.items()]) for r in self.records]
266
+
267
+ if columns is not None:
268
+ repr_cols = columns
269
+ elif self.repr_cols:
270
+ repr_cols = self.repr_cols
271
+ elif len(self.records):
272
+ repr_cols = []
273
+ for r in row_data: repr_cols += [k for k in r[0].keys() if not k.startswith('_')]
274
+ repr_cols = list(set(repr_cols))
275
+ else:
276
+ repr_cols = self.repr_cols if self.repr_cols else [k for k, _ in self.DATA_ITEM_CLS.DICT_VALID_KEYS]
277
+
278
+ if bypass_styles:
279
+ row_styles = [{} for r in self.records]
280
+ elif row_styles is not None:
281
+ if len(row_styles) != len(self.records):
282
+ raise ValueError("Row styles must match the number of records")
283
+ else:
284
+ row_styles = [{'background-color': '#EEEEEE'} if ii%2 else {} for ii in range(len(self.records))]
285
+ row_styles = [{**rs, **r.default_html_row_style} for r, rs in zip(self.records, row_styles)]
286
+
287
+ if bypass_styles:
288
+ cell_styles = [[{} for k in r.printable_values.keys()] for r in self.records]
289
+ elif cell_styles is not None:
290
+ pass
291
+ else:
292
+ cell_styles = [[r.default_html_cell_styles.get(k,{}) for k in self._repr_cols] for r in self.records]
293
+
294
+ table = PowerTable(
295
+ column_fields=repr_cols,
296
+ row_data=row_data,
297
+ row_styles=row_styles,
298
+ cell_styles=cell_styles,
299
+ **kwargs
300
+ )
301
+ if superheader:
302
+ table.add_superheader(superheader)
303
+ table.add_header(column_names=repr_cols)
304
+ return table
305
+
306
+ @property
307
+ def default_html_row_style(self):
308
+ """Returns default CSS dictionary for HTML rows."""
309
+ return {}
310
+
311
+ @property
312
+ def default_time_label(self):
313
+ """Returns the primary key used for time-based operations."""
314
+ return None
315
+
316
+ @property
317
+ def valid(self):
318
+ """Returns True if all records pass validation (or if validation is bypassed)."""
319
+ return self._bypass_validation or all(_.valid for _ in self)
320
+
321
+ @property
322
+ def source(self):
323
+ """Returns a list of raw source dictionaries for all contained records."""
324
+ return [_.source for _ in self]
325
+
326
+ def __iter__(self):
327
+ """
328
+ Iterates through the container's records.
329
+
330
+ Allows the DataContainer to be used in loops:
331
+ `for record in container: ...`
332
+ """
333
+ for r in self.records:
334
+ yield r
335
+
336
+ def __len__(self):
337
+ """
338
+ Returns the total number of records currently held in the container.
339
+ """
340
+ return len(self.records)
341
+
342
+ def __getitem__(self, ii):
343
+ """
344
+ Provides flexible access to data using indexing, slicing, or column keys.
345
+
346
+ **Supported Behaviors:**
347
+ * **Integer (`int`):** Returns the specific `DataItem` at that index.
348
+ * **Slice (`slice`):** Returns a new `DataContainer` containing the subset of records.
349
+ * **String (`str`):** Returns a list of all values found in the specified column.
350
+ * **List (`list[str]`):** (Not yet implemented) Intended to return a container with subset columns.
351
+
352
+ :param ii: The index, slice, or column name requested.
353
+ :type ii: int | slice | str | list[str]
354
+ :return: A DataItem, a new DataContainer, or a list of values.
355
+ """
356
+ if isinstance(ii, slice): # Handle slicing
357
+ new_obj = self._copy(self.records[ii.start:ii.stop:ii.step])
358
+ indexes = f'{ii.start}:{ii.stop}:{ii.step}'
359
+ elif isinstance(ii, int): # Handle index
360
+ return self.records[ii]
361
+ elif isinstance(ii, str):
362
+ if isinstance(self.records, list):
363
+ return [r[ii] for r in self.records]
364
+ else:
365
+ return self.records[ii]
366
+ elif isinstance(ii, list):
367
+ raise NotImplementedError("Lists not implemented yet")
368
+ else:
369
+ raise TypeError("Invalid argument type")
370
+
371
+ # Performance check and history logging for sliced objects
372
+ if len(self.records):
373
+ percent_remaining = f'{len(new_obj.records)/len(self.records)*100:.2f}'
374
+ else:
375
+ percent_remaining = 'NA'
376
+
377
+ new_obj.history._add_record({
378
+ 'Action': 'Get Item',
379
+ 'Description': f'Sliced to {indexes}',
380
+ 'Ending Count': len(new_obj.records),
381
+ 'Starting Count': len(self.records),
382
+ 'Percent Remaining': percent_remaining
383
+ })
384
+
385
+ return new_obj
386
+
387
+ def __str__(self):
388
+ """
389
+ Returns the human-readable name of the container.
390
+ """
391
+ return self.name
392
+
393
+ def __repr__(self):
394
+ """
395
+ Produces a formatted ASCII grid table for terminal display.
396
+
397
+ **Concept:**
398
+ Uses the `tabulate` library to render rows. It filters for columns
399
+ defined in `self.repr_cols` and uses printable_values to ensure proper formatting.
400
+ """
401
+ if self._repr_filters:
402
+ for repr_filter in self._repr_filters:
403
+ records = self.records
404
+ else:
405
+ records = self.records
406
+
407
+ # Use printable_values instead of raw values to ensure proper formatting
408
+ rows = [{k: v for k,v in r.printable_values.items() if k in self.repr_cols} for r in records]
409
+
410
+ if len(rows):
411
+ headers = 'keys'
412
+ else:
413
+ headers = self.repr_cols
414
+
415
+ return tabulate(rows, headers=headers, tablefmt="grid")
416
+
417
+ def _repr_html_(self):
418
+ """
419
+ IPython/Jupyter hook to automatically render an interactive
420
+ PowerTable when the container is displayed in a notebook.
421
+ """
422
+ table = self.power_table()
423
+ return table.render()
424
+
425
+ def __add__(self, other, sort_by=None):
426
+ """
427
+ Concatenates two DataContainers together using the '+' operator.
428
+
429
+ **The Concept:**
430
+ This allows for intuitive dataset combination (e.g., `combined = list_a + list_b`).
431
+ The operation creates a new container copy, preserves the history of the
432
+ original, and logs the merge event with updated record counts.
433
+
434
+ :param other: The other DataContainer to append to this one.
435
+ :type other: DataContainer
436
+ :param sort_by: (Placeholder) Optional key to sort by after merging.
437
+ :return: A new DataContainer containing records from both parents.
438
+ """
439
+ new_obj = self._copy(self.records + other.records)
440
+
441
+ if len(self.records):
442
+ percent_remaining = f'{len(new_obj.records)/len(self.records)*100:.2f}'
443
+ else:
444
+ percent_remaining = 'NA'
445
+
446
+ new_obj.history._add_record({
447
+ 'Action': 'Merged',
448
+ 'Description': 'TBD, need to figure out how to represent this',
449
+ 'Ending Count': len(new_obj.records),
450
+ 'Starting Count': len(self.records),
451
+ 'Percent Remaining': percent_remaining
452
+ })
453
+ return new_obj
454
+
455
+
456
+ def table(self, columns=None):
457
+ """
458
+ Explicitly prints the ASCII grid table representation to standard output.
459
+
460
+ **Concept:**
461
+ While `__repr__` handles automatic display in the terminal, this method
462
+ allows for programmatic printing with an optional subset of columns.
463
+
464
+ :param columns: List of column labels to include. Defaults to `self.repr_cols`.
465
+ :type columns: list[str], optional
466
+ """
467
+ if columns is None: columns = self.repr_cols
468
+
469
+ # Use printable_values instead of raw values to ensure proper formatting
470
+ rows = [{k: v for k,v in r.printable_values.items() if k in columns} for r in self.records]
471
+
472
+ if len(rows):
473
+ headers = 'keys'
474
+ else:
475
+ headers = self.repr_cols
476
+
477
+ print(tabulate(rows, headers=headers, tablefmt="grid"))
478
+
479
+ def _diff(self, left, right):
480
+ """
481
+ Internal stub for shared diffing logic.
482
+ Override or implement to provide custom comparison behaviors.
483
+ """
484
+ return
485
+ #make this the common diff
486
+
487
+ def diff(self, left='48vf34VD)$', right='48vf34VD)$', name='', ancestors='', diff_container=None, summarize=False, debug=False, do_not_diff_keys=[], ignore=[], float_tol=1e-10):
488
+ """
489
+ Generates a DiffContainer with a comprehensive comparison between two objects.
490
+ Recursively trees down through all attributes until the structures are fully diffed.
491
+
492
+ **The Concept:**
493
+ This method is the backbone of the library's regression testing suite. It is designed
494
+ to compare a runtime DataContainer against a "vetted" baseline (typically from a CSV).
495
+ It identifies missing keys, mismatched values, and type discrepancies across
496
+ nested lists and dictionaries.
497
+
498
+ **Handling Differently Ordered Data:**
499
+ Note that this method does not yet handle reordered containers gracefully; it is
500
+ optimized for structures that are expected to be very similar in sequence.
501
+
502
+ **The Null Guard:**
503
+ The default value '48vf34VD)$' is used instead of None to allow `None` to be
504
+ passed as a valid value to be diffed without triggering the "missing argument" logic.
505
+
506
+ :param left: The primary value or container to compare.
507
+ :param right: The second value or container to compare. If omitted, `self` is
508
+ treated as `left` and the first argument is treated as `right`.
509
+ :param name: Internal tracker for the current field name (used in recursion).
510
+ :param ancestors: Internal tracker for the breadcrumb path (used in recursion).
511
+ :param diff_container: The accumulator for diff results.
512
+ :param summarize: If True, returns a boolean (True if all match) instead of the container.
513
+ :param do_not_diff_keys: Keys to skip (useful for history or dynamic IDs).
514
+ :param ignore: Output paths to prune from the final results.
515
+ :param float_tol: Maximum allowance for floating-point precision drift.
516
+ :return: A DiffContainer object or a boolean result.
517
+ """
518
+
519
+ # Logic to handle self-diffing if only one argument is provided
520
+ if left == '48vf34VD)$':
521
+ raise Exception('Need something to diff against!')
522
+ if right == '48vf34VD)$':
523
+ # This allows a user to call self.diff(other_obj).
524
+ # self becomes left, and the argument becomes right.
525
+ right = left
526
+ left = self
527
+
528
+ if isinstance(ignore, str): ignore = [ignore]
529
+
530
+ ancestors += '/' + name
531
+
532
+ if len(ancestors) >= 2:
533
+ if ancestors[:2] == '//': ancestors = ancestors[1:]
534
+
535
+ # Import locally to avoid circular dependency issues
536
+ if diff_container is None:
537
+ from tts_data_utils.core.diff import DiffContainer
538
+ diff_container = DiffContainer('tbd', 'tdb')
539
+
540
+ # 1. Compare Types
541
+ if type(left) != type(right):
542
+ left_str = f'Type: {type(left).__name__}'
543
+ right_str = f'Type: {type(right).__name__}'
544
+ typename = 'various'
545
+ same = False
546
+
547
+ # 2. Compare Floats with Tolerance
548
+ elif isinstance(left, float):
549
+ left_str = str(left)
550
+ right_str = str(right)
551
+ typename = f'float ({float_tol} diff tolerance)'
552
+ if abs(left - right) < float_tol:
553
+ same = True
554
+ else:
555
+ same = False
556
+
557
+ # 3. Compare Base Types (Int, Bool, Str, Datetime)
558
+ elif isinstance(left, (int, bool, str, datetime)):
559
+ left_str = str(left)
560
+ right_str = str(right)
561
+ typename = type(left).__name__
562
+ if left == right:
563
+ same = True
564
+ else:
565
+ same = False
566
+
567
+ # 4. Compare via Identity (Fallback for complex objects)
568
+ elif left is right:
569
+ left_str = f'Same memory location'
570
+ right_str = f'Same memory location'
571
+ same = True
572
+ typename = type(left).__name__
573
+
574
+ # 5. Recursive List Comparison
575
+ elif isinstance(left, list):
576
+ same = True
577
+ left_str = 'Children All Same'
578
+ right_str = 'Children All Same'
579
+ typename = 'list'
580
+ if len(left) != len(right):
581
+ same = False
582
+ left_str = f'List size differs ({len(left)})'
583
+ right_str = f'List size differs ({len(right)})'
584
+ else:
585
+ for ii, (l, r) in enumerate(zip(left, right)):
586
+ ii_same = self.diff(l, r, name=str(ii), ancestors=ancestors, diff_container=diff_container, summarize=True, debug=debug)
587
+ if not ii_same:
588
+ same = False
589
+ left_str = 'Children Differ'
590
+ right_str = 'Children Differ'
591
+
592
+ # 6. Recursive Dictionary Comparison
593
+ elif isinstance(left, dict):
594
+ same = True
595
+ keys_with_different_values = []
596
+ keys_in_left_not_right = []
597
+ keys_in_right_not_left = []
598
+ typename = 'dict'
599
+ for k, v in left.items():
600
+ if k in do_not_diff_keys:
601
+ # we only skip keys if the dict is an internal __dict__
602
+ # of a DataContainer or DataItem.
603
+ continue
604
+ elif k in right.keys():
605
+ if isinstance(do_not_diff_keys, dict) and k in do_not_diff_keys.keys():
606
+ shared_kv_same = self.diff(left[k], right[k], name=k, ancestors=ancestors, diff_container=diff_container, summarize=True, do_not_diff_keys=do_not_diff_keys[k])
607
+ else:
608
+ shared_kv_same = self.diff(left[k], right[k], name=k, ancestors=ancestors, diff_container=diff_container, summarize=True)
609
+ if not shared_kv_same: keys_with_different_values.append(k)
610
+ else:
611
+ keys_in_left_not_right.append(k)
612
+
613
+ for k, v in right.items():
614
+ if k in do_not_diff_keys:
615
+ continue
616
+ elif k in left.keys():
617
+ pass
618
+ else:
619
+ keys_in_right_not_left.append(k)
620
+
621
+ left_comments = []
622
+ right_comments = []
623
+ if keys_with_different_values:
624
+ same = False
625
+ left_comments.append('Keys with diffs: ' + ', '.join(keys_with_different_values))
626
+ right_comments.append('Keys with diffs: ' + ', '.join(keys_with_different_values))
627
+ if keys_in_left_not_right:
628
+ same = False
629
+ right_comments.append('Missing Keys: ' + ', '.join(keys_in_left_not_right))
630
+ if keys_in_right_not_left:
631
+ same = False
632
+ left_comments.append('Missing Keys: ' + ', '.join(keys_in_right_not_left))
633
+ if same:
634
+ left_comments.append('All k/v pairs same')
635
+ right_comments.append('All k/v pairs same')
636
+
637
+ left_str = '\n'.join(left_comments)
638
+ right_str = '\n'.join(right_comments)
639
+
640
+ # 7. Library Object Comparison (DataItem/DataContainer)
641
+ elif isinstance(left, (globals().get('DataContainer'), DataItem)):
642
+ from tts_data_utils.core.diff import DiffItem
643
+ same = self.diff(left.__dict__, right.__dict__, name=self.name, ancestors=ancestors, diff_container=diff_container, summarize=True, debug=debug, do_not_diff_keys=left.DO_NOT_DIFF)
644
+
645
+ if not same:
646
+ left_str = 'See children'
647
+ right_str = 'See children'
648
+ else:
649
+ left_str = 'All children same'
650
+ right_str = 'All children same'
651
+ typename = type(left).__name__
652
+ else:
653
+ same = False
654
+ left_str = 'Undiffable Type'
655
+ right_str = 'Undiffable Type'
656
+ typename = type(left).__name__
657
+
658
+ # Log results to the accumulator
659
+ diff_container.append({
660
+ 'Key': ancestors ,
661
+ 'Same': same,
662
+ 'Type': typename,
663
+ 'Left': left_str,
664
+ 'Right': right_str,
665
+ 'left': left,
666
+ 'right': right
667
+ })
668
+
669
+ # Apply ignoring logic for output pruning
670
+ for ignored_path in ignore:
671
+ diff_container = diff_container.ne('Key', '/')
672
+ diff_container = diff_container.ne('Key', f'/{self.name}')
673
+ diff_container = diff_container.ne('Key', f'/{self.name}/{ignored_path}')
674
+ diff_container = diff_container.doesnotmatch('Key', f'/{self.name}/{ignored_path}/.*')
675
+
676
+ if summarize:
677
+ return same
678
+ return diff_container
679
+
680
+ def compare_rows(self, l, r):
681
+ """
682
+ Calculates the similarity between two DataItems by counting matching values.
683
+
684
+ **Concept:**
685
+ This is used by the visual diff engine to determine if two rows are similar
686
+ enough to be considered a 'replacement' rather than an 'insertion' and
687
+ 'deletion'. It iterates through keys in the left item and checks for
688
+ equality in the right item.
689
+
690
+ :param l: The left DataItem.
691
+ :type l: DataItem
692
+ :param r: The right DataItem.
693
+ :type r: DataItem
694
+ :return: Integer count of identical fields.
695
+ :rtype: int
696
+ """
697
+ return sum(1 for key in l.values if l.values.get(key) == r.values.get(key))
698
+
699
+ def _get_index_from_hash(self, target_hash):
700
+ """
701
+ Returns the index of the first element whose hash() matches the target.
702
+
703
+ **Concept:**
704
+ Used to re-align records after visual diff processing. It performs a
705
+ linear search through `self.records` comparing the Python `hash()`
706
+ of each record to the target.
707
+
708
+ :param target_hash: The hash value to locate.
709
+ :type target_hash: int
710
+ :raises ValueError: If no record matching the hash is found.
711
+ :return: The index of the matching record.
712
+ :rtype: int
713
+ """
714
+ for i, rec in enumerate(self.records):
715
+ if hash(rec) == target_hash: # compare the hash
716
+ return i
717
+
718
+ # Trigger debugger to investigate why a record signature was lost
719
+ pdb.set_trace()
720
+ raise ValueError(f'Hash "{target_hash}" not found in records')
721
+
722
+ def visual_diff(self, right, ignore_cols=[], tolerance={}):
723
+ """
724
+ Generates a side-by-side visual alignment between this container and another.
725
+
726
+ **The Concept:**
727
+ This uses `SequenceMatcher` to find the best horizontal alignment between two
728
+ datasets. It identifies identical rows, modified rows (replace), and
729
+ inserted/deleted rows. It then injects "empty" placeholders into the
730
+ resulting containers so that matching records stay horizontally synchronized
731
+ when rendered.
732
+
733
+ :param right: The DataContainer to compare against.
734
+ :type right: DataContainer
735
+ :param ignore_cols: Columns to exclude from the row-matching signature.
736
+ :type ignore_cols: list[str]
737
+ :param tolerance: Drift allowance for numeric or datetime columns.
738
+ :type tolerance: dict[str, float]
739
+ :return: A tuple of two VisualDiffContainers (left, right).
740
+ """
741
+
742
+ #Avoids circular dependency. Probably a better way to do it
743
+ #but here we are...
744
+ from tts_data_utils.core.visual_diff import VisualDiffContainer
745
+
746
+ left = self._copy()
747
+ right = right._copy()
748
+
749
+ # Generate row signatures
750
+ aa = [{k: v for k, v in r.values.items() if k not in ignore_cols} for r in left.records]
751
+ bb = [{k: v for k, v in r.values.items() if k not in ignore_cols} for r in right.records]
752
+
753
+ for a, b in zip(aa, bb):
754
+ for k, v in tolerance.items():
755
+ match = False
756
+ if isinstance(a[k], datetime) and isinstance(b[k], datetime):
757
+ match = abs((a[k] - b[k]).total_seconds()) < v
758
+ elif isinstance(a[k], datetime) or isinstance(b[k], datetime):
759
+ raise Exception(f'Compared values for "{k}" must either both be datetimes or both not be datetimes. Got "{type(a[k]).__name__}" and "{type(b[k]).__name__}"')
760
+ else:
761
+ try:
762
+ match = abs(a[k] - b[k]) < v
763
+ except TypeError:
764
+ raise Exception(f'Compared values for "{k}" must both be numbers or both datetimes. Got "{type(a[k]).__name__} "and "{type(b[k]).__name__}"')
765
+ a[k] = match
766
+ b[k] = match
767
+
768
+ aa = [tuple(sorted((k, v) for k, v in a.items())) for a in aa]
769
+ bb = [tuple(sorted((k, v) for k, v in b.items())) for b in bb]
770
+ sm = SequenceMatcher(None, aa, bb)
771
+ # Iterate over opcodes to assign row status
772
+ for tag, i1, i2, j1, j2 in sm.get_opcodes():
773
+ if tag == 'equal':
774
+ # Rows are the same in both, mark as equal
775
+ for i, j in zip(range(i1, i2), range(j1, j2)):
776
+ left[i]['_visdiff_match'] = 'equal'
777
+ left[i]['_visdiff_index'] = j
778
+ right[j]['_visdiff_match'] = 'equal'
779
+ right[j]['_visdiff_index'] = i
780
+ left[i]['_mismatched_keys'] = []
781
+ right[j]['_mismatched_keys'] = []
782
+
783
+ elif tag == 'replace':
784
+ # Who hurt you?
785
+ if i2 - i1 != j2 - j1:
786
+ left_chunk = left[i1:i2]
787
+ right_chunk = right[j1:j2]
788
+
789
+ #to begin, assume all lefts are deleted and all rights
790
+ #are added. When we attempt to find best matches below
791
+ #we will overwrite the best we can
792
+ for i in range(i1, i2):
793
+ left[i]['_visdiff_match'] = 'delete'
794
+ left[i]['_visdiff_index'] = None
795
+ left[i]['_mismatched_keys'] = []
796
+ for j in range(j1, j2):
797
+ right[j]['_visdiff_match'] = 'insert'
798
+ right[j]['_visdiff_index'] = None
799
+ right[j]['_mismatched_keys'] = []
800
+
801
+ L = [left[ii] for ii in range(i1,i2)]
802
+ R = [right[jj] for jj in range(j1,j2)]
803
+ comparison_triples = [(l, r, self.compare_rows(l,r)) for l, r in product(L,R)]
804
+ #only consider it a diff if fewer than half the fields have changed. Otherwise it's an add and a delete
805
+ comparison_triples = [(l, r, score) for l, r, score in comparison_triples if score > len(r.values.keys())/2]
806
+ comparison_triples.sort(key=lambda x: x[2], reverse=True)
807
+
808
+ assigned_r = set()
809
+ assigned_l = set()
810
+ final_matches = {}
811
+
812
+ for l, r, score in comparison_triples:
813
+ if l not in assigned_l and r not in assigned_r:
814
+ final_matches[l] = r
815
+ assigned_l.add(l)
816
+ assigned_r.add(r)
817
+
818
+ for l, r in final_matches.items():
819
+ l['_visdiff_match'] = 'replace'
820
+ # this shouldn't be none, but the way I've done this the code doesn't
821
+ # have the index at this moment.
822
+ l['_visdiff_index'] = right._get_index_from_hash(hash(r))
823
+ l['_mismatched_keys'] = [k for k in l.values.keys() if l[k] != r[k]]
824
+ r['_visdiff_match'] = 'replace'
825
+ # this shouldn't be none, but the way I've done this the code doesn't
826
+ # have the index at this moment.
827
+ r['_visdiff_index'] = left._get_index_from_hash(hash(l))
828
+ r['_mismatched_keys'] = [k for k in l.values.keys() if l[k] != r[k]]
829
+
830
+ else:
831
+ for i, j in zip(range(i1, i2), range(j1, j2)):
832
+ #if the rows don't have at least half their cells in common, then
833
+ #treat them as add/delete instead of as replace
834
+ if self.compare_rows(left[i], right[j]) <= (i2 - i1)/2 and i2 - i1 >= 3:
835
+ left[i]['_visdiff_match'] = 'delete'
836
+ right[j]['_visdiff_match'] = 'insert'
837
+ mismatched_keys = []
838
+ left[i]['_visdiff_index'] = None
839
+ right[j]['_visdiff_index'] = None
840
+ left[i]['_mismatched_keys'] = []
841
+ right[j]['_mismatched_keys'] = []
842
+ else:
843
+ left[i]['_visdiff_match'] = 'replace'
844
+ right[j]['_visdiff_match'] = 'replace'
845
+ mismatched_keys = [k for k in left[i].values.keys() if left[i][k] != right[j][k]]
846
+
847
+ left[i]['_visdiff_index'] = j
848
+ right[j]['_visdiff_index'] = i
849
+ left[i]['_mismatched_keys'] = mismatched_keys
850
+ right[j]['_mismatched_keys'] = mismatched_keys
851
+
852
+ elif tag == 'delete':
853
+ # Rows exist in left only
854
+ for i in range(i1, i2):
855
+ left[i]['_visdiff_match'] = 'delete'
856
+ left[i]['_visdiff_index'] = None
857
+ left[i]['_mismatched_keys'] = []
858
+
859
+ elif tag == 'insert':
860
+ # Rows exist in right only
861
+ for j in range(j1, j2):
862
+ right[j]['_visdiff_match'] = 'insert'
863
+ right[j]['_visdiff_index'] = None
864
+ right[j]['_mismatched_keys'] = []
865
+
866
+
867
+ longer_table_len = max(len(left), len(right))
868
+ new_left = left._copy(new_records=[])
869
+ new_right = right._copy(new_records=[])
870
+
871
+ if len(left) > len(right):
872
+ longer_table = left
873
+ shorter_table = right
874
+ new_longer_table = new_left
875
+ new_shorter_table = new_right
876
+ else:
877
+ longer_table = right
878
+ shorter_table = left
879
+ new_longer_table = new_right
880
+ new_shorter_table = new_left
881
+
882
+
883
+ ii = 0
884
+ jj = 0
885
+
886
+ while max(ii, jj) < longer_table_len:
887
+ try:
888
+ l = left[ii]
889
+ r = right[jj]
890
+ except:
891
+ pdb.set_trace()
892
+ empty_left_record = self.DATA_ITEM_CLS(source=r.values)
893
+ empty_left_record['_mismatched_keys'] = []
894
+ empty_left_record['_visdiff_match'] = 'empty_from_insert'
895
+
896
+ empty_right_record = self.DATA_ITEM_CLS(source=l.values)
897
+ empty_right_record['_mismatched_keys'] = []
898
+ empty_right_record['_visdiff_match'] = 'empty_from_delete'
899
+
900
+ if l['_visdiff_index'] is None and r['_visdiff_index'] is None:
901
+ new_right.append(empty_right_record)
902
+ new_right.append(r)
903
+ new_left.append(l)
904
+ new_left.append(empty_left_record)
905
+ ii += 1
906
+ jj += 1
907
+ elif l['_visdiff_index'] is None:
908
+ new_left.append(l)
909
+ new_right.append(empty_right_record)
910
+ ii += 1
911
+ elif r['_visdiff_index'] is None:
912
+ new_left.append(empty_left_record)
913
+ new_right.append(r)
914
+ jj += 1
915
+ else:
916
+ new_left.append(l)
917
+ new_right.append(r)
918
+ ii += 1
919
+ jj += 1
920
+
921
+ left = new_left
922
+ right = new_right
923
+
924
+ empties = []
925
+
926
+ right = self._de_interlace_diffs(['insert', 'empty_from_delete'], right)
927
+ left = self._de_interlace_diffs(['delete', 'empty_from_insert'], left, False)
928
+ # pdb.set_trace()
929
+
930
+ try:
931
+ visdiff_left, visdiff_right = VisualDiffContainer(raw_data=[r.values for r in left.records]), VisualDiffContainer(raw_data=[r.values for r in right.records])
932
+ except:
933
+ pdb.set_trace()
934
+
935
+ return visdiff_left, visdiff_right
936
+
937
+ def _de_interlace_diffs(self, target_values, container, empties_first=True):
938
+ """
939
+ Internal helper: Reorganizes alignment blocks so that empty placeholders
940
+ and actual data rows are grouped logically for display.
941
+ """
942
+ matching_sections = []
943
+ non_matching_sections = []
944
+ start = None
945
+ in_matching = None
946
+
947
+ for i, val in enumerate(container):
948
+ if val['_visdiff_match'] in target_values:
949
+ if in_matching is False:
950
+ non_matching_sections.append((start, i - 1))
951
+ start = i
952
+ elif in_matching is None:
953
+ start = i
954
+ in_matching = True
955
+ else:
956
+ if in_matching is True:
957
+ matching_sections.append((start, i - 1))
958
+ start = i
959
+ elif in_matching is None:
960
+ start = i
961
+ in_matching = False
962
+
963
+ # Finalize the last section
964
+ if start is not None:
965
+ if in_matching:
966
+ matching_sections.append((start, len(container) - 1))
967
+ else:
968
+ non_matching_sections.append((start, len(container) - 1))
969
+
970
+ all_sections = [{'start': x, 'end': y, 'reorder': True} for x, y in matching_sections] + \
971
+ [{'start': x, 'end': y, 'reorder': False} for x, y in non_matching_sections]
972
+ all_sections.sort(key=lambda x:x['start'])
973
+
974
+ de_interlaced_container = None
975
+ for section in all_sections:
976
+ new_chunk = container[section['start']:section['end']+1]
977
+ if section['reorder']:
978
+ if empties_first:
979
+ new_chunk = new_chunk.contains('_visdiff_match', 'empty') + \
980
+ new_chunk.doesnotcontain('_visdiff_match', 'empty')
981
+ else:
982
+ new_chunk = new_chunk.doesnotcontain('_visdiff_match', 'empty') + \
983
+ new_chunk.contains('_visdiff_match', 'empty')
984
+ if de_interlaced_container is None:
985
+ de_interlaced_container = new_chunk
986
+ else:
987
+ de_interlaced_container += new_chunk
988
+
989
+ return de_interlaced_container
990
+
991
+ def insert(self, index, record):
992
+ """
993
+ Inserts a record at the specified index and returns a new container instance.
994
+ """
995
+ records_before = [r for r in self[:index]]
996
+ records_after = [r for r in self[index:]]
997
+ new_records = records_before + [record] + records_after
998
+ return self._copy(new_records=new_records)
999
+
1000
+ def _add_record(self, record):
1001
+ """
1002
+ Internal: Directly appends a record dictionary converted to a DATA_ITEM_CLS.
1003
+ """
1004
+ self.records.append(self.DATA_ITEM_CLS(record))
1005
+
1006
+ def _copy(self, new_records=None):
1007
+ """
1008
+ Creates a deep copy of the container, its history, and its records.
1009
+ """
1010
+ if new_records is None: new_records = self.records
1011
+ new_obj = copy(self)
1012
+ new_obj._unlock()
1013
+ new_obj.history = deepcopy(self.history)
1014
+ new_obj.records = [r._copy() for r in new_records]
1015
+ new_obj._lock()
1016
+ return new_obj
1017
+
1018
+ def _lock(self):
1019
+ """
1020
+ Placeholder for locking state, primarily to support TOWER InputClient compatibility.
1021
+ """
1022
+ # Note: In the OCO-2 implementation of TOWER, input clients inherit from both
1023
+ # TOWER InputClient and this class. Stubs here prevent failures when calling
1024
+ # these methods outside of the TOWER environment.
1025
+ # TO DO: Consider bringing TOWER-like lock/unlock functionality into this class.
1026
+ return
1027
+
1028
+ def _unlock(self):
1029
+ """
1030
+ Placeholder for unlocking state. See _lock for details.
1031
+ """
1032
+ return
1033
+
1034
+ def _filter(self, filter_lambda, minimum=None, maximum=None, exactly=None):
1035
+ """
1036
+ The core engine for all chainable filtering methods.
1037
+
1038
+ **Performance Optimization:**
1039
+ Rather than a full deep copy of the entire container (which is slow for large datasets),
1040
+ we perform a targeted copy that excludes the full record set, then apply the filter
1041
+ logic to generate the new decimated record list.
1042
+ """
1043
+ condition = filter_lambda[0]
1044
+ comparison_string = filter_lambda[1]
1045
+
1046
+ #Note that this uses private methods from InputClient, but I propose
1047
+ #we move all of this down into InputClient, so that won't be a big
1048
+ #deal anymore once we do.
1049
+
1050
+ #what the heck is going on here? Glad you asked.
1051
+ #I'm basically doing a deep copy of self. In fact, when
1052
+ #I originally did this, that's exactly what I did.
1053
+ #but it was slooooooooooow. So instead I loop through the
1054
+ #attributes of the copy one at a time and only copy them
1055
+ #if they're NOT the records attribute.
1056
+ #
1057
+ #That way we don't waste time copying a huge amount of data
1058
+ #that we're not going to use most of anyway. Instead of copying
1059
+ #I just do the filterign that we want to do right there.
1060
+
1061
+ #TO DO: EXPLAIN WTF YOU ARE DOING HERE
1062
+
1063
+ # Filter the records based on the provided condition
1064
+ filtered_records = [r for r in self.records if condition(r)]
1065
+
1066
+ # Validate result counts based on constraints
1067
+ if exactly is not None and minimum is not None:
1068
+ log.warning('"exactly" and "minimum" kwargs are both set. Only "exactly" will be honored.')
1069
+ if exactly is not None and maximum is not None:
1070
+ log.warning('"exactly" and "maximum" kwargs are both set. Only "exactly" will be honored.')
1071
+
1072
+ if exactly is not None and len(filtered_records) != exactly:
1073
+ raise Exception(f'Filtered length is not exactly {exactly} as specified')
1074
+ if minimum is not None and len(filtered_records) < minimum:
1075
+ raise Exception(f'Filtered length is not at least {minimum} as specified')
1076
+ if maximum is not None and len(filtered_records) > maximum:
1077
+ raise Exception(f'Filtered length is not less than or equal to {maximum} as specified')
1078
+
1079
+ # Create the new decimated container
1080
+ new_obj = self._copy(filtered_records)
1081
+
1082
+ if len(self.records):
1083
+ percent_remaining = f'{len(new_obj.records)/len(self.records)*100:.2f}'
1084
+ else:
1085
+ percent_remaining = 'NA'
1086
+
1087
+ # Log the filter action to the audit history
1088
+ new_obj.history._add_record({
1089
+ 'Action': 'Filtered',
1090
+ 'Description': comparison_string,
1091
+ 'Ending Count': len(new_obj.records),
1092
+ 'Starting Count': len(self.records),
1093
+ 'Percent Remaining': percent_remaining
1094
+ })
1095
+
1096
+ # Special case: return the DataItem itself if exactly=1 requested
1097
+ if exactly == 1:
1098
+ new_obj._unlock()
1099
+ new_obj.records = new_obj.records[0]
1100
+ new_obj._lock()
1101
+
1102
+ return new_obj
1103
+
1104
+ def with_cols(self, columns):
1105
+ """
1106
+ Returns a new version of this container with the display columns changed.
1107
+ Will add new (empty) columns if they do not currently exist in the records.
1108
+
1109
+ :param columns: List of column names to display/return.
1110
+ :type columns: list[str]
1111
+ :return: A new DataContainer instance with updated column settings.
1112
+ """
1113
+ self._unlock()
1114
+ new_obj = self._copy()
1115
+ new_obj._repr_cols = columns
1116
+ new_obj._csv_cols = columns
1117
+ self._lock()
1118
+ return new_obj
1119
+
1120
+ def summarize(self, key, expected_values=None, include_times=True):
1121
+ """
1122
+ Generates a summary table counting occurrences and time ranges for unique values
1123
+ in a specific column.
1124
+
1125
+ **The Concept:**
1126
+ This method transforms the current data into a frequency report. If `expected_values`
1127
+ are provided, it validates the data against them and ensures the output table
1128
+ follows the user's preferred ordering, while still appending any unexpected "rogue"
1129
+ values at the end of the list.
1130
+
1131
+ :param key: The column name to summarize.
1132
+ :type key: str
1133
+ :param expected_values: Optional list of values to check for and order by.
1134
+ :type expected_values: list, optional
1135
+ :param include_times: If True, adds "First Occurrence" and "Last Occurrence" columns.
1136
+ :type include_times: bool
1137
+ :return: A GenericContainer containing the summary records.
1138
+ """
1139
+ unique_values = self.unique(key)
1140
+
1141
+ if expected_values is not None:
1142
+ unexpected_values = [u for u in unique_values if u not in expected_values]
1143
+ if len(unexpected_values):
1144
+ log.warning(f'Unexpected values found in column "{key}": {unexpected_values}')
1145
+
1146
+ # This logic preserves the user's requested order for expected values
1147
+ # while ensuring any actual values found in the data are included.
1148
+ summary_values = []
1149
+ for value in expected_values + unique_values:
1150
+ if value not in summary_values:
1151
+ summary_values.append(value)
1152
+ else:
1153
+ summary_values = unique_values
1154
+
1155
+ summary_records = []
1156
+ for summary_value in summary_values:
1157
+ summary_record = {key: summary_value}
1158
+
1159
+ # Filter for current value
1160
+ filtered = self.eq(key, summary_value)
1161
+
1162
+ # Fix: Only sort if we have a valid time label,
1163
+ # otherwise just use the filtered results order
1164
+ if self.default_time_label:
1165
+ sorted_records = filtered.sort()
1166
+ else:
1167
+ sorted_records = filtered
1168
+
1169
+ summary_record['Occurances'] = str(len(sorted_records))
1170
+
1171
+ if include_times:
1172
+ if len(sorted_records):
1173
+ # TO DO: Turn this into a string instead of relying on time_str property
1174
+ summary_record['First Occurence'] = sorted_records[0].time_str
1175
+ summary_record['Last Occurence'] = sorted_records[-1].time_str
1176
+ else:
1177
+ summary_record['First Occurence'] = 'NA'
1178
+ summary_record['Last Occurence'] = 'NA'
1179
+
1180
+ summary_records.append(summary_record)
1181
+
1182
+ # Import here to avoid circular dependency since GenericContainer inherits from DataContainer
1183
+ from tts_data_utils.core.generic import GenericContainer
1184
+ return GenericContainer(raw_data=summary_records)
1185
+
1186
+ def unique(self, key, exclude=[], sort=True):
1187
+ """
1188
+ Returns a list of unique values found in a specific column.
1189
+
1190
+ :param key: Name of the column to inspect.
1191
+ :type key: str
1192
+ :param exclude: List of values to filter out of the final unique list.
1193
+ :type exclude: list
1194
+ :param sort: If True, the resulting list is sorted ascending.
1195
+ :type sort: bool
1196
+ :return: A list of unique values.
1197
+ """
1198
+ if not isinstance(exclude, list):
1199
+ exclude = [exclude]
1200
+
1201
+ unique = list(set([x[key] for x in self.records if x[key] not in exclude]))
1202
+
1203
+ if sort:
1204
+ unique.sort()
1205
+
1206
+ return unique
1207
+
1208
+ def gt(self, key, value, minimum=None, maximum=None, exactly=None):
1209
+ """
1210
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1211
+ field is greater than value in "value" parameter.
1212
+
1213
+ :param key: Name of column to filter on
1214
+ :type key: str
1215
+
1216
+ :param value: Value to compare against
1217
+ :type value: Varies depending on contents of "key" column
1218
+
1219
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1220
+ :type minimum: int
1221
+
1222
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1223
+ :type maximum: int
1224
+
1225
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1226
+ :type exactly: int
1227
+
1228
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1229
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1230
+ :rtype: DataContainer or DataItem
1231
+ """
1232
+ return self._filter(gt(key, value), minimum=minimum, maximum=maximum, exactly=exactly)
1233
+
1234
+ def lt(self, key, value, minimum=None, maximum=None, exactly=None):
1235
+ """
1236
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1237
+ field is less than value in "value" parameter.
1238
+
1239
+ :param key: Name of column to filter on
1240
+ :type key: str
1241
+
1242
+ :param value: Value to compare against
1243
+ :type value: Varies depending on contents of "key" column
1244
+
1245
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1246
+ :type minimum: int
1247
+
1248
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1249
+ :type maximum: int
1250
+
1251
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1252
+ :type exactly: int
1253
+
1254
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1255
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1256
+ :rtype: DataContainer or DataItem
1257
+ """
1258
+ return self._filter(lt(key, value), minimum=minimum, maximum=maximum, exactly=exactly)
1259
+
1260
+ def gte(self, key, value, minimum=None, maximum=None, exactly=None):
1261
+ """
1262
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1263
+ field is greater than or equal to value in "value" parameter.
1264
+
1265
+ :param key: Name of column to filter on
1266
+ :type key: str
1267
+
1268
+ :param value: Value to compare against
1269
+ :type value: Varies depending on contents of "key" column
1270
+
1271
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1272
+ :type minimum: int
1273
+
1274
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1275
+ :type maximum: int
1276
+
1277
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1278
+ :type exactly: int
1279
+
1280
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1281
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1282
+ :rtype: DataContainer or DataItem
1283
+ """
1284
+ return self._filter(gte(key, value), minimum=minimum, maximum=maximum, exactly=exactly)
1285
+
1286
+ def lte(self, key, value, minimum=None, maximum=None, exactly=None):
1287
+ """
1288
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1289
+ field is less than or equal to value in "value" parameter.
1290
+
1291
+ :param key: Name of column to filter on
1292
+ :type key: str
1293
+
1294
+ :param value: Value to compare against
1295
+ :type value: Varies depending on contents of "key" column
1296
+
1297
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1298
+ :type minimum: int
1299
+
1300
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1301
+ :type maximum: int
1302
+
1303
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1304
+ :type exactly: int
1305
+
1306
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1307
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1308
+ :rtype: DataContainer or DataItem
1309
+ """
1310
+ return self._filter(lte(key, value), minimum=minimum, maximum=maximum, exactly=exactly)
1311
+
1312
+ def eq(self, key, value, minimum=None, maximum=None, exactly=None, tolerance=0):
1313
+ """
1314
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1315
+ field matches value in "value" field.
1316
+
1317
+ :param key: Name of column to filter on
1318
+ :type key: str
1319
+
1320
+ :param value: Value to compare against
1321
+ :type value: Varies depending on contents of "key" column
1322
+
1323
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1324
+ :type minimum: int
1325
+
1326
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1327
+ :type maximum: int
1328
+
1329
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1330
+ :type exactly: int
1331
+
1332
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1333
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1334
+ :rtype: DataContainer or DataItem
1335
+ """
1336
+ return self._filter(eq(key, value, tolerance=tolerance), minimum=minimum, maximum=maximum, exactly=exactly)
1337
+
1338
+ def ne(self, key, value, minimum=None, maximum=None, exactly=None):
1339
+ """
1340
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1341
+ field does not match value in "value" field.
1342
+
1343
+ :param key: Name of column to filter on
1344
+ :type key: str
1345
+
1346
+ :param value: Value to compare against
1347
+ :type value: Varies depending on contents of "key" column
1348
+
1349
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1350
+ :type minimum: int
1351
+
1352
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1353
+ :type maximum: int
1354
+
1355
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1356
+ :type exactly: int
1357
+
1358
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1359
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1360
+ :rtype: DataContainer or DataItem
1361
+ """
1362
+
1363
+ return self._filter(ne(key, value), minimum=minimum, maximum=maximum, exactly=exactly)
1364
+
1365
+ def isin(self, key, values, minimum=None, maximum=None, exactly=None):
1366
+ """
1367
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1368
+ field matches value any of the values in the list "values".
1369
+
1370
+ :param key: Name of column to filter on
1371
+ :type key: str
1372
+
1373
+ :param value: Value to compare against
1374
+ :type values: list. Type of list contents depends on contents of "key" column
1375
+
1376
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1377
+ :type minimum: int
1378
+
1379
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1380
+ :type maximum: int
1381
+
1382
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1383
+ :type exactly: int
1384
+
1385
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1386
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1387
+ :rtype: DataContainer or DataItem
1388
+ """
1389
+
1390
+ return self._filter(isin(key, values), minimum=minimum, maximum=maximum, exactly=exactly)
1391
+
1392
+ def notin(self, key, values, minimum=None, maximum=None, exactly=None):
1393
+ """
1394
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1395
+ field does not match any value in the list "values".
1396
+
1397
+ :param key: Name of column to filter on
1398
+ :type key: str
1399
+
1400
+ :param value: Value to compare against
1401
+ :type values: list. Type of list contents depends on contents of "key" column
1402
+
1403
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1404
+ :type minimum: int
1405
+
1406
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1407
+ :type maximum: int
1408
+
1409
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1410
+ :type exactly: int
1411
+
1412
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1413
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1414
+ :rtype: DataContainer or DataItem
1415
+ """
1416
+ return self._filter(notin(key, values), minimum=minimum, maximum=maximum, exactly=exactly)
1417
+
1418
+ def contains(self, key, substring, case_sensitive=True, minimum=None, maximum=None, exactly=None):
1419
+ """
1420
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1421
+ field contains the value in the "substring" parameter as a substring.
1422
+
1423
+ :param key: Name of column to filter on
1424
+ :type key: str
1425
+
1426
+ :param substring: Value to compare against
1427
+ :type substring: str
1428
+
1429
+ :param value: Should the substring match be case sensitive?
1430
+ :type value: bool
1431
+
1432
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1433
+ :type minimum: int
1434
+
1435
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1436
+ :type maximum: int
1437
+
1438
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1439
+ :type exactly: int
1440
+
1441
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1442
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1443
+ :rtype: DataContainer or DataItem
1444
+ """
1445
+ return self._filter(contains(key, substring, case_sensitive=case_sensitive), minimum=minimum, maximum=maximum, exactly=exactly)
1446
+
1447
+ def doesnotcontain(self, key, substring, case_sensitive=True, minimum=None, maximum=None, exactly=None):
1448
+ """
1449
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1450
+ field does not contain the value in the "substring" parameter as a substring.
1451
+
1452
+ :param key: Name of column to filter on
1453
+ :type key: str
1454
+
1455
+ :param substring: Value to compare against
1456
+ :type substring: str
1457
+
1458
+ :param value: Should the substring match be case sensitive?
1459
+ :type value: bool
1460
+
1461
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1462
+ :type minimum: int
1463
+
1464
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1465
+ :type maximum: int
1466
+
1467
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1468
+ :type exactly: int
1469
+
1470
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1471
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1472
+ :rtype: DataContainer or DataItem
1473
+ """
1474
+ return self._filter(doesnotcontain(key, substring, case_sensitive=case_sensitive), minimum=minimum, maximum=maximum, exactly=exactly)
1475
+
1476
+ def before(self, time, time_label=None, inclusive=False, minimum=None, maximum=None, exactly=None):
1477
+ """
1478
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1479
+ occur before the value in the "time" parameter.
1480
+
1481
+ Unlike most other filter methods, time MUST be a datetime.
1482
+
1483
+ Note that "key" is not requried since DataItems have default time columns. If an object takes multiple
1484
+ time columns (or if using something like GenericContainer with no default time label), the time_label
1485
+ kwarg is provided.
1486
+
1487
+ :param value: Time to compare against
1488
+ :type value: datetime
1489
+
1490
+ :param time_label: Name of time column to use if not the default
1491
+ :type time_label: str
1492
+
1493
+ :param inclusive: If a row's time matches "time" exactly, should it be included?
1494
+ :type inclusive: bool
1495
+
1496
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1497
+ :type minimum: int
1498
+
1499
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1500
+ :type maximum: int
1501
+
1502
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1503
+ :type exactly: int
1504
+
1505
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1506
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1507
+ :rtype: DataContainer or DataItem
1508
+ """
1509
+
1510
+ #TO DO: Fix this.. shouldn't have to reference a record just to get this
1511
+ if len(self.records):
1512
+ if time_label is None: time_label = [x for x in self.records[0].TIME_FORMATS.keys()][0]
1513
+ else:
1514
+ if time_label is None: time_label = 'NA'
1515
+ return self._filter(before(time, time_label, inclusive=inclusive), minimum=minimum, maximum=maximum, exactly=exactly)
1516
+
1517
+ def after(self, time, time_label=None, inclusive=False, minimum=None, maximum=None, exactly=None):
1518
+ """
1519
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1520
+ occur after the value in the "time" parameter.
1521
+
1522
+ Unlike most other filter methods, time MUST be a datetime.
1523
+
1524
+ Note that "key" is not requried since DataItems have default time columns. If an object takes multiple
1525
+ time columns (or if using something like GenericContainer with no default time label), the time_label
1526
+ kwarg is provided.
1527
+
1528
+ :param value: Time to compare against
1529
+ :type value: datetime
1530
+
1531
+ :param time_label: Name of time column to use if not the default
1532
+ :type time_label: str
1533
+
1534
+ :param inclusive: If a row's time matches "time" exactly, should it be included?
1535
+ :type inclusive: bool
1536
+
1537
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1538
+ :type minimum: int
1539
+
1540
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1541
+ :type maximum: int
1542
+
1543
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1544
+ :type exactly: int
1545
+
1546
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1547
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1548
+ :rtype: DataContainer or DataItem
1549
+ """
1550
+
1551
+ #TO DO: Fix this.. shouldn't have to reference a record just to get this
1552
+ if len(self.records):
1553
+ if time_label is None: time_label = [x for x in self.records[0].TIME_FORMATS.keys()][0]
1554
+ else:
1555
+ if time_label is None: time_label = 'NA'
1556
+ return self._filter(after(time, time_label, inclusive=inclusive), minimum=minimum, maximum=maximum, exactly=exactly)
1557
+
1558
+ def between(self, key, lower, upper, inclusive="both", minimum=None, maximum=None, exactly=None):
1559
+ """
1560
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1561
+ occur between the values in the "lower" and "upper" parameters.
1562
+
1563
+ Unlike most other filter methods, time MUST be a datetime.
1564
+
1565
+ Note that "key" is required on this method unlike the before and after methods. This is just an error
1566
+ by the developer. It is slated to be fixed at the next major release since it will be a breaking change:
1567
+ issue #32 (TO DO: Migrate out of JPL-internal issues)
1568
+
1569
+ :param key: Name of time column to use
1570
+ :type key: str
1571
+
1572
+ :param value: Time to compare against
1573
+ :type value: datetime
1574
+
1575
+ :param inclusive: If a row's time matches "time" exactly, should it be included?
1576
+ :type inclusive: str (should be upper, lower, both, or neither)
1577
+
1578
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1579
+ :type minimum: int
1580
+
1581
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1582
+ :type maximum: int
1583
+
1584
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1585
+ :type exactly: int
1586
+
1587
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1588
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1589
+ :rtype: DataContainer or DataItem
1590
+ """
1591
+ return self._filter(between(key, lower, upper, inclusive=inclusive), minimum=minimum, maximum=maximum, exactly=exactly)
1592
+
1593
+ def matches(self, key, pattern, minimum=None, maximum=None, exactly=None):
1594
+ """
1595
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1596
+ matches the regex in the parameter "pattern".
1597
+
1598
+ :param key: Name of column to match against
1599
+ :type key: str
1600
+
1601
+ :param pattern: Regex pattern
1602
+ :type pattern: r-string
1603
+
1604
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1605
+ :type minimum: int
1606
+
1607
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1608
+ :type maximum: int
1609
+
1610
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1611
+ :type exactly: int
1612
+
1613
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1614
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1615
+ :rtype: DataContainer or DataItem
1616
+ """
1617
+ return self._filter(matches(key, pattern), minimum=minimum, maximum=maximum, exactly=exactly)
1618
+
1619
+ def doesnotmatch(self, key, pattern, minimum=None, maximum=None, exactly=None):
1620
+ """
1621
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1622
+ does not match the regex in the parameter "pattern".
1623
+
1624
+ :param key: Name of column to not match against
1625
+ :type key: str
1626
+
1627
+ :param time_label: Name of time column to use if not the default
1628
+ :type time_label: str
1629
+
1630
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1631
+ :type minimum: int
1632
+
1633
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1634
+ :type maximum: int
1635
+
1636
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1637
+ :type exactly: int
1638
+
1639
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1640
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1641
+ :rtype: DataContainer or DataItem
1642
+ """
1643
+ return self._filter(doesnotmatch(key, pattern), minimum=minimum, maximum=maximum, exactly=exactly)
1644
+
1645
+
1646
+ def on_change(self, key, minimum=None, maximum=None, exactly=None):
1647
+ """
1648
+ Return a decimated verison of this DataContainer where all rows where column in "key"
1649
+ is different than in the row before. Will always include first row.
1650
+
1651
+ :param key: Name of column to inspect for changes
1652
+ :type key: str
1653
+
1654
+ :param time_label: Name of time column to use if not the default
1655
+ :type time_label: str
1656
+
1657
+ :param minimum: Minimum number of records to return. Will raise an exception if too few records match
1658
+ :type minimum: int
1659
+
1660
+ :param maximum: Maximum number of records to return. Will raise an exception if too many records match
1661
+ :type maximum: int
1662
+
1663
+ :param exactly: Exact number of records to return. Will raise an exception any other number of records match
1664
+ :type exactly: int
1665
+
1666
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1667
+ filtered outputs (except if exactly=1, in which case it will return a DataItem only).
1668
+ :rtype: DataContainer or DataItem
1669
+ """
1670
+ return self._filter(on_change(key), minimum=minimum, maximum=maximum, exactly=exactly)
1671
+
1672
+ def sort(self, by=None, lam=None, reverse=False):
1673
+ """
1674
+ Return a version of the DataContainer with rows sorted by the row in the "by" kwarg or by
1675
+ the lambda funciton in the "lam" kwarg.
1676
+
1677
+ Always sorts by ascending (for now, see https://github.jpl.nasa.gov/teamtools-studio/data_utils/issues/33)
1678
+
1679
+ :param by: Name of column to sort by
1680
+ :type by: str
1681
+
1682
+ :param lam: Lambda to control how values are sorted
1683
+ :type lam: lambda
1684
+
1685
+ :param reverse: By default, sorts like Python list sort. This works the same as reverse kwarg on default list sort
1686
+ :type reverse: bool
1687
+
1688
+ :return: Returns a new DataContainer exactly the same as this one, but with updated history and
1689
+ sorted outputs.
1690
+ :rtype: DataContainer
1691
+ """
1692
+ if by is not None and lam is not None:
1693
+ log.warning('"by" and "lamb" are both defined. Honoring "lamb" and not "by"')
1694
+ by = None
1695
+
1696
+ records_copy = copy(self.records)
1697
+ if by is None and lam is None:
1698
+ by = self.default_time_label
1699
+ records_copy.sort(key=lambda r: r.source[by], reverse=reverse)
1700
+ elif lam is not None:
1701
+ records_copy.sort(key=lam, reverse=reverse)
1702
+ else:
1703
+ records_copy.sort(key=lambda r: r.values[by], reverse=reverse)
1704
+ return self._copy(records_copy)
1705
+
1706
+ def append(self, items, cast_fields=False, fill=False):
1707
+ """
1708
+ Adds one or more items to the end of the container's records.
1709
+
1710
+ :param items: A dictionary, DataItem, or a list of either to append.
1711
+ :param cast_fields: If True, attempts to force data into types defined in DataItem.
1712
+ :param fill: If True, fills in missing keys with default values.
1713
+ """
1714
+ self._unlock()
1715
+ if not isinstance(items, list):
1716
+ items = [items]
1717
+
1718
+ for item in items:
1719
+ if isinstance(item, dict):
1720
+ self.records.append(self.DATA_ITEM_CLS(item, cast_fields=cast_fields, fill=fill))
1721
+ elif isinstance(item, self.DATA_ITEM_CLS):
1722
+ self.records.append(item)
1723
+ else:
1724
+ raise Exception('Unexpected item type!')
1725
+ self._lock()
1726
+
1727
+ def inject_error(self, lamb):
1728
+ """
1729
+ Iterates through records and applies a transformation lambda.
1730
+ Useful for error injection or data simulation.
1731
+
1732
+ :param lamb: A function that accepts a record and returns (bool, key, value).
1733
+ """
1734
+ # not super crazy about this implementation but I like the idea and
1735
+ # should continue to iterate on it
1736
+ self._unlock()
1737
+ for r in self.records:
1738
+ change_this_record, key, value = lamb(r)
1739
+ if change_this_record:
1740
+ r.source[key] = value
1741
+ self._lock()
1742
+
1743
+ def simple_record_table(self, *args, **kwargs):
1744
+ """Placeholder for breaking tests. One day at a time here..."""
1745
+ return
1746
+
1747
+ def file_contents_as_string(self, *args, **kwargs):
1748
+ """Placeholder for breaking tests. One day at a time here..."""
1749
+ return
1750
+
1751
+ def to_csv(self, csv_path, mkdirs=False):
1752
+ """
1753
+ Writes the container's records to a CSV file.
1754
+
1755
+ :param csv_path: Target file path.
1756
+ :param mkdirs: If True, creates the target directory if it does not exist.
1757
+ """
1758
+ if len(self.records):
1759
+ formatted_records = []
1760
+
1761
+ # Internal helper to handle JSON serialization of complex types
1762
+ def json_serialize_fallback(obj):
1763
+ if isinstance(obj, datetime):
1764
+ return obj.isoformat()
1765
+ return str(obj) # Fallback for HistoryItems and other objects
1766
+
1767
+ for r in self.records:
1768
+ row = r.printable_values.copy()
1769
+
1770
+ for k, v in row.items():
1771
+ if isinstance(v, (dict, list)):
1772
+ # Use the default parameter to handle datetimes/custom objects
1773
+ row[k] = json.dumps(v, default=json_serialize_fallback)
1774
+ formatted_records.append(row)
1775
+
1776
+ df = pd.DataFrame(formatted_records)
1777
+
1778
+ for col in self._csv_cols:
1779
+ if col not in df.columns:
1780
+ df[col] = [''] * len(df)
1781
+ df = df[self._csv_cols]
1782
+ else:
1783
+ df = pd.DataFrame(columns=self._csv_cols)
1784
+
1785
+ if mkdirs: os.makedirs(os.path.dirname(csv_path), exist_ok=True)
1786
+
1787
+ df.to_csv(csv_path, index=False, date_format=None)
1788
+ def read_csv(self, csv_path):
1789
+ """Reads a CSV file into a list of record dictionaries."""
1790
+ return pd.read_csv(csv_path).to_dict('records')
1791
+
1792
+ def read_xlsx(self, xlsx_path):
1793
+ """
1794
+ Reads an Excel file into a list of record dictionaries,
1795
+ handling NaN values as None.
1796
+ """
1797
+ raw_data = pd.read_excel(xlsx_path).to_dict('records')
1798
+ for row in raw_data:
1799
+ for k, v in row.items():
1800
+ if not isinstance(v, (int, float)): continue
1801
+ if isnan(v): row[k] = None
1802
+ return raw_data
1803
+
1804
+ def calculate_records_hash(self):
1805
+ """
1806
+ Generates a SHA256 hash representing the current state of all records.
1807
+
1808
+ **The Process:**
1809
+ Normalizes timestamps based on DataItem time formats to ensure consistent
1810
+ string representation before hashing the JSON-encoded record set.
1811
+ """
1812
+ records = [deepcopy(r.source) for r in self.records]
1813
+ for time_label, time_format in self.records[0].TIME_FORMATS.items():
1814
+ for record in records:
1815
+ record[time_label] = record[time_label].strftime(time_format)
1816
+
1817
+ dict_str = json.dumps(records)
1818
+ dict_hash = hashlib.sha256(dict_str.encode()).hexdigest()
1819
+ return dict_hash
1820
+
1821
+ ##################################################################################
1822
+ # Dexter-specific methods, consider reorganizing this so not every DataContainer gets this
1823
+ ##################################################################################
1824
+
1825
+ def assert_records_match_hash(self, expected_hash):
1826
+ """
1827
+ Validates the integrity of the records against a known hash.
1828
+ """
1829
+ actual_hash = self.calculate_records_hash()
1830
+ log.info(f'Expected hash: {expected_hash}')
1831
+ log.info(f'Actual hash: {actual_hash}')
1832
+ if expected_hash != actual_hash:
1833
+ raise Exception(f'Expected and Actual hashes do not match!\n'
1834
+ f'Expected hash: {expected_hash}\n'
1835
+ f'Actual hash: {actual_hash}')
1836
+
1837
+ def stamp_all(self, dispo_choice, dispo_format):
1838
+ """Iterates through all data and applies a disposition stamp."""
1839
+ for _data in self:
1840
+ _data.choose_and_stamp(dispo_choice, dispo_format)
1841
+
1842
+ def subdivide_f(self, sub_f):
1843
+ """Returns a subdivided container based on a filter function."""
1844
+ sub_data = [_ for _ in self if sub_f(_)]
1845
+ return self.subdivide(sub_data, bypass_validation=self._bypass_validation)
1846
+
1847
+ @classmethod
1848
+ def subdivide(cls, sub_data, bypass_validation=False):
1849
+ """
1850
+ Class method to create a new 'sub-container' instance.
1851
+ """
1852
+ return cls(
1853
+ sub_data,
1854
+ bypass_validation=bypass_validation,
1855
+ sub_container=True
1856
+ )
1857
+
1858
+ def gt(key, value):
1859
+ """
1860
+ Returns a predicate for: field > value.
1861
+
1862
+ :param key: column to filter on
1863
+ :type key: str
1864
+ :param value: value to check against
1865
+ :type value: int or float
1866
+ """
1867
+ return lambda r: r[key] > value, f'"{key}" > {value}'
1868
+
1869
+ def lt(key, value):
1870
+ """
1871
+ Returns a predicate for: field < value.
1872
+
1873
+ :param key: column to filter on
1874
+ :type key: str
1875
+ :param value: value to check against
1876
+ :type value: int or float
1877
+ """
1878
+ return lambda r: r[key] < value, f'"{key}" < {value}'
1879
+
1880
+ def gte(key, value):
1881
+ """
1882
+ Returns a predicate for: field >= value.
1883
+
1884
+ :param key: column to filter on
1885
+ :type key: str
1886
+ :param value: value to check against
1887
+ :type value: int or float
1888
+ """
1889
+ return lambda r: r[key] >= value, f'"{key}" >= {value}'
1890
+
1891
+ def lte(key, value):
1892
+ """
1893
+ Returns a predicate for: field <= value.
1894
+
1895
+ :param key: column to filter on
1896
+ :type key: str
1897
+ :param value: value to check against
1898
+ :type value: int or float
1899
+ """
1900
+ return lambda r: r[key] <= value, f'"{key}" <= {value}'
1901
+
1902
+ def eq(key, value, tolerance=0):
1903
+ """
1904
+ Returns a predicate for: field == value.
1905
+
1906
+ :param key: column to filter on
1907
+ :type key: str, int, float, datetime
1908
+ :param value: value to check against
1909
+ :type value: any
1910
+ """
1911
+ if isinstance(value, str):
1912
+ return lambda r: r[key] == value, f'"{key}" == {value}'
1913
+ elif isinstance(value, datetime):
1914
+ if not isinstance(tolerance, timedelta):
1915
+ tolerance = timedelta(seconds=tolerance)
1916
+ return lambda r: abs(r[key] - value) <= tolerance, f'"{key}" == {value}'
1917
+ elif isinstance(value, (int, float)):
1918
+ return lambda r: abs(r[key] - value) <= tolerance, f'"{key}" == {value}'
1919
+
1920
+
1921
+ def ne(key, value):
1922
+ """
1923
+ Returns a predicate for: field != value.
1924
+
1925
+ :param key: column to filter on
1926
+ :type key: str
1927
+ :param value: value to check against
1928
+ :type value: any
1929
+ """
1930
+ return lambda r: r[key] != value, f'"{key}" != {value}'
1931
+
1932
+ def isin(key, values):
1933
+ """
1934
+ Returns a predicate for: field in list_of_values.
1935
+
1936
+ :param key: column to filter on
1937
+ :type key: str
1938
+ :param value: value to check against
1939
+ :type value: list
1940
+ """
1941
+ return lambda r: r[key] in values, f'"{key}" is in {values}'
1942
+
1943
+ def notin(key, values):
1944
+ """
1945
+ Returns a predicate for: field not in list_of_values.
1946
+
1947
+ :param key: column to filter on
1948
+ :type key: str
1949
+ :param value: value to check against
1950
+ :type value: list
1951
+ """
1952
+ return lambda r: r[key] not in values, f'"{key}" is not in {values}'
1953
+
1954
+ def contains(key, substring, case_sensitive=True):
1955
+ """
1956
+ Returns a predicate for substring matching.
1957
+
1958
+ :param key: column to filter on
1959
+ :type key: str
1960
+ :param substring: Substring to check values for
1961
+ :type substring: str
1962
+ :param case_sensitive: whether to check with case sensitiveiy or not. Defaults to True
1963
+ :type case_sensitive: bool
1964
+ """
1965
+ if case_sensitive:
1966
+ return lambda r: substring in r[key], f'"{key}" contains {substring} (case sensitive)'
1967
+ else:
1968
+ return lambda r: substring.lower() in r[key].lower(), f'"{key}" contains {substring} (case insensitive)'
1969
+
1970
+ def doesnotcontain(key, substring, case_sensitive=True):
1971
+ """
1972
+ Returns a predicate for negative substring matching.
1973
+
1974
+ :param key: column to filter on
1975
+ :type key: str
1976
+ :param substring: Substring to check values for
1977
+ :type substring: str
1978
+ :param case_sensitive: whether to check with case sensitiveiy or not. Defaults to True
1979
+ :type case_sensitive: bool
1980
+ """
1981
+ if case_sensitive:
1982
+ return lambda r: substring not in r[key], f'"{key}" does not contain {substring} (case sensitive)'
1983
+ else:
1984
+ return lambda r: substring.lower() not in r[key].lower(), f'"{key}" does not contain {substring} (case insensitive)'
1985
+
1986
+ def before(time, time_label, inclusive=False):
1987
+ """
1988
+ Returns a predicate for datetime comparison (earlier than).
1989
+
1990
+ :param time: Time for comparison
1991
+ :type time: datetime
1992
+ :param time_label: Label for time column
1993
+ :type time_label: str
1994
+ :param inclusive: Should we include a time that is exactly equal? Defaults to False
1995
+ :type inclusive: bool
1996
+ """
1997
+ if inclusive:
1998
+ return lambda r: r[time_label] <= time, f'"{time_label}" <= {time}'
1999
+ else:
2000
+ return lambda r: r[time_label] < time, f'"{time_label}" < {time}'
2001
+
2002
+ def after(time, time_label, inclusive=False):
2003
+ """
2004
+ Returns a predicate for datetime comparison (later than).
2005
+
2006
+ :param time: Time for comparison
2007
+ :type time: datetime
2008
+ :param time_label: Label for time column
2009
+ :type time_label: str
2010
+ :param inclusive: Should we include a time that is exactly equal? Defaults to False
2011
+ :type inclusive: bool
2012
+ """
2013
+ if inclusive:
2014
+ return lambda r: r[time_label] >= time, f'"{time_label}" >= {time}'
2015
+ else:
2016
+ return lambda r: r[time_label] > time, f'"{time_label}" > {time}'
2017
+
2018
+ def between(key, lower, upper, inclusive="both"):
2019
+ """
2020
+ Returns a predicate for range comparison.
2021
+
2022
+ :param key: column to filter on
2023
+ :type key: str
2024
+ :param lower: Lower value for range comparison
2025
+ :param upper: Upper value for range comparison
2026
+ :param inclusive: One of 'both', 'neither', 'lower', or 'upper'.
2027
+ """
2028
+ if inclusive == "both":
2029
+ return lambda r: lower <= r[key] <= upper, f'{lower} <= "{key}" <= {upper}'
2030
+ elif inclusive == "neither":
2031
+ return lambda r: lower < r[key] < upper, f'{lower} < "{key}" < {upper}'
2032
+ elif inclusive == "lower":
2033
+ return lambda r: lower <= r[key] < upper, f'{lower} <= "{key}" < {upper}'
2034
+ elif inclusive == "upper":
2035
+ return lambda r: lower < r[key] <= upper, f'{lower} < "{key}" <= {upper}'
2036
+ else:
2037
+ raise ValueError("inclusive must be 'both', 'neither', 'lower', or 'upper'")
2038
+
2039
+ def matches(key, pattern):
2040
+ """
2041
+ Returns a predicate for regex matching.
2042
+
2043
+ :param key: column to filter on
2044
+ :type key: str
2045
+ :param pattern: regex pattern to match with
2046
+ :type pattern: str
2047
+ """
2048
+ return lambda r: re.match(pattern, r[key]), 'matches'
2049
+
2050
+ def doesnotmatch(key, pattern):
2051
+ """
2052
+ Returns a predicate for negative regex matching.
2053
+
2054
+ :param key: column to filter on
2055
+ :type key: str
2056
+ :param pattern: regex pattern to match with
2057
+ :type pattern: str
2058
+ """
2059
+ return lambda r: not(re.match(pattern, r[key])), 'does not match'
2060
+
2061
+ def on_change(key):
2062
+ """
2063
+ Returns a stateful predicate that triggers when the value in a column
2064
+ changes relative to the previous record.
2065
+
2066
+ :param key: Column to check for changes in
2067
+ :type key: str
2068
+ """
2069
+ last_value = [None] # mutable closure to track previous value
2070
+ is_first = [True] # flag to catch the first record
2071
+
2072
+ def condition(record):
2073
+ nonlocal last_value, is_first
2074
+ if is_first[0]:
2075
+ is_first[0] = False
2076
+ last_value[0] = record[key]
2077
+ return True
2078
+ elif record[key] != last_value[0]:
2079
+ last_value[0] = record[key]
2080
+ return True
2081
+ else:
2082
+ return False
2083
+
2084
+ return (condition, f"on_change({key})")
2085
+
2086
+
2087
+ FILTERS = {
2088
+ 'gt': gt,
2089
+ 'lt': lt,
2090
+ 'gte': gte,
2091
+ 'lte': lte,
2092
+ 'eq': eq,
2093
+ 'ne': ne,
2094
+ 'isin': isin,
2095
+ 'notin': notin,
2096
+ 'contains': contains,
2097
+ 'doesnotcontain': doesnotcontain,
2098
+ 'before': before,
2099
+ 'after': after,
2100
+ 'between': between,
2101
+ 'matches': matches
2102
+ }