gea-program 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. gea/BENCH_TEST_PROTOCOL.md +97 -0
  2. gea/IMPORT_RECORD.md +61 -0
  3. gea/__init__.py +160 -0
  4. gea/__main__.py +661 -0
  5. gea/acceptance_tests.py +1315 -0
  6. gea/accuracy_statement.py +181 -0
  7. gea/alarm_engine.py +307 -0
  8. gea/bench.py +144 -0
  9. gea/blind_harness.py +108 -0
  10. gea/case_study.py +229 -0
  11. gea/catalog/acex_lomonosov_age_depth_model.provenance.json +23 -0
  12. gea/catalog/acex_lomonosov_age_depth_model.txt +28 -0
  13. gea/catalog/agassiz77_canada_temperature.csv +68 -0
  14. gea/catalog/agassiz77_canada_temperature.provenance.json +17 -0
  15. gea/catalog/barbados_110_consolidation.provenance.json +24 -0
  16. gea/catalog/barbados_110_consolidation.txt +91 -0
  17. gea/catalog/bengal_u1452_grain_size.provenance.json +20 -0
  18. gea/catalog/bengal_u1452_grain_size.txt +252 -0
  19. gea/catalog/blake_164_methane_isotopes.provenance.json +22 -0
  20. gea/catalog/blake_164_methane_isotopes.txt +68 -0
  21. gea/catalog/chicxulub_m0077a_pwave_velocity.provenance.json +23 -0
  22. gea/catalog/chicxulub_m0077a_pwave_velocity.txt +735 -0
  23. gea/catalog/collingwood_1_28_ks_complete.las +128 -0
  24. gea/catalog/collingwood_1_28_ks_complete.provenance.json +17 -0
  25. gea/catalog/costa_rica_odp_friction_envelope.provenance.json +24 -0
  26. gea/catalog/costa_rica_odp_friction_envelope.txt +57 -0
  27. gea/catalog/dead_sea_5017_debrite_xrf_ms.provenance.json +21 -0
  28. gea/catalog/dead_sea_5017_debrite_xrf_ms.txt +94 -0
  29. gea/catalog/dsdp_504b_physical_properties.provenance.json +24 -0
  30. gea/catalog/dsdp_504b_physical_properties.txt +82 -0
  31. gea/catalog/dsdp_504b_sound_velocity.provenance.json +21 -0
  32. gea/catalog/dsdp_504b_sound_velocity.txt +81 -0
  33. gea/catalog/elgygytgyn_5011_turbidites.provenance.json +20 -0
  34. gea/catalog/elgygytgyn_5011_turbidites.txt +193 -0
  35. gea/catalog/epica_domec_co2_800kyr.provenance.json +21 -0
  36. gea/catalog/epica_domec_co2_800kyr.txt +265 -0
  37. gea/catalog/fram_909_organic_petrography.provenance.json +22 -0
  38. gea/catalog/fram_909_organic_petrography.txt +40 -0
  39. gea/catalog/gbr_325_coral_uth_ages.provenance.json +23 -0
  40. gea/catalog/gbr_325_coral_uth_ages.txt +71 -0
  41. gea/catalog/gisp2_greenland_temperature.csv +599 -0
  42. gea/catalog/gisp2_greenland_temperature.provenance.json +17 -0
  43. gea/catalog/gom_308_t2p_insitu.provenance.json +21 -0
  44. gea/catalog/gom_308_t2p_insitu.txt +40 -0
  45. gea/catalog/guaymas_385_dom_d13c.provenance.json +21 -0
  46. gea/catalog/guaymas_385_dom_d13c.txt +103 -0
  47. gea/catalog/hikurangi_u1520_friction_insitu.provenance.json +26 -0
  48. gea/catalog/hikurangi_u1520_friction_insitu.txt +74 -0
  49. gea/catalog/hydrate_ridge_204_ncr_hydrate.provenance.json +22 -0
  50. gea/catalog/hydrate_ridge_204_ncr_hydrate.txt +60 -0
  51. gea/catalog/iodp_u1324_pore_pressure.provenance.json +22 -0
  52. gea/catalog/iodp_u1324_pore_pressure.txt +42 -0
  53. gea/catalog/jfast_c0019_slow_slip_events.provenance.json +28 -0
  54. gea/catalog/jfast_c0019_slow_slip_events.txt +35 -0
  55. gea/catalog/kennetcook_2_p129_excerpt.las +139 -0
  56. gea/catalog/kennetcook_2_p129_excerpt.provenance.json +17 -0
  57. gea/catalog/ktb_hb_bhgm_density.dat +227 -0
  58. gea/catalog/ktb_hb_bhgm_density.provenance.json +22 -0
  59. gea/catalog/ktb_hb_complog_6020_excerpt.provenance.json +20 -0
  60. gea/catalog/ktb_hb_complog_6020_excerpt.txt +72 -0
  61. gea/catalog/ktb_hb_hlog246_temperature.dat +1636 -0
  62. gea/catalog/ktb_hb_hlog246_temperature.provenance.json +22 -0
  63. gea/catalog/ktb_hb_rockmech_compress.dat +33 -0
  64. gea/catalog/ktb_hb_rockmech_compress.provenance.json +22 -0
  65. gea/catalog/ktb_hb_tvd_0_9080_excerpt.dat +2817 -0
  66. gea/catalog/ktb_hb_tvd_0_9080_excerpt.provenance.json +20 -0
  67. gea/catalog/ktb_vb_rockmech_compress.dat +125 -0
  68. gea/catalog/ktb_vb_rockmech_compress.provenance.json +22 -0
  69. gea/catalog/ktb_vb_vlog251_temperature.dat +1127 -0
  70. gea/catalog/ktb_vb_vlog251_temperature.provenance.json +24 -0
  71. gea/catalog/l06_06_nl_survey.csv +201 -0
  72. gea/catalog/l06_06_nl_survey.provenance.json +17 -0
  73. gea/catalog/l07_01_nl_excerpt.las +90 -0
  74. gea/catalog/l07_01_nl_excerpt.provenance.json +17 -0
  75. gea/catalog/mariana_1200_serpentinite_geochem.provenance.json +21 -0
  76. gea/catalog/mariana_1200_serpentinite_geochem.txt +63 -0
  77. gea/catalog/med_160_sapropels.provenance.json +22 -0
  78. gea/catalog/med_160_sapropels.txt +43 -0
  79. gea/catalog/nankai_megasplay_shear_strength.provenance.json +26 -0
  80. gea/catalog/nankai_megasplay_shear_strength.txt +42 -0
  81. gea/catalog/odp_1027b_thermal_conductivity.provenance.json +21 -0
  82. gea/catalog/odp_1027b_thermal_conductivity.txt +52 -0
  83. gea/catalog/odp_1027c_cork_temperature.provenance.json +20 -0
  84. gea/catalog/odp_1027c_cork_temperature.txt +26 -0
  85. gea/catalog/odp_1165b_thermal_conductivity.provenance.json +22 -0
  86. gea/catalog/odp_1165b_thermal_conductivity.txt +102 -0
  87. gea/catalog/odp_1274a_mantle_peridotite_mad.provenance.json +27 -0
  88. gea/catalog/odp_1274a_mantle_peridotite_mad.txt +44 -0
  89. gea/catalog/odp_504b_dike_elastic_moduli.provenance.json +27 -0
  90. gea/catalog/odp_504b_dike_elastic_moduli.txt +85 -0
  91. gea/catalog/odp_504b_leg137_borehole_fluids.provenance.json +19 -0
  92. gea/catalog/odp_504b_leg137_borehole_fluids.txt +68 -0
  93. gea/catalog/odp_735b_gabbro_elastic_moduli.provenance.json +28 -0
  94. gea/catalog/odp_735b_gabbro_elastic_moduli.txt +127 -0
  95. gea/catalog/peru_201_sulfate_reduction.provenance.json +22 -0
  96. gea/catalog/peru_201_sulfate_reduction.txt +322 -0
  97. gea/catalog/scorpio_e1_sa_excerpt.las +113 -0
  98. gea/catalog/scorpio_e1_sa_excerpt.provenance.json +17 -0
  99. gea/catalog/sumatra_362_cohesion.provenance.json +21 -0
  100. gea/catalog/sumatra_362_cohesion.txt +38 -0
  101. gea/catalog/university_6_17_no1_tx_excerpt.las +119 -0
  102. gea/catalog/university_6_17_no1_tx_excerpt.provenance.json +17 -0
  103. gea/catalog/ursa_308_xrd_mineralogy.provenance.json +21 -0
  104. gea/catalog/ursa_308_xrd_mineralogy.txt +46 -0
  105. gea/catalog/volve_15_9_19_sr_excerpt.las +183 -0
  106. gea/catalog/volve_15_9_19_sr_excerpt.provenance.json +16 -0
  107. gea/catalog/volve_15_9_19a_core_excerpt.csv +88 -0
  108. gea/catalog/volve_15_9_19a_core_excerpt.provenance.json +18 -0
  109. gea/catalog/volve_f12_f14_production_excerpt.csv +167 -0
  110. gea/catalog/volve_f12_f14_production_excerpt.provenance.json +21 -0
  111. gea/catalog/walvis_208_petm_carbonate.provenance.json +21 -0
  112. gea/catalog/walvis_208_petm_carbonate.txt +268 -0
  113. gea/catalog/woodlark_1109_rock_eval.provenance.json +23 -0
  114. gea/catalog/woodlark_1109_rock_eval.txt +30 -0
  115. gea/cli.py +125 -0
  116. gea/client_reports.py +943 -0
  117. gea/config_versioning.py +133 -0
  118. gea/correlation.py +155 -0
  119. gea/dashboard.py +390 -0
  120. gea/deviation.py +70 -0
  121. gea/downhole_engine.py +395 -0
  122. gea/drift_monitor.py +310 -0
  123. gea/earth_model.py +230 -0
  124. gea/example_register_map.json +14 -0
  125. gea/fat_sat.py +68 -0
  126. gea/follower.py +98 -0
  127. gea/forward_model.py +139 -0
  128. gea/gamma.py +176 -0
  129. gea/gauge_specs.py +112 -0
  130. gea/gravity_reference.py +116 -0
  131. gea/inverse_engine.py +215 -0
  132. gea/matplotlib_demo.py +85 -0
  133. gea/modbus.py +229 -0
  134. gea/model_card.py +248 -0
  135. gea/operator_app.py +442 -0
  136. gea/ports.py +340 -0
  137. gea/profile_catalog.py +773 -0
  138. gea/project.py +213 -0
  139. gea/qt6_downhole_app.py +144 -0
  140. gea/quartz_hpht_extension.py +152 -0
  141. gea/reconciler.py +206 -0
  142. gea/rock_inventory.py +404 -0
  143. gea/sample_record.py +430 -0
  144. gea/sample_well_profile.csv +15 -0
  145. gea/sbom.py +116 -0
  146. gea/segy.py +181 -0
  147. gea/service_life.py +173 -0
  148. gea/shell.py +107 -0
  149. gea/sla_report.py +199 -0
  150. gea/store_forward.py +234 -0
  151. gea/strata_join.py +186 -0
  152. gea/survey_cmd.py +264 -0
  153. gea/survey_view.py +138 -0
  154. gea/telemetry.py +306 -0
  155. gea/tool_library.py +260 -0
  156. gea/well_assembler.py +457 -0
  157. gea/well_test_validation.py +369 -0
  158. gea_program-0.1.0.dist-info/METADATA +138 -0
  159. gea_program-0.1.0.dist-info/RECORD +163 -0
  160. gea_program-0.1.0.dist-info/WHEEL +5 -0
  161. gea_program-0.1.0.dist-info/entry_points.txt +2 -0
  162. gea_program-0.1.0.dist-info/licenses/LICENSE +373 -0
  163. gea_program-0.1.0.dist-info/top_level.txt +1 -0
gea/sample_record.py ADDED
@@ -0,0 +1,430 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """sample_record — the canonical measurement record and tag catalogue.
5
+
6
+ Every measurement the client-facing reports touch passes through ONE record
7
+ shape, so quality flags, gaps, staleness and ingest latency can be counted
8
+ and reported per tag in the vocabulary a production-operations client uses
9
+ (tag catalogue with owner / unit / engineering range; per-sample quality
10
+ flag with the rule that fired; latency measured per source layer).
11
+
12
+ SampleRecord(tag_id, timestamp_utc, value, unit, quality_flag,
13
+ rule_fired, source_layer, ingest_timestamp_utc)
14
+
15
+ Quality flags (fixed enumeration):
16
+
17
+ GOOD passed every rule in force for the tag
18
+ RANGE outside the tag's engineering range
19
+ ROC rate of change above the tag's limit
20
+ FLATLINE value unchanged for at least the tag's flatline run length
21
+ SPIKE single-sample excursion (from the source's own despike pass
22
+ or the rule here)
23
+ STALE the sample arrived after the tag's allowed silence
24
+ GAP no value (missing sample)
25
+
26
+ Source flags carried by an existing stream are mapped, never discarded:
27
+ OK -> GOOD, MISSING -> GAP, STUCK -> FLATLINE, SPIKE -> SPIKE.
28
+
29
+ Rules are engineering configuration, not derivations: each TagDefinition
30
+ holds its own limits, and every flagged record names the rule and the limit
31
+ that fired, so the client can audit the flag against the tag definition.
32
+
33
+ Headless-safe: numpy only.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import csv
39
+ from dataclasses import dataclass, field, asdict
40
+ from datetime import datetime, timedelta, timezone
41
+ from typing import Dict, Iterable, List, Optional, Tuple
42
+
43
+ import numpy as np
44
+
45
+ QUALITY_FLAGS = ('GOOD', 'RANGE', 'ROC', 'FLATLINE', 'SPIKE', 'STALE', 'GAP')
46
+ SOURCE_LAYERS = ('FIELD_EDGE', 'OT_LAKE', 'DOF')
47
+ LEGACY_FLAG_MAP = {'OK': 'GOOD', 'MISSING': 'GAP', 'STUCK': 'FLATLINE',
48
+ 'SPIKE': 'SPIKE', '': 'GOOD'}
49
+
50
+
51
+ def _iso(dt: datetime) -> str:
52
+ return dt.astimezone(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')
53
+
54
+
55
+ def parse_utc(s: str) -> datetime:
56
+ """ISO-8601 (date or datetime, optional Z) -> aware UTC datetime."""
57
+ s = s.strip()
58
+ if s.endswith('Z'):
59
+ s = s[:-1]
60
+ dt = datetime.fromisoformat(s)
61
+ if dt.tzinfo is None:
62
+ dt = dt.replace(tzinfo=timezone.utc)
63
+ return dt.astimezone(timezone.utc)
64
+
65
+
66
+ # ---------------------------------------------------------------------------
67
+ # Tag catalogue
68
+ # ---------------------------------------------------------------------------
69
+ @dataclass
70
+ class TagDefinition:
71
+ tag_id: str
72
+ description: str = ''
73
+ unit: str = ''
74
+ owner: str = 'Production Operations'
75
+ tag_class: str = 'downhole_pressure'
76
+ eng_range: Tuple[Optional[float], Optional[float]] = (None, None)
77
+ roc_limit_per_s: Optional[float] = None # |dv/dt| above this -> ROC
78
+ flatline_min_samples: int = 0 # 0 disables the FLATLINE rule
79
+ stale_after_s: Optional[float] = None # silence longer than this -> STALE
80
+ cadence_s: Optional[float] = None
81
+ source_layer: str = 'FIELD_EDGE'
82
+ source: str = '' # where the tag comes from (file, port, catalogue entry)
83
+ spike_n_sigma: Optional[float] = None # |v - rolling median| > n x MAD -> SPIKE (None disables)
84
+ spike_window: int = 21 # rolling window (odd) for the spike rule
85
+ limits_basis: str = 'class defaults' # where the limits came from (datasheet citation or setting)
86
+
87
+ def row(self) -> dict:
88
+ lo, hi = self.eng_range
89
+ return {
90
+ 'tag_id': self.tag_id, 'description': self.description, 'unit': self.unit,
91
+ 'owner': self.owner, 'tag_class': self.tag_class,
92
+ 'eng_range_lo': '' if lo is None else lo, 'eng_range_hi': '' if hi is None else hi,
93
+ 'roc_limit_per_s': '' if self.roc_limit_per_s is None else self.roc_limit_per_s,
94
+ 'flatline_min_samples': self.flatline_min_samples,
95
+ 'stale_after_s': '' if self.stale_after_s is None else self.stale_after_s,
96
+ 'cadence_s': '' if self.cadence_s is None else self.cadence_s,
97
+ 'source_layer': self.source_layer, 'source': self.source,
98
+ 'spike_n_sigma': '' if self.spike_n_sigma is None else self.spike_n_sigma,
99
+ 'spike_window': self.spike_window, 'limits_basis': self.limits_basis,
100
+ }
101
+
102
+
103
+ # Default rule sets per tag class. Values are engineering conventions for a
104
+ # downhole quartz gauge historian at the stated cadence; a client overrides
105
+ # them per tag in the catalogue, and the report prints whatever is in force.
106
+ DEFAULT_TAG_CLASSES: Dict[str, dict] = {
107
+ 'downhole_pressure': dict(unit='psi', eng_range=(0.0, 30000.0),
108
+ roc_limit_per_s=None, flatline_min_samples=3,
109
+ stale_after_s=None),
110
+ 'downhole_temperature': dict(unit='degF', eng_range=(-40.0, 500.0),
111
+ roc_limit_per_s=None, flatline_min_samples=3,
112
+ stale_after_s=None),
113
+ 'generic': dict(unit='', eng_range=(None, None), roc_limit_per_s=None,
114
+ flatline_min_samples=0, stale_after_s=None),
115
+ }
116
+
117
+
118
+ def rules_from_gauge_spec(spec, tag_class: str, cadence_s: Optional[float] = None,
119
+ roc_limit_per_s: Optional[float] = None,
120
+ stale_multiple: float = 3.0, flatline_samples: int = 3,
121
+ spike_n_sigma: Optional[float] = 6.0) -> dict:
122
+ """Quality limits for a tag class from a gauge datasheet (`GaugeSpec`).
123
+
124
+ What the datasheet supplies: the engineering RANGE - pressure 0 to
125
+ full_scale_psi (absolute gauge), temperature up to max_temp_C (converted
126
+ to degF; the lower bound is the class default since datasheets state a
127
+ rating, not a floor). What the datasheet does not supply and is therefore
128
+ an operations setting, printed as such: the rate-of-change limit (a real
129
+ shut-in is a fast transient the gauge sees correctly), the flatline run
130
+ length and the staleness multiple (both from cadence), and the spike
131
+ threshold. `limits_basis` carries the citation so the report can print
132
+ where each limit came from."""
133
+ d = dict(DEFAULT_TAG_CLASSES.get(tag_class, DEFAULT_TAG_CLASSES['generic']))
134
+ basis = []
135
+ if tag_class == 'downhole_pressure' and getattr(spec, 'full_scale_psi', None):
136
+ d['eng_range'] = (0.0, float(spec.full_scale_psi))
137
+ basis.append(f"range 0 to {spec.full_scale_psi:g} psi from datasheet '{spec.name}' full scale")
138
+ elif tag_class == 'downhole_temperature' and getattr(spec, 'max_temp_C', None) is not None:
139
+ hi_f = float(spec.max_temp_C) * 9.0 / 5.0 + 32.0
140
+ d['eng_range'] = (d['eng_range'][0], round(hi_f, 1))
141
+ basis.append(f"range upper bound {hi_f:.0f} degF from datasheet '{spec.name}' rating {spec.max_temp_C:g} C; lower bound class default")
142
+ else:
143
+ basis.append('range: class default')
144
+ d['roc_limit_per_s'] = roc_limit_per_s
145
+ basis.append('rate-of-change: operations setting' + (f' {roc_limit_per_s:g}/s' if roc_limit_per_s is not None else ' (off)'))
146
+ d['flatline_min_samples'] = flatline_samples
147
+ basis.append(f'flatline: {flatline_samples} samples at cadence')
148
+ d['stale_after_s'] = (cadence_s * stale_multiple) if cadence_s else None
149
+ basis.append(f'staleness: {stale_multiple:g} x cadence' if cadence_s else 'staleness: off (cadence unknown)')
150
+ d['spike_n_sigma'] = spike_n_sigma
151
+ basis.append(f'spike: {spike_n_sigma:g} x MAD of rolling median' if spike_n_sigma else 'spike: off')
152
+ d['limits_basis'] = '; '.join(basis)
153
+ d['cadence_s'] = cadence_s
154
+ return d
155
+
156
+
157
+ def _class_for(name: str, unit: str) -> str:
158
+ n = name.lower()
159
+ if unit == 'psi' or n.startswith('p_') or 'press' in n:
160
+ return 'downhole_pressure'
161
+ if unit in ('degF', 'degC') or n.startswith('t_') or 'temp' in n:
162
+ return 'downhole_temperature'
163
+ return 'generic'
164
+
165
+
166
+ class TagCatalogue:
167
+ """The canonical data model: one TagDefinition per tag."""
168
+
169
+ def __init__(self, tags: Optional[Iterable[TagDefinition]] = None):
170
+ self.tags: Dict[str, TagDefinition] = {}
171
+ for t in tags or ():
172
+ self.add(t)
173
+
174
+ def add(self, tag: TagDefinition) -> TagDefinition:
175
+ self.tags[tag.tag_id] = tag
176
+ return tag
177
+
178
+ def get(self, tag_id: str) -> TagDefinition:
179
+ return self.tags[tag_id]
180
+
181
+ def __contains__(self, tag_id: str) -> bool:
182
+ return tag_id in self.tags
183
+
184
+ def __len__(self) -> int:
185
+ return len(self.tags)
186
+
187
+ def rows(self) -> List[dict]:
188
+ return [t.row() for t in self.tags.values()]
189
+
190
+ @classmethod
191
+ def from_stream(cls, stream, owner: str = 'Production Operations',
192
+ source_layer: str = 'FIELD_EDGE', cadence_s: Optional[float] = None,
193
+ stale_multiple: float = 3.0, gauge_spec=None,
194
+ roc_limits: Optional[Dict[str, float]] = None,
195
+ spike_n_sigma: Optional[float] = 6.0) -> 'TagCatalogue':
196
+ """Build definitions for every channel of a LiveStream. With a gauge
197
+ datasheet (`gauge_spec`) the engineering ranges come from the
198
+ datasheet and the basis is recorded per tag; `roc_limits` maps a tag
199
+ class to a rate-of-change limit per second (operations setting).
200
+ Cadence from the argument or the stream meta."""
201
+ cad = cadence_s
202
+ if cad is None:
203
+ try:
204
+ cad = float(stream.meta.get('cadence_s'))
205
+ except (TypeError, ValueError):
206
+ cad = None
207
+ if cad is None and getattr(stream, 'index_kind', '') == 'time_s' and len(stream.index) > 1:
208
+ d = np.diff(np.asarray(stream.index, dtype=float))
209
+ d = d[d > 0]
210
+ cad = float(np.median(d)) if len(d) else None
211
+ cat = cls()
212
+ for name, ch in stream.channels.items():
213
+ k = _class_for(name, ch.unit)
214
+ roc = (roc_limits or {}).get(k)
215
+ if gauge_spec is not None:
216
+ d = rules_from_gauge_spec(gauge_spec, k, cadence_s=cad, roc_limit_per_s=roc,
217
+ stale_multiple=stale_multiple, spike_n_sigma=spike_n_sigma)
218
+ else:
219
+ d = dict(DEFAULT_TAG_CLASSES[k])
220
+ d['roc_limit_per_s'] = roc
221
+ d['stale_after_s'] = (cad * stale_multiple) if cad else None
222
+ d['spike_n_sigma'] = spike_n_sigma
223
+ d['limits_basis'] = ('class defaults; rate-of-change: ' + (f'operations setting {roc:g}/s' if roc is not None else 'off')
224
+ + f'; staleness: {stale_multiple:g} x cadence'
225
+ + (f'; spike: {spike_n_sigma:g} x MAD of rolling median' if spike_n_sigma else '; spike: off'))
226
+ if ch.unit:
227
+ d['unit'] = ch.unit
228
+ cat.add(TagDefinition(
229
+ tag_id=name, description=f'{name} from {stream.name}',
230
+ unit=d['unit'], owner=owner, tag_class=k,
231
+ eng_range=d['eng_range'], roc_limit_per_s=d['roc_limit_per_s'],
232
+ flatline_min_samples=d['flatline_min_samples'],
233
+ stale_after_s=d['stale_after_s'], cadence_s=cad, source_layer=source_layer,
234
+ source=str(stream.meta.get('path') or stream.meta.get('source_channel') or stream.source_format),
235
+ spike_n_sigma=d.get('spike_n_sigma'), limits_basis=d.get('limits_basis', 'class defaults')))
236
+ return cat
237
+
238
+
239
+ # ---------------------------------------------------------------------------
240
+ # The record
241
+ # ---------------------------------------------------------------------------
242
+ @dataclass
243
+ class SampleRecord:
244
+ tag_id: str
245
+ timestamp_utc: str
246
+ value: Optional[float]
247
+ unit: str
248
+ quality_flag: str = 'GOOD'
249
+ rule_fired: str = ''
250
+ source_layer: str = 'FIELD_EDGE'
251
+ ingest_timestamp_utc: str = ''
252
+
253
+ def latency_s(self) -> Optional[float]:
254
+ if not self.ingest_timestamp_utc:
255
+ return None
256
+ return (parse_utc(self.ingest_timestamp_utc) - parse_utc(self.timestamp_utc)).total_seconds()
257
+
258
+ def row(self) -> dict:
259
+ d = asdict(self)
260
+ d['value'] = '' if self.value is None or (isinstance(self.value, float) and np.isnan(self.value)) else self.value
261
+ return d
262
+
263
+
264
+ RECORD_COLUMNS = ['tag_id', 'timestamp_utc', 'value', 'unit', 'quality_flag',
265
+ 'rule_fired', 'source_layer', 'ingest_timestamp_utc']
266
+
267
+
268
+ # ---------------------------------------------------------------------------
269
+ # Quality rules
270
+ # ---------------------------------------------------------------------------
271
+ def apply_quality_rules(values: np.ndarray, times_s: np.ndarray, tag: TagDefinition,
272
+ source_flags: Optional[List[str]] = None) -> List[Tuple[str, str]]:
273
+ """Return (quality_flag, rule_fired) per sample.
274
+
275
+ Precedence when several rules fire on one sample: GAP > source flag
276
+ (FLATLINE/SPIKE from the stream's own pass) > RANGE > ROC > FLATLINE >
277
+ SPIKE > STALE > GOOD. The first rule in that order that fires is reported."""
278
+ v = np.asarray(values, dtype=float)
279
+ t = np.asarray(times_s, dtype=float)
280
+ n = len(v)
281
+ out: List[Tuple[str, str]] = [('GOOD', '')] * n
282
+ lo, hi = tag.eng_range
283
+ # SPIKE (rolling median / MAD, computed once; NaNs ignored inside the window)
284
+ spike = np.zeros(n, dtype=bool)
285
+ spike_dev = np.zeros(n)
286
+ if tag.spike_n_sigma and n >= 5:
287
+ h = max(1, int(tag.spike_window) // 2)
288
+ for i in range(n):
289
+ if np.isnan(v[i]):
290
+ continue
291
+ w = v[max(0, i - h):i + h + 1]
292
+ w = w[~np.isnan(w)]
293
+ if len(w) < 5:
294
+ continue
295
+ med = float(np.median(w))
296
+ mad = 1.4826 * float(np.median(np.abs(w - med)))
297
+ if mad > 0 and abs(v[i] - med) > tag.spike_n_sigma * mad:
298
+ spike[i] = True
299
+ spike_dev[i] = abs(v[i] - med) / mad
300
+ # FLATLINE runs (computed once)
301
+ flat = np.zeros(n, dtype=bool)
302
+ if tag.flatline_min_samples and tag.flatline_min_samples > 1:
303
+ run_start = 0
304
+ for i in range(1, n + 1):
305
+ if i == n or np.isnan(v[i]) or np.isnan(v[i - 1]) or v[i] != v[i - 1]:
306
+ if i - run_start >= tag.flatline_min_samples and not np.isnan(v[run_start]):
307
+ flat[run_start:i] = True
308
+ run_start = i
309
+ last_good_t: Optional[float] = None
310
+ for i in range(n):
311
+ if np.isnan(v[i]):
312
+ out[i] = ('GAP', 'no value')
313
+ continue
314
+ sf = LEGACY_FLAG_MAP.get((source_flags[i] if source_flags and i < len(source_flags) else '').strip().upper(), None) \
315
+ if source_flags else None
316
+ if sf in ('FLATLINE', 'SPIKE'):
317
+ out[i] = (sf, f'source flag {source_flags[i].strip()}')
318
+ elif lo is not None and v[i] < lo:
319
+ out[i] = ('RANGE', f'value {v[i]:g} < eng_range_lo {lo:g}')
320
+ elif hi is not None and v[i] > hi:
321
+ out[i] = ('RANGE', f'value {v[i]:g} > eng_range_hi {hi:g}')
322
+ elif tag.roc_limit_per_s is not None and i > 0 and not np.isnan(v[i - 1]) and t[i] > t[i - 1] \
323
+ and abs((v[i] - v[i - 1]) / (t[i] - t[i - 1])) > tag.roc_limit_per_s:
324
+ out[i] = ('ROC', f'|dv/dt| {abs((v[i]-v[i-1])/(t[i]-t[i-1])):.4g}/s > roc_limit {tag.roc_limit_per_s:g}/s')
325
+ elif flat[i]:
326
+ out[i] = ('FLATLINE', f'unchanged >= {tag.flatline_min_samples} samples')
327
+ elif spike[i]:
328
+ out[i] = ('SPIKE', f'|v - rolling median| = {spike_dev[i]:.1f} x MAD > {tag.spike_n_sigma:g} x MAD (window {tag.spike_window})')
329
+ elif tag.stale_after_s is not None and last_good_t is not None and (t[i] - last_good_t) > tag.stale_after_s:
330
+ out[i] = ('STALE', f'silence {t[i]-last_good_t:.0f} s > stale_after {tag.stale_after_s:g} s')
331
+ else:
332
+ out[i] = ('GOOD', '')
333
+ last_good_t = t[i]
334
+ return out
335
+
336
+
337
+ def records_from_stream(stream, catalogue: Optional[TagCatalogue] = None,
338
+ t0_utc: Optional[str] = None, source_layer: Optional[str] = None,
339
+ ingest_utc: Optional[str] = None) -> List[SampleRecord]:
340
+ """Every channel of a time-indexed LiveStream -> SampleRecords with the
341
+ catalogue's rules applied. t0_utc anchors elapsed seconds; when absent
342
+ the stream's start_date meta is used, and failing that 1970-01-01 with
343
+ the timestamps understood as elapsed time from an unknown origin."""
344
+ if getattr(stream, 'index_kind', 'time_s') != 'time_s':
345
+ raise ValueError('records_from_stream needs a time-indexed stream')
346
+ if catalogue is None:
347
+ catalogue = TagCatalogue.from_stream(stream, source_layer=source_layer or 'FIELD_EDGE')
348
+ t0s = t0_utc or stream.meta.get('start_date') or stream.meta.get('start_time') or '1970-01-01T00:00:00Z'
349
+ t0 = parse_utc(t0s)
350
+ times = np.asarray(stream.index, dtype=float)
351
+ recs: List[SampleRecord] = []
352
+ for name, ch in stream.channels.items():
353
+ tag = catalogue.get(name) if name in catalogue else catalogue.add(
354
+ TagDefinition(tag_id=name, unit=ch.unit, tag_class='generic', source=stream.name))
355
+ flags = apply_quality_rules(ch.values, times, tag, getattr(ch, 'quality', None))
356
+ layer = source_layer or tag.source_layer
357
+ for i, (q, rule) in enumerate(flags):
358
+ val = float(ch.values[i])
359
+ recs.append(SampleRecord(
360
+ tag_id=name, timestamp_utc=_iso(t0 + timedelta(seconds=float(times[i]))),
361
+ value=None if np.isnan(val) else val, unit=tag.unit,
362
+ quality_flag=q, rule_fired=rule, source_layer=layer,
363
+ ingest_timestamp_utc=ingest_utc or ''))
364
+ return recs
365
+
366
+
367
+ # ---------------------------------------------------------------------------
368
+ # Data quality summary (the §4.2.2 table)
369
+ # ---------------------------------------------------------------------------
370
+ def quality_summary(records: Iterable[SampleRecord]) -> Dict[str, dict]:
371
+ """Per tag: n, counts per flag, % GOOD, longest gap (samples and seconds),
372
+ first/last timestamp, latency p95 where ingest stamps exist."""
373
+ by: Dict[str, List[SampleRecord]] = {}
374
+ for r in records:
375
+ by.setdefault(r.tag_id, []).append(r)
376
+ out: Dict[str, dict] = {}
377
+ for tag, rs in by.items():
378
+ rs = sorted(rs, key=lambda r: r.timestamp_utc)
379
+ counts = {f: 0 for f in QUALITY_FLAGS}
380
+ for r in rs:
381
+ counts[r.quality_flag] = counts.get(r.quality_flag, 0) + 1
382
+ n = len(rs)
383
+ # longest GAP run
384
+ best = cur = 0
385
+ best_span = 0.0
386
+ run_start = None
387
+ for i, r in enumerate(rs):
388
+ if r.quality_flag == 'GAP':
389
+ if cur == 0:
390
+ run_start = i
391
+ cur += 1
392
+ if cur > best:
393
+ best = cur
394
+ a = parse_utc(rs[run_start].timestamp_utc)
395
+ b = parse_utc(rs[i].timestamp_utc)
396
+ best_span = (b - a).total_seconds()
397
+ else:
398
+ cur = 0
399
+ lat = [r.latency_s() for r in rs if r.ingest_timestamp_utc]
400
+ lat = [x for x in lat if x is not None]
401
+ out[tag] = {
402
+ 'n': n, 'unit': rs[0].unit, 'source_layer': rs[0].source_layer,
403
+ 'counts': counts,
404
+ 'pct_good': round(100.0 * counts['GOOD'] / n, 2) if n else 0.0,
405
+ 'longest_gap_samples': best, 'longest_gap_s': round(best_span, 0),
406
+ 'first_utc': rs[0].timestamp_utc, 'last_utc': rs[-1].timestamp_utc,
407
+ 'latency_p95_s': (round(float(np.percentile(lat, 95)), 1) if lat else None),
408
+ 'rule_examples': {f: next(r.rule_fired for r in rs if r.quality_flag == f)
409
+ for f in QUALITY_FLAGS if counts.get(f) and f != 'GOOD'},
410
+ }
411
+ return out
412
+
413
+
414
+ def write_records_csv(records: Iterable[SampleRecord], path) -> str:
415
+ with open(path, 'w', newline='', encoding='utf-8') as f:
416
+ w = csv.DictWriter(f, fieldnames=RECORD_COLUMNS)
417
+ w.writeheader()
418
+ for r in records:
419
+ w.writerow(r.row())
420
+ return str(path)
421
+
422
+
423
+ def write_catalogue_csv(catalogue: TagCatalogue, path) -> str:
424
+ rows = catalogue.rows()
425
+ with open(path, 'w', newline='', encoding='utf-8') as f:
426
+ w = csv.DictWriter(f, fieldnames=list(rows[0].keys()) if rows else ['tag_id'])
427
+ w.writeheader()
428
+ for r in rows:
429
+ w.writerow(r)
430
+ return str(path)
@@ -0,0 +1,15 @@
1
+ depth_ft,pressure_psi,temp_F
2
+ 0,14.7,75
3
+ 2000,960,110
4
+ 4000,1900,146
5
+ 6000,2850,182
6
+ 8000,3800,218
7
+ 10000,4750,254
8
+ 12000,5700,290
9
+ 14000,6650,326
10
+ 15000,7600,348
11
+ 16500,9800,382
12
+ 17500,12600,408
13
+ 18500,15400,428
14
+ 19500,17600,444
15
+ 20300,18900,452
gea/sbom.py ADDED
@@ -0,0 +1,116 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """sbom — the software bill of materials, generated from the running
5
+ environment (SCC 16.0; software with licences).
6
+
7
+ Eight fields per component, the minimum a client inspection asks for:
8
+ name, version, supplier, licence, hash, identifier (purl), relationship
9
+ (root / direct / optional), generated timestamp.
10
+
11
+ Nothing is typed in by hand: versions and licences come from the installed
12
+ distributions' metadata (importlib.metadata); the program's own components
13
+ are hashed from the source files that ship; optional components (plotting,
14
+ desktop UI) are listed as optional whether or not they are installed, with
15
+ 'not installed' recorded when absent. A component whose licence is not
16
+ declared in its metadata is recorded as 'not declared in metadata', never
17
+ guessed.
18
+
19
+ Headless-safe: standard library only.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import hashlib
25
+ import json
26
+ import os
27
+ import platform
28
+ import sys
29
+ from datetime import datetime, timezone
30
+ from typing import Dict, List, Optional
31
+
32
+ _HERE = os.path.dirname(os.path.abspath(__file__))
33
+ OPTIONAL = {'matplotlib': 'plotting (figures in reports)', 'PyQt6': 'desktop operator application'}
34
+ DIRECT = {'numpy': 'numerical arrays (all engines)'}
35
+
36
+
37
+ def _iso() -> str:
38
+ return datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ')
39
+
40
+
41
+ def _dist(name: str) -> Optional[dict]:
42
+ try:
43
+ import importlib.metadata as m
44
+ md = m.metadata(name)
45
+ except Exception:
46
+ return None
47
+ lic = md.get('License-Expression') or md.get('License') or ''
48
+ if not lic or len(lic) > 80:
49
+ cls = [c for c in md.get_all('Classifier') or [] if c.startswith('License ::')]
50
+ lic = cls[-1].split('::')[-1].strip() if cls else (lic[:80] if lic else 'not declared in metadata')
51
+ files = []
52
+ try:
53
+ files = [f for f in (m.files(name) or []) if str(f).endswith('RECORD')]
54
+ except Exception:
55
+ pass
56
+ h = 'n/a'
57
+ try:
58
+ if files:
59
+ h = hashlib.sha256(files[0].read_binary()).hexdigest()[:16]
60
+ except Exception:
61
+ pass
62
+ return {'version': md.get('Version', ''), 'supplier': (md.get('Author') or md.get('Maintainer') or md.get('Author-email') or 'not declared in metadata')[:60],
63
+ 'licence': lic, 'hash': h, 'home': md.get('Home-page', '')}
64
+
65
+
66
+ def _tree_hash(dirpath: str) -> str:
67
+ h = hashlib.sha256()
68
+ for root, _, files in sorted(os.walk(dirpath)):
69
+ for fn in sorted(files):
70
+ if fn.endswith('.py'):
71
+ p = os.path.join(root, fn)
72
+ h.update(fn.encode()); h.update(open(p, 'rb').read())
73
+ return h.hexdigest()[:16]
74
+
75
+
76
+ def generate(program_name: str = 'Downhole Gauge Monitoring', program_version: str = '') -> dict:
77
+ from . import __version__
78
+ ver = program_version or __version__
79
+ ts = _iso()
80
+ comps: List[dict] = []
81
+ comps.append({'name': program_name, 'version': ver, 'supplier': 'ENRGYONE', 'licence': 'MPL-2.0',
82
+ 'hash': _tree_hash(_HERE), 'identifier': f'pkg:generic/downhole-gauge-monitoring@{ver}',
83
+ 'relationship': 'root', 'generated_utc': ts})
84
+ comps.append({'name': 'Python', 'version': platform.python_version(), 'supplier': 'Python Software Foundation',
85
+ 'licence': 'PSF License', 'hash': 'n/a (runtime)', 'identifier': f'pkg:generic/python@{platform.python_version()}',
86
+ 'relationship': 'runtime', 'generated_utc': ts})
87
+ for name, role in DIRECT.items():
88
+ d = _dist(name)
89
+ comps.append({'name': name, 'version': d['version'] if d else 'not installed', 'supplier': d['supplier'] if d else '-',
90
+ 'licence': d['licence'] if d else '-', 'hash': d['hash'] if d else '-',
91
+ 'identifier': f"pkg:pypi/{name.lower()}@{d['version']}" if d else f'pkg:pypi/{name.lower()}',
92
+ 'relationship': f'direct ({role})', 'generated_utc': ts})
93
+ for name, role in OPTIONAL.items():
94
+ d = _dist(name)
95
+ comps.append({'name': name, 'version': d['version'] if d else 'not installed', 'supplier': d['supplier'] if d else '-',
96
+ 'licence': d['licence'] if d else '-', 'hash': d['hash'] if d else '-',
97
+ 'identifier': f"pkg:pypi/{name.lower()}@{d['version']}" if d else f'pkg:pypi/{name.lower()}',
98
+ 'relationship': f'optional ({role})', 'generated_utc': ts})
99
+ return {'format': 'eight-field SBOM (name, version, supplier, licence, hash, identifier, relationship, generated)',
100
+ 'generated_utc': ts, 'platform': platform.platform(), 'python': sys.version.split()[0],
101
+ 'components': comps, 'n_components': len(comps)}
102
+
103
+
104
+ def write(sbom: dict, out_dir: str, basename: str = 'sbom') -> Dict[str, str]:
105
+ import csv
106
+ os.makedirs(out_dir, exist_ok=True)
107
+ pj = os.path.join(out_dir, basename + '.json')
108
+ pc = os.path.join(out_dir, basename + '.csv')
109
+ with open(pj, 'w', encoding='utf-8') as f:
110
+ json.dump(sbom, f, indent=1)
111
+ with open(pc, 'w', newline='', encoding='utf-8') as f:
112
+ w = csv.DictWriter(f, fieldnames=['name', 'version', 'supplier', 'licence', 'hash', 'identifier', 'relationship', 'generated_utc'])
113
+ w.writeheader()
114
+ for c in sbom['components']:
115
+ w.writerow(c)
116
+ return {'json': pj, 'csv': pc}