gea-program 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. gea/BENCH_TEST_PROTOCOL.md +97 -0
  2. gea/IMPORT_RECORD.md +61 -0
  3. gea/__init__.py +160 -0
  4. gea/__main__.py +661 -0
  5. gea/acceptance_tests.py +1315 -0
  6. gea/accuracy_statement.py +181 -0
  7. gea/alarm_engine.py +307 -0
  8. gea/bench.py +144 -0
  9. gea/blind_harness.py +108 -0
  10. gea/case_study.py +229 -0
  11. gea/catalog/acex_lomonosov_age_depth_model.provenance.json +23 -0
  12. gea/catalog/acex_lomonosov_age_depth_model.txt +28 -0
  13. gea/catalog/agassiz77_canada_temperature.csv +68 -0
  14. gea/catalog/agassiz77_canada_temperature.provenance.json +17 -0
  15. gea/catalog/barbados_110_consolidation.provenance.json +24 -0
  16. gea/catalog/barbados_110_consolidation.txt +91 -0
  17. gea/catalog/bengal_u1452_grain_size.provenance.json +20 -0
  18. gea/catalog/bengal_u1452_grain_size.txt +252 -0
  19. gea/catalog/blake_164_methane_isotopes.provenance.json +22 -0
  20. gea/catalog/blake_164_methane_isotopes.txt +68 -0
  21. gea/catalog/chicxulub_m0077a_pwave_velocity.provenance.json +23 -0
  22. gea/catalog/chicxulub_m0077a_pwave_velocity.txt +735 -0
  23. gea/catalog/collingwood_1_28_ks_complete.las +128 -0
  24. gea/catalog/collingwood_1_28_ks_complete.provenance.json +17 -0
  25. gea/catalog/costa_rica_odp_friction_envelope.provenance.json +24 -0
  26. gea/catalog/costa_rica_odp_friction_envelope.txt +57 -0
  27. gea/catalog/dead_sea_5017_debrite_xrf_ms.provenance.json +21 -0
  28. gea/catalog/dead_sea_5017_debrite_xrf_ms.txt +94 -0
  29. gea/catalog/dsdp_504b_physical_properties.provenance.json +24 -0
  30. gea/catalog/dsdp_504b_physical_properties.txt +82 -0
  31. gea/catalog/dsdp_504b_sound_velocity.provenance.json +21 -0
  32. gea/catalog/dsdp_504b_sound_velocity.txt +81 -0
  33. gea/catalog/elgygytgyn_5011_turbidites.provenance.json +20 -0
  34. gea/catalog/elgygytgyn_5011_turbidites.txt +193 -0
  35. gea/catalog/epica_domec_co2_800kyr.provenance.json +21 -0
  36. gea/catalog/epica_domec_co2_800kyr.txt +265 -0
  37. gea/catalog/fram_909_organic_petrography.provenance.json +22 -0
  38. gea/catalog/fram_909_organic_petrography.txt +40 -0
  39. gea/catalog/gbr_325_coral_uth_ages.provenance.json +23 -0
  40. gea/catalog/gbr_325_coral_uth_ages.txt +71 -0
  41. gea/catalog/gisp2_greenland_temperature.csv +599 -0
  42. gea/catalog/gisp2_greenland_temperature.provenance.json +17 -0
  43. gea/catalog/gom_308_t2p_insitu.provenance.json +21 -0
  44. gea/catalog/gom_308_t2p_insitu.txt +40 -0
  45. gea/catalog/guaymas_385_dom_d13c.provenance.json +21 -0
  46. gea/catalog/guaymas_385_dom_d13c.txt +103 -0
  47. gea/catalog/hikurangi_u1520_friction_insitu.provenance.json +26 -0
  48. gea/catalog/hikurangi_u1520_friction_insitu.txt +74 -0
  49. gea/catalog/hydrate_ridge_204_ncr_hydrate.provenance.json +22 -0
  50. gea/catalog/hydrate_ridge_204_ncr_hydrate.txt +60 -0
  51. gea/catalog/iodp_u1324_pore_pressure.provenance.json +22 -0
  52. gea/catalog/iodp_u1324_pore_pressure.txt +42 -0
  53. gea/catalog/jfast_c0019_slow_slip_events.provenance.json +28 -0
  54. gea/catalog/jfast_c0019_slow_slip_events.txt +35 -0
  55. gea/catalog/kennetcook_2_p129_excerpt.las +139 -0
  56. gea/catalog/kennetcook_2_p129_excerpt.provenance.json +17 -0
  57. gea/catalog/ktb_hb_bhgm_density.dat +227 -0
  58. gea/catalog/ktb_hb_bhgm_density.provenance.json +22 -0
  59. gea/catalog/ktb_hb_complog_6020_excerpt.provenance.json +20 -0
  60. gea/catalog/ktb_hb_complog_6020_excerpt.txt +72 -0
  61. gea/catalog/ktb_hb_hlog246_temperature.dat +1636 -0
  62. gea/catalog/ktb_hb_hlog246_temperature.provenance.json +22 -0
  63. gea/catalog/ktb_hb_rockmech_compress.dat +33 -0
  64. gea/catalog/ktb_hb_rockmech_compress.provenance.json +22 -0
  65. gea/catalog/ktb_hb_tvd_0_9080_excerpt.dat +2817 -0
  66. gea/catalog/ktb_hb_tvd_0_9080_excerpt.provenance.json +20 -0
  67. gea/catalog/ktb_vb_rockmech_compress.dat +125 -0
  68. gea/catalog/ktb_vb_rockmech_compress.provenance.json +22 -0
  69. gea/catalog/ktb_vb_vlog251_temperature.dat +1127 -0
  70. gea/catalog/ktb_vb_vlog251_temperature.provenance.json +24 -0
  71. gea/catalog/l06_06_nl_survey.csv +201 -0
  72. gea/catalog/l06_06_nl_survey.provenance.json +17 -0
  73. gea/catalog/l07_01_nl_excerpt.las +90 -0
  74. gea/catalog/l07_01_nl_excerpt.provenance.json +17 -0
  75. gea/catalog/mariana_1200_serpentinite_geochem.provenance.json +21 -0
  76. gea/catalog/mariana_1200_serpentinite_geochem.txt +63 -0
  77. gea/catalog/med_160_sapropels.provenance.json +22 -0
  78. gea/catalog/med_160_sapropels.txt +43 -0
  79. gea/catalog/nankai_megasplay_shear_strength.provenance.json +26 -0
  80. gea/catalog/nankai_megasplay_shear_strength.txt +42 -0
  81. gea/catalog/odp_1027b_thermal_conductivity.provenance.json +21 -0
  82. gea/catalog/odp_1027b_thermal_conductivity.txt +52 -0
  83. gea/catalog/odp_1027c_cork_temperature.provenance.json +20 -0
  84. gea/catalog/odp_1027c_cork_temperature.txt +26 -0
  85. gea/catalog/odp_1165b_thermal_conductivity.provenance.json +22 -0
  86. gea/catalog/odp_1165b_thermal_conductivity.txt +102 -0
  87. gea/catalog/odp_1274a_mantle_peridotite_mad.provenance.json +27 -0
  88. gea/catalog/odp_1274a_mantle_peridotite_mad.txt +44 -0
  89. gea/catalog/odp_504b_dike_elastic_moduli.provenance.json +27 -0
  90. gea/catalog/odp_504b_dike_elastic_moduli.txt +85 -0
  91. gea/catalog/odp_504b_leg137_borehole_fluids.provenance.json +19 -0
  92. gea/catalog/odp_504b_leg137_borehole_fluids.txt +68 -0
  93. gea/catalog/odp_735b_gabbro_elastic_moduli.provenance.json +28 -0
  94. gea/catalog/odp_735b_gabbro_elastic_moduli.txt +127 -0
  95. gea/catalog/peru_201_sulfate_reduction.provenance.json +22 -0
  96. gea/catalog/peru_201_sulfate_reduction.txt +322 -0
  97. gea/catalog/scorpio_e1_sa_excerpt.las +113 -0
  98. gea/catalog/scorpio_e1_sa_excerpt.provenance.json +17 -0
  99. gea/catalog/sumatra_362_cohesion.provenance.json +21 -0
  100. gea/catalog/sumatra_362_cohesion.txt +38 -0
  101. gea/catalog/university_6_17_no1_tx_excerpt.las +119 -0
  102. gea/catalog/university_6_17_no1_tx_excerpt.provenance.json +17 -0
  103. gea/catalog/ursa_308_xrd_mineralogy.provenance.json +21 -0
  104. gea/catalog/ursa_308_xrd_mineralogy.txt +46 -0
  105. gea/catalog/volve_15_9_19_sr_excerpt.las +183 -0
  106. gea/catalog/volve_15_9_19_sr_excerpt.provenance.json +16 -0
  107. gea/catalog/volve_15_9_19a_core_excerpt.csv +88 -0
  108. gea/catalog/volve_15_9_19a_core_excerpt.provenance.json +18 -0
  109. gea/catalog/volve_f12_f14_production_excerpt.csv +167 -0
  110. gea/catalog/volve_f12_f14_production_excerpt.provenance.json +21 -0
  111. gea/catalog/walvis_208_petm_carbonate.provenance.json +21 -0
  112. gea/catalog/walvis_208_petm_carbonate.txt +268 -0
  113. gea/catalog/woodlark_1109_rock_eval.provenance.json +23 -0
  114. gea/catalog/woodlark_1109_rock_eval.txt +30 -0
  115. gea/cli.py +125 -0
  116. gea/client_reports.py +943 -0
  117. gea/config_versioning.py +133 -0
  118. gea/correlation.py +155 -0
  119. gea/dashboard.py +390 -0
  120. gea/deviation.py +70 -0
  121. gea/downhole_engine.py +395 -0
  122. gea/drift_monitor.py +310 -0
  123. gea/earth_model.py +230 -0
  124. gea/example_register_map.json +14 -0
  125. gea/fat_sat.py +68 -0
  126. gea/follower.py +98 -0
  127. gea/forward_model.py +139 -0
  128. gea/gamma.py +176 -0
  129. gea/gauge_specs.py +112 -0
  130. gea/gravity_reference.py +116 -0
  131. gea/inverse_engine.py +215 -0
  132. gea/matplotlib_demo.py +85 -0
  133. gea/modbus.py +229 -0
  134. gea/model_card.py +248 -0
  135. gea/operator_app.py +442 -0
  136. gea/ports.py +340 -0
  137. gea/profile_catalog.py +773 -0
  138. gea/project.py +213 -0
  139. gea/qt6_downhole_app.py +144 -0
  140. gea/quartz_hpht_extension.py +152 -0
  141. gea/reconciler.py +206 -0
  142. gea/rock_inventory.py +404 -0
  143. gea/sample_record.py +430 -0
  144. gea/sample_well_profile.csv +15 -0
  145. gea/sbom.py +116 -0
  146. gea/segy.py +181 -0
  147. gea/service_life.py +173 -0
  148. gea/shell.py +107 -0
  149. gea/sla_report.py +199 -0
  150. gea/store_forward.py +234 -0
  151. gea/strata_join.py +186 -0
  152. gea/survey_cmd.py +264 -0
  153. gea/survey_view.py +138 -0
  154. gea/telemetry.py +306 -0
  155. gea/tool_library.py +260 -0
  156. gea/well_assembler.py +457 -0
  157. gea/well_test_validation.py +369 -0
  158. gea_program-0.1.0.dist-info/METADATA +138 -0
  159. gea_program-0.1.0.dist-info/RECORD +163 -0
  160. gea_program-0.1.0.dist-info/WHEEL +5 -0
  161. gea_program-0.1.0.dist-info/entry_points.txt +2 -0
  162. gea_program-0.1.0.dist-info/licenses/LICENSE +373 -0
  163. gea_program-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,181 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """accuracy_statement — the accuracy statement in the client's statistic.
5
+
6
+ A production-operations client states tool accuracy as MAPE (mean absolute
7
+ percentage error) at a stated confidence interval, per predicted quantity,
8
+ per well, with n and the back-test window, and reads the result against an
9
+ accuracy band. This module produces exactly that from per-trial back-test
10
+ results, and a back-test runner that re-scores the library's blind
11
+ (leave-one-out) predictions trial by trial so the statistic is computed on
12
+ the raw errors, never on a summary.
13
+
14
+ Definitions in force (printed on every statement; client-configurable):
15
+
16
+ APE_i = |estimate_i - truth_i| / |truth_i| x 100
17
+ MAPE = mean(APE_i)
18
+ Accuracy = 100 - MAPE
19
+ CI = bootstrap percentile interval on MAPE (default 90 %,
20
+ 2,000 resamples, fixed seed - reproducible)
21
+ Coverage90 = fraction of trials whose truth lies inside the estimate's
22
+ own +/- 1.645 sigma band (calibration check; ~0.90 expected)
23
+ Band = accuracy read against the thresholds
24
+ MEETS_TARGET >= 95 / BAND_2 90-95 / BAND_3 85-90 /
25
+ NOT_ACCEPTABLE < 85, using the CONSERVATIVE end of the CI
26
+ (accuracy at the upper CI bound of MAPE) so a statement
27
+ never claims a band the interval does not support.
28
+
29
+ Trials below MIN_TRIALS are reported PENDING with n, never scored.
30
+
31
+ Headless-safe: numpy only.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ import statistics
37
+ from dataclasses import dataclass, asdict
38
+ from typing import Dict, List, Optional, Sequence
39
+
40
+ import numpy as np
41
+
42
+ MIN_TRIALS = 10
43
+ DEFAULT_CI = 0.90
44
+ N_BOOT = 2000
45
+ SEED = 20260928
46
+ Z90 = 1.6448536269514722
47
+
48
+ BANDS = [(95.0, 'MEETS_TARGET'), (90.0, 'BAND_2'), (85.0, 'BAND_3'), (-1e9, 'NOT_ACCEPTABLE')]
49
+
50
+
51
+ def band_for(accuracy_pct: float) -> str:
52
+ for thr, name in BANDS:
53
+ if accuracy_pct >= thr:
54
+ return name
55
+ return 'NOT_ACCEPTABLE'
56
+
57
+
58
+ @dataclass
59
+ class Trial:
60
+ truth: float
61
+ estimate: float
62
+ std: float # the estimate's own 1-sigma spread (0 if none)
63
+
64
+
65
+ def statement_from_trials(trials: Sequence[Trial], ci: float = DEFAULT_CI,
66
+ n_boot: int = N_BOOT, seed: int = SEED) -> dict:
67
+ """MAPE with a bootstrap CI, coverage at the CI level, and the band."""
68
+ n = len(trials)
69
+ if n < MIN_TRIALS:
70
+ return {'status': 'PENDING', 'n': n, 'min_n': MIN_TRIALS}
71
+ truth = np.array([t.truth for t in trials], dtype=float)
72
+ est = np.array([t.estimate for t in trials], dtype=float)
73
+ std = np.array([t.std for t in trials], dtype=float)
74
+ nz = np.abs(truth) > 0
75
+ if nz.sum() < MIN_TRIALS:
76
+ return {'status': 'PENDING', 'n': int(nz.sum()), 'min_n': MIN_TRIALS,
77
+ 'note': 'truth values at zero cannot carry a percentage error'}
78
+ ape = np.abs(est[nz] - truth[nz]) / np.abs(truth[nz]) * 100.0
79
+ mape = float(ape.mean())
80
+ rng = np.random.default_rng(seed)
81
+ idx = rng.integers(0, len(ape), size=(n_boot, len(ape)))
82
+ boots = ape[idx].mean(axis=1)
83
+ alpha = (1.0 - ci) / 2.0
84
+ lo, hi = float(np.percentile(boots, 100 * alpha)), float(np.percentile(boots, 100 * (1 - alpha)))
85
+ z = Z90 if abs(ci - 0.90) < 1e-9 else float(_z_for(ci))
86
+ with_std = std > 0
87
+ cov = float(np.mean(np.abs(est[with_std] - truth[with_std]) <= z * std[with_std])) if with_std.any() else None
88
+ acc = 100.0 - mape
89
+ acc_conservative = 100.0 - hi
90
+ return {
91
+ 'status': 'OK', 'n': int(len(ape)), 'ci': ci,
92
+ 'mape_pct': round(mape, 3), 'mape_ci_lo_pct': round(lo, 3), 'mape_ci_hi_pct': round(hi, 3),
93
+ 'accuracy_pct': round(acc, 3), 'accuracy_conservative_pct': round(acc_conservative, 3),
94
+ 'coverage_at_ci': (round(cov, 3) if cov is not None else None),
95
+ 'n_with_spread': int(with_std.sum()),
96
+ 'band': band_for(acc_conservative), 'band_point': band_for(acc),
97
+ 'max_ape_pct': round(float(ape.max()), 3), 'median_ape_pct': round(float(np.median(ape)), 3),
98
+ }
99
+
100
+
101
+ def _z_for(ci: float) -> float:
102
+ # two-sided normal quantile by bisection on the error function (no scipy)
103
+ import math
104
+ target = ci
105
+ lo, hi = 0.0, 10.0
106
+ for _ in range(80):
107
+ mid = (lo + hi) / 2
108
+ if math.erf(mid / math.sqrt(2)) < target:
109
+ lo = mid
110
+ else:
111
+ hi = mid
112
+ return (lo + hi) / 2
113
+
114
+
115
+ # ---------------------------------------------------------------------------
116
+ # Back-test runner: the library's blind predictions, trial by trial
117
+ # ---------------------------------------------------------------------------
118
+ def _loo_trials(pairs: List[tuple], k: int = 7) -> List[Trial]:
119
+ from .inverse_engine import _conditional_from_pairs
120
+ out: List[Trial] = []
121
+ for i in range(len(pairs)):
122
+ rest = pairs[:i] + pairs[i + 1:]
123
+ g, t = pairs[i]
124
+ c = _conditional_from_pairs(rest, g, k)
125
+ if c['status'] != 'OK':
126
+ continue
127
+ out.append(Trial(truth=float(t), estimate=float(c['estimate']), std=float(c['std'])))
128
+ return out
129
+
130
+
131
+ def library_backtest(ci: float = DEFAULT_CI, k: int = 7) -> dict:
132
+ """Leave-one-out back-test of every supported property pair in the
133
+ library, scored as an accuracy statement per pair. Same estimator, same
134
+ library, same hold-out as the standing blind harness; the difference is
135
+ that the raw trials are kept so MAPE and its CI can be computed."""
136
+ from . import strata_join as SJ
137
+ from .inverse_engine import _site_native_pairs
138
+ rows = []
139
+ for well, (_, props) in SJ.WELL_GROUPS.items():
140
+ names = sorted(props)
141
+ for i, a in enumerate(names):
142
+ for b in names[i + 1:]:
143
+ pairs, _bin = SJ._colocated(well, a, b, None)
144
+ st = statement_from_trials(_loo_trials(pairs, k), ci=ci)
145
+ st.update({'well': well, 'given': a, 'target': b, 'source': 'strata_join',
146
+ 'n_pairs': len(pairs)})
147
+ rows.append(st)
148
+ try:
149
+ pairs, washouts = _site_native_pairs('ktb_hb_complog_6020_excerpt')
150
+ st = statement_from_trials(_loo_trials(pairs, k), ci=ci)
151
+ st.update({'well': 'ktb_complog', 'given': 'rho (g/cc)', 'target': 'Vp (m/s)',
152
+ 'source': 'site_pairs', 'n_pairs': len(pairs), 'washouts_excluded': washouts})
153
+ rows.append(st)
154
+ except KeyError:
155
+ pass
156
+ ok = [r for r in rows if r['status'] == 'OK']
157
+ pending = [r for r in rows if r['status'] != 'OK']
158
+ ok.sort(key=lambda r: r['mape_pct'])
159
+ bands: Dict[str, int] = {}
160
+ for r in ok:
161
+ bands[r['band']] = bands.get(r['band'], 0) + 1
162
+ return {
163
+ 'method': {'hold_out': 'leave-one-out', 'estimator': f'k-nearest empirical conditional, k={k}',
164
+ 'ci': ci, 'bootstrap_resamples': N_BOOT, 'seed': SEED, 'min_trials': MIN_TRIALS,
165
+ 'band_rule': 'band from accuracy at the upper CI bound of MAPE (conservative)'},
166
+ 'statements': ok, 'pending': pending, 'n_ok': len(ok), 'n_pending': len(pending),
167
+ 'bands': bands,
168
+ 'worst_mape_pct': (ok[-1]['mape_pct'] if ok else None),
169
+ 'best_mape_pct': (ok[0]['mape_pct'] if ok else None),
170
+ 'median_coverage_at_ci': (statistics.median([r['coverage_at_ci'] for r in ok if r['coverage_at_ci'] is not None])
171
+ if any(r['coverage_at_ci'] is not None for r in ok) else None),
172
+ }
173
+
174
+
175
+ def statement_from_arrays(truth, estimate, std=None, ci: float = DEFAULT_CI) -> dict:
176
+ """Convenience: arrays -> statement (for any predictor, gauge track included)."""
177
+ truth = np.asarray(truth, dtype=float)
178
+ estimate = np.asarray(estimate, dtype=float)
179
+ std = np.zeros_like(truth) if std is None else np.asarray(std, dtype=float)
180
+ m = ~(np.isnan(truth) | np.isnan(estimate))
181
+ return statement_from_trials([Trial(float(t), float(e), float(s)) for t, e, s in zip(truth[m], estimate[m], std[m])], ci=ci)
gea/alarm_engine.py ADDED
@@ -0,0 +1,307 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """alarm_engine — alarm definitions, the alarm state machine, the event log
5
+ and the alarm-management KPIs (SOW 4.2.4; ISA-18.2 / IEC 62682 practice).
6
+
7
+ Definitions are engineering configuration: each alarm names its tag, kind,
8
+ setpoint, deadband (return-to-normal hysteresis in engineering units),
9
+ on-delay (seconds the condition must persist before the alarm activates),
10
+ priority (P1 highest .. P5 lowest) and the BASIS of the setpoint (a
11
+ datasheet, a client setting, the record layer). Nothing here invents a
12
+ setpoint: `defaults_from_catalogue` derives only instrument over-range
13
+ alarms from the engineering range the tag catalogue already carries, plus a
14
+ quality alarm per tag; every other setpoint is loaded from the client's
15
+ definitions file.
16
+
17
+ State machine per alarm (ISA-18.2 states, simplified):
18
+
19
+ NORMAL --condition true for >= on_delay--> ACTIVE_UNACKED
20
+ ACTIVE_UNACKED --acknowledge--> ACTIVE_ACKED
21
+ ACTIVE_* --condition false past the deadband--> NORMAL (event CLEARED;
22
+ an unacknowledged alarm that clears is logged RTN_UNACKED)
23
+ any --shelve--> SHELVED (suppressed, logged) --unshelve--> NORMAL
24
+
25
+ Kinds: HIGH, HIGH_HIGH, LOW, LOW_LOW (value vs setpoint with deadband),
26
+ RATE (|dv/dt| vs setpoint per second), QUALITY (sample flag not GOOD).
27
+
28
+ Event log: one line per transition - timestamp, alarm, tag, kind, priority,
29
+ event (ACTIVATED / ACKNOWLEDGED / CLEARED / RTN_UNACKED / SHELVED /
30
+ UNSHELVED), value, operator, note. Append-only JSON lines when a path is
31
+ given; always kept in memory.
32
+
33
+ KPIs over a window (targets as commonly stated in ISA-18.2 practice, printed
34
+ as targets, not as this program's claims): average alarm rate per 10 min
35
+ (target <= 1, manageable <= 2), peak 10-min rate, alarm floods (> 10 alarms
36
+ in 10 min) and time in flood, standing alarms (active longer than 24 h),
37
+ chattering alarms (>= 5 activations in 10 min), priority distribution
38
+ (target about 80 / 15 / 5 for P3 / P2 / P1 when three priorities are used),
39
+ top-10 most frequent alarms.
40
+
41
+ Headless-safe: numpy only.
42
+ """
43
+
44
+ from __future__ import annotations
45
+
46
+ import json
47
+ import os
48
+ from dataclasses import dataclass, field, asdict
49
+ from datetime import datetime, timedelta, timezone
50
+ from typing import Dict, Iterable, List, Optional
51
+
52
+ import numpy as np
53
+
54
+ from .sample_record import SampleRecord, TagCatalogue, parse_utc
55
+
56
+ KINDS = ('HIGH', 'HIGH_HIGH', 'LOW', 'LOW_LOW', 'RATE', 'QUALITY')
57
+ PRIORITIES = ('P1', 'P2', 'P3', 'P4', 'P5')
58
+ EVENTS = ('ACTIVATED', 'ACKNOWLEDGED', 'CLEARED', 'RTN_UNACKED', 'SHELVED', 'UNSHELVED')
59
+
60
+
61
+ @dataclass
62
+ class AlarmDefinition:
63
+ alarm_id: str
64
+ tag_id: str
65
+ kind: str
66
+ priority: str = 'P3'
67
+ setpoint: Optional[float] = None
68
+ deadband: float = 0.0
69
+ on_delay_s: float = 0.0
70
+ description: str = ''
71
+ basis: str = 'client setting'
72
+ enabled: bool = True
73
+
74
+ def __post_init__(self):
75
+ if self.kind not in KINDS:
76
+ raise ValueError(f'unknown alarm kind {self.kind}; kinds {KINDS}')
77
+ if self.priority not in PRIORITIES:
78
+ raise ValueError(f'unknown priority {self.priority}; priorities {PRIORITIES}')
79
+ if self.kind != 'QUALITY' and self.setpoint is None:
80
+ raise ValueError(f'{self.alarm_id}: {self.kind} needs a setpoint')
81
+
82
+ def row(self) -> dict:
83
+ return asdict(self)
84
+
85
+
86
+ def load_alarm_definitions(path) -> List[AlarmDefinition]:
87
+ with open(path, encoding='utf-8') as f:
88
+ d = json.load(f)
89
+ return [AlarmDefinition(**x) for x in (d['alarms'] if isinstance(d, dict) else d)]
90
+
91
+
92
+ def write_alarm_definitions(defs: Iterable[AlarmDefinition], path) -> str:
93
+ with open(path, 'w', encoding='utf-8') as f:
94
+ json.dump({'alarms': [d.row() for d in defs]}, f, indent=1)
95
+ return str(path)
96
+
97
+
98
+ def defaults_from_catalogue(catalogue: TagCatalogue, quality_priority: str = 'P4',
99
+ range_priority: str = 'P2') -> List[AlarmDefinition]:
100
+ """Instrument over-range alarms at the tag's engineering range (basis:
101
+ the catalogue / datasheet) and one QUALITY alarm per tag. No process
102
+ setpoints - those come from the client's definitions file."""
103
+ out: List[AlarmDefinition] = []
104
+ for r in catalogue.rows():
105
+ tag = r['tag_id']
106
+ lo, hi = r['eng_range_lo'], r['eng_range_hi']
107
+ unit = r['unit']
108
+ if hi != '':
109
+ out.append(AlarmDefinition(alarm_id=f'{tag}.HH_RANGE', tag_id=tag, kind='HIGH_HIGH', priority=range_priority,
110
+ setpoint=float(hi), deadband=abs(float(hi)) * 0.005,
111
+ description=f'{tag} above instrument range {hi:g} {unit}',
112
+ basis=f"engineering range upper bound ({r['limits_basis'].split(';')[0]})"))
113
+ if lo != '':
114
+ out.append(AlarmDefinition(alarm_id=f'{tag}.LL_RANGE', tag_id=tag, kind='LOW_LOW', priority=range_priority,
115
+ setpoint=float(lo), deadband=max(abs(float(hi)) * 0.005 if hi != '' else 0.0, 0.0),
116
+ description=f'{tag} below instrument range {lo:g} {unit}',
117
+ basis=f"engineering range lower bound ({r['limits_basis'].split(';')[0]})"))
118
+ out.append(AlarmDefinition(alarm_id=f'{tag}.QUALITY', tag_id=tag, kind='QUALITY', priority=quality_priority,
119
+ description=f'{tag} sample quality not GOOD', basis='record layer quality rules'))
120
+ return out
121
+
122
+
123
+ def _iso(dt: datetime) -> str:
124
+ return dt.strftime('%Y-%m-%dT%H:%M:%SZ')
125
+
126
+
127
+ @dataclass
128
+ class _State:
129
+ state: str = 'NORMAL'
130
+ cond_since: Optional[datetime] = None
131
+ active_since: Optional[datetime] = None
132
+ activations: List[datetime] = field(default_factory=list)
133
+ last_value: Optional[float] = None
134
+ last_time: Optional[datetime] = None
135
+
136
+
137
+ class AlarmEngine:
138
+ """Feeds SampleRecords through every enabled definition on their tag."""
139
+
140
+ def __init__(self, definitions: Iterable[AlarmDefinition], event_log_path: Optional[str] = None,
141
+ operator_positions: int = 1):
142
+ self.defs: Dict[str, AlarmDefinition] = {d.alarm_id: d for d in definitions}
143
+ self.by_tag: Dict[str, List[AlarmDefinition]] = {}
144
+ for d in self.defs.values():
145
+ self.by_tag.setdefault(d.tag_id, []).append(d)
146
+ self.states: Dict[str, _State] = {a: _State() for a in self.defs}
147
+ self.events: List[dict] = []
148
+ self.log_path = event_log_path
149
+ self.operator_positions = max(1, int(operator_positions))
150
+ if event_log_path and os.path.exists(event_log_path):
151
+ with open(event_log_path, encoding='utf-8') as f:
152
+ self.events = [json.loads(l) for l in f if l.strip()]
153
+
154
+ # -- events -----------------------------------------------------------------
155
+ def _event(self, now: datetime, d: AlarmDefinition, event: str, value=None, operator: str = '', note: str = '') -> dict:
156
+ e = {'timestamp_utc': _iso(now), 'alarm_id': d.alarm_id, 'tag_id': d.tag_id, 'kind': d.kind,
157
+ 'priority': d.priority, 'event': event, 'value': (None if value is None else float(value)),
158
+ 'setpoint': d.setpoint, 'operator': operator, 'note': note}
159
+ self.events.append(e)
160
+ if self.log_path:
161
+ with open(self.log_path, 'a', encoding='utf-8') as f:
162
+ f.write(json.dumps(e, sort_keys=True) + '\n')
163
+ return e
164
+
165
+ # -- condition evaluation ------------------------------------------------------
166
+ @staticmethod
167
+ def _condition(d: AlarmDefinition, rec: SampleRecord, st: _State, now: datetime) -> Optional[bool]:
168
+ """True = in alarm, False = clearly normal (past deadband), None = inside the deadband (hold)."""
169
+ if d.kind == 'QUALITY':
170
+ return rec.quality_flag != 'GOOD'
171
+ v = rec.value
172
+ if v is None or (isinstance(v, float) and np.isnan(v)):
173
+ return None
174
+ sp, db = float(d.setpoint), float(d.deadband)
175
+ if d.kind in ('HIGH', 'HIGH_HIGH'):
176
+ return True if v >= sp else (False if v < sp - db else None)
177
+ if d.kind in ('LOW', 'LOW_LOW'):
178
+ return True if v <= sp else (False if v > sp + db else None)
179
+ if d.kind == 'RATE':
180
+ if st.last_value is None or st.last_time is None:
181
+ return False
182
+ dt = (now - st.last_time).total_seconds()
183
+ if dt <= 0:
184
+ return None
185
+ r = abs((v - st.last_value) / dt)
186
+ return True if r >= sp else (False if r < max(sp - db, 0.0) else None)
187
+ return None
188
+
189
+ def process(self, records: Iterable[SampleRecord]) -> List[dict]:
190
+ """Process records in time order per tag. Returns the events raised."""
191
+ start = len(self.events)
192
+ recs = sorted(records, key=lambda r: (r.timestamp_utc, r.tag_id))
193
+ for rec in recs:
194
+ for d in self.by_tag.get(rec.tag_id, []):
195
+ if not d.enabled:
196
+ continue
197
+ st = self.states[d.alarm_id]
198
+ now = parse_utc(rec.timestamp_utc)
199
+ if st.state == 'SHELVED':
200
+ st.last_value, st.last_time = rec.value, now
201
+ continue
202
+ cond = self._condition(d, rec, st, now)
203
+ if cond is True:
204
+ if st.cond_since is None:
205
+ st.cond_since = now
206
+ if st.state == 'NORMAL' and (now - st.cond_since).total_seconds() >= d.on_delay_s:
207
+ st.state = 'ACTIVE_UNACKED'
208
+ st.active_since = now
209
+ st.activations.append(now)
210
+ self._event(now, d, 'ACTIVATED', rec.value, note=(rec.rule_fired if d.kind == 'QUALITY' else ''))
211
+ elif cond is False:
212
+ st.cond_since = None
213
+ if st.state in ('ACTIVE_UNACKED', 'ACTIVE_ACKED'):
214
+ self._event(now, d, 'CLEARED' if st.state == 'ACTIVE_ACKED' else 'RTN_UNACKED', rec.value)
215
+ st.state = 'NORMAL'
216
+ st.active_since = None
217
+ # cond None: inside the deadband - hold the current state
218
+ if rec.value is not None and not (isinstance(rec.value, float) and np.isnan(rec.value)):
219
+ st.last_value, st.last_time = rec.value, now
220
+ return self.events[start:]
221
+
222
+ # -- operator actions -------------------------------------------------------------
223
+ def acknowledge(self, alarm_id: str, operator: str, now: datetime, note: str = '') -> dict:
224
+ st, d = self.states[alarm_id], self.defs[alarm_id]
225
+ if st.state != 'ACTIVE_UNACKED':
226
+ raise ValueError(f'{alarm_id} is {st.state}, nothing to acknowledge')
227
+ st.state = 'ACTIVE_ACKED'
228
+ return self._event(now, d, 'ACKNOWLEDGED', st.last_value, operator, note)
229
+
230
+ def shelve(self, alarm_id: str, operator: str, now: datetime, note: str = '') -> dict:
231
+ st, d = self.states[alarm_id], self.defs[alarm_id]
232
+ st.state = 'SHELVED'
233
+ st.cond_since = None
234
+ return self._event(now, d, 'SHELVED', st.last_value, operator, note)
235
+
236
+ def unshelve(self, alarm_id: str, operator: str, now: datetime) -> dict:
237
+ st, d = self.states[alarm_id], self.defs[alarm_id]
238
+ st.state = 'NORMAL'
239
+ return self._event(now, d, 'UNSHELVED', st.last_value, operator)
240
+
241
+ def active(self) -> List[dict]:
242
+ return [{'alarm_id': a, 'state': s.state, 'priority': self.defs[a].priority, 'tag_id': self.defs[a].tag_id,
243
+ 'active_since_utc': _iso(s.active_since) if s.active_since else None, 'last_value': s.last_value}
244
+ for a, s in self.states.items() if s.state.startswith('ACTIVE')]
245
+
246
+ # -- KPIs -------------------------------------------------------------------------
247
+ def kpis(self, window_start: Optional[datetime] = None, window_end: Optional[datetime] = None,
248
+ standing_hours: float = 24.0) -> dict:
249
+ acts = [e for e in self.events if e['event'] == 'ACTIVATED']
250
+ if not acts:
251
+ return {'n_activations': 0, 'window': None, 'note': 'no activations in the window'}
252
+ times = [parse_utc(e['timestamp_utc']) for e in acts]
253
+ ws = window_start or min(times)
254
+ we = window_end or max(times)
255
+ span_min = max((we - ws).total_seconds() / 60.0, 10.0)
256
+ acts_w = [(t, e) for t, e in zip(times, acts) if ws <= t <= we]
257
+ n = len(acts_w)
258
+ # 10-minute bins
259
+ nb = int(np.ceil(span_min / 10.0))
260
+ bins = np.zeros(nb, dtype=int)
261
+ for t, _ in acts_w:
262
+ bins[min(int((t - ws).total_seconds() // 600), nb - 1)] += 1
263
+ per_pos = bins / self.operator_positions
264
+ flood_bins = int(np.sum(per_pos > 10))
265
+ # chattering: >= 5 activations of one alarm within any 10-minute span
266
+ chatter = []
267
+ for a, s in self.states.items():
268
+ ts = sorted(s.activations)
269
+ for i in range(len(ts)):
270
+ if i + 4 < len(ts) and (ts[i + 4] - ts[i]).total_seconds() <= 600:
271
+ chatter.append(a)
272
+ break
273
+ # standing alarms
274
+ standing = [a for a, s in self.states.items() if s.active_since and (we - s.active_since).total_seconds() > standing_hours * 3600]
275
+ # priority distribution
276
+ pr = {p: 0 for p in PRIORITIES}
277
+ for _, e in acts_w:
278
+ pr[e['priority']] += 1
279
+ counts: Dict[str, int] = {}
280
+ for _, e in acts_w:
281
+ counts[e['alarm_id']] = counts.get(e['alarm_id'], 0) + 1
282
+ top = sorted(counts.items(), key=lambda kv: -kv[1])[:10]
283
+ unacked = [a for a, s in self.states.items() if s.state == 'ACTIVE_UNACKED']
284
+ return {
285
+ 'window': [_iso(ws), _iso(we)], 'window_hours': round(span_min / 60.0, 2),
286
+ 'operator_positions': self.operator_positions,
287
+ 'n_activations': n, 'per_day': round(n / (span_min / 1440.0), 1),
288
+ 'avg_per_10min_per_position': round(float(per_pos.mean()), 3),
289
+ 'peak_per_10min_per_position': round(float(per_pos.max()), 2),
290
+ 'flood_10min_bins': flood_bins, 'pct_time_in_flood': round(100.0 * flood_bins / nb, 2),
291
+ 'standing_alarms_over_24h': standing, 'chattering_alarms': chatter,
292
+ 'active_unacknowledged': unacked,
293
+ 'priority_distribution': pr,
294
+ 'priority_distribution_pct': {p: round(100.0 * c / n, 1) for p, c in pr.items()} if n else pr,
295
+ 'top_alarms': [{'alarm_id': a, 'activations': c, 'pct_of_total': round(100.0 * c / n, 1)} for a, c in top],
296
+ 'targets': {'avg_per_10min': '<= 1 acceptable, <= 2 manageable', 'per_day': '<= 150 acceptable, <= 300 manageable',
297
+ 'flood': '> 10 alarms in 10 min per operator position', 'chattering': '>= 5 activations in 10 min',
298
+ 'standing': f'active > {standing_hours:g} h', 'priority_split': 'about 80 / 15 / 5 (P3 / P2 / P1) for a three-priority scheme',
299
+ 'basis': 'targets as commonly stated in ISA-18.2 / IEC 62682 alarm-management practice; printed as targets, not as results'},
300
+ }
301
+
302
+
303
+ def records_from_series(tag_id: str, times: List[str], values: List[float], unit: str = '',
304
+ flags: Optional[List[str]] = None) -> List[SampleRecord]:
305
+ """Convenience for tests and for feeding a single series."""
306
+ return [SampleRecord(tag_id=tag_id, timestamp_utc=t, value=v, unit=unit,
307
+ quality_flag=(flags[i] if flags else 'GOOD')) for i, (t, v) in enumerate(zip(times, values))]
gea/bench.py ADDED
@@ -0,0 +1,144 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """Bench-test analysis (v1.48.0) - field-tier step 8.
5
+
6
+ The analysis half of BENCH_TEST_PROTOCOL.md: fit each leg's drift slope,
7
+ propagate uncertainties into the conventional/GEA ratio, and return a
8
+ verdict from the protocol's vocabulary (MEASURED_CONFIRMS /
9
+ MEASURED_REFUTES / INSUFFICIENT_SPAN / INSUFFICIENT_SNR).
10
+
11
+ Honesty rules:
12
+ - The 18-day span floor is READ FROM the reconciler's own configuration -
13
+ the bench cannot be rushed past the product's standing rule.
14
+ - The self-test is labeled SIMULATION_SELF_TEST in its own output: it
15
+ verifies the analysis arithmetic against the engine's twin models, and
16
+ proves NOTHING about physical gauges.
17
+ - A refutation is a first-class outcome, not an error.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ from typing import Optional, Tuple
22
+
23
+ import numpy as np
24
+
25
+ from .quartz_hpht_extension import (calculate_quartz_transducer_hpht_program,
26
+ canonical_suppression,
27
+ conventional_drift)
28
+ from .reconciler import ReconcilerConfig
29
+
30
+ YEAR_S = 365.25 * 86400.0
31
+
32
+
33
+ def _leg_slope(times_s: np.ndarray, p_psi: np.ndarray,
34
+ full_scale_psi: float) -> dict:
35
+ """OLS drift slope of one leg at constant setpoint: %FS/yr with the
36
+ standard error of the slope propagated from the fit residuals."""
37
+ t = np.asarray(times_s, dtype=float) / YEAR_S
38
+ p = np.asarray(p_psi, dtype=float)
39
+ m = ~(np.isnan(t) | np.isnan(p))
40
+ t, p = t[m], p[m]
41
+ n = len(t)
42
+ if n < 8:
43
+ raise ValueError(f"leg needs >= 8 finite samples (got {n})")
44
+ span_years = float(t.max() - t.min())
45
+ A = np.vstack([t - t.mean(), np.ones(n)]).T
46
+ coef, res, _, _ = np.linalg.lstsq(A, p, rcond=None)
47
+ slope_psi_yr = float(coef[0])
48
+ dof = max(n - 2, 1)
49
+ sigma2 = float(res[0]) / dof if len(res) else float(np.var(p - A @ coef))
50
+ se_slope = float(np.sqrt(sigma2 / np.sum((t - t.mean()) ** 2)))
51
+ return {"n": n, "span_years": round(span_years, 4),
52
+ "slope_psi_yr": round(slope_psi_yr, 3),
53
+ "se_slope_psi_yr": round(se_slope, 3),
54
+ "drift_pct_fs_yr": round(slope_psi_yr / full_scale_psi * 100.0, 5),
55
+ "se_drift_pct_fs_yr": round(se_slope / full_scale_psi * 100.0, 5)}
56
+
57
+
58
+ def bench_analysis(times_s, p_psi, conv_times_s, conv_p_psi,
59
+ full_scale_psi: float = 30000.0,
60
+ k_sigma: float = 2.0) -> dict:
61
+ """The protocol section 5 analysis. Inputs are the two legs' raw
62
+ pressure series at a constant setpoint; output carries the measured
63
+ ratio, its propagated uncertainty, the prediction, and the verdict."""
64
+ cfg = ReconcilerConfig()
65
+ uq = _leg_slope(times_s, p_psi, full_scale_psi)
66
+ cv = _leg_slope(conv_times_s, conv_p_psi, full_scale_psi)
67
+ prediction = canonical_suppression()
68
+ out = {"protocol": "BENCH_TEST_PROTOCOL.md section 5",
69
+ "prediction_ratio": round(prediction, 4),
70
+ "prediction_status": ("DERIVED_HYBRID composition - the quantity "
71
+ "this bench exists to confirm or refute"),
72
+ "leg": uq, "conventional_leg": cv,
73
+ "full_scale_psi": full_scale_psi}
74
+ min_span = float(cfg.min_trend_span_years)
75
+ if uq["span_years"] < min_span or cv["span_years"] < min_span:
76
+ out["verdict"] = "INSUFFICIENT_SPAN"
77
+ out["detail"] = (f"span floor {min_span:g} yr "
78
+ f"(the reconciler's own >=18-day slope rule) not met "
79
+ f"- no verdict; the bench cannot be rushed")
80
+ return out
81
+ if uq["slope_psi_yr"] <= 0:
82
+ out["verdict"] = "INSUFFICIENT_SNR"
83
+ out["detail"] = ("program-model-leg slope is non-positive - a drift ratio has "
84
+ "no meaning here; check setpoint stability")
85
+ return out
86
+ r = cv["slope_psi_yr"] / uq["slope_psi_yr"]
87
+ se_r = abs(r) * float(np.sqrt(
88
+ (cv["se_slope_psi_yr"] / cv["slope_psi_yr"]) ** 2
89
+ + (uq["se_slope_psi_yr"] / uq["slope_psi_yr"]) ** 2))
90
+ out["measured_ratio"] = round(r, 4)
91
+ out["se_ratio"] = round(se_r, 4)
92
+ lo, hi = r - k_sigma * se_r, r + k_sigma * se_r
93
+ contains_pred = lo <= prediction <= hi
94
+ contains_unity = lo <= 1.0 <= hi
95
+ if contains_pred and contains_unity:
96
+ out["verdict"] = "INSUFFICIENT_SNR"
97
+ out["detail"] = (f"the {k_sigma:.0f}-sigma band [{lo:.4f}, {hi:.4f}] "
98
+ "contains BOTH the prediction and 1.0 - suppression "
99
+ "cannot be distinguished from no-suppression; more "
100
+ "data required, no verdict")
101
+ elif contains_pred:
102
+ out["verdict"] = "MEASURED_CONFIRMS"
103
+ out["detail"] = (f"measured R = {r:.4f} +/- {se_r:.4f} contains the "
104
+ f"predicted {prediction:.4f} and excludes 1.0 - on "
105
+ "REAL bench data this outcome would support "
106
+ "relabeling per protocol section 6")
107
+ else:
108
+ out["verdict"] = "MEASURED_REFUTES"
109
+ out["detail"] = (f"measured R = {r:.4f} +/- {se_r:.4f} excludes the "
110
+ f"predicted {prediction:.4f} - a first-class "
111
+ "outcome: the composition is falsified at these "
112
+ "conditions; the label stays DERIVED_HYBRID with "
113
+ "this refutation on record (protocol section 6)")
114
+ return out
115
+
116
+
117
+ def bench_selftest(days: int = 120, setpoint_psi: float = 10000.0,
118
+ setpoint_temp_C: float = 150.0,
119
+ noise_psi: float = 0.5, seed: int = 8,
120
+ conv_scale: float = 1.0) -> dict:
121
+ """SIMULATION_SELF_TEST: synthesize both legs from the engine's own
122
+ drift models at the protocol's setpoint class and run the analysis.
123
+ Verifies the ARITHMETIC, proves nothing about gauges - and says so in
124
+ its own output. conv_scale != 1 exercises the refutation path."""
125
+ r_uq = float(calculate_quartz_transducer_hpht_program(
126
+ depth_m=3000.0, temp_c=setpoint_temp_C,
127
+ pressure_psi=setpoint_psi)["value"]["drift_pct"])
128
+ r_cv = conventional_drift(setpoint_temp_C, setpoint_psi) * conv_scale
129
+ fs = 30000.0
130
+ rng = np.random.default_rng(seed)
131
+ t = np.arange(days) * 86400.0
132
+ ty = t / YEAR_S
133
+ p_uq = setpoint_psi + r_uq / 100.0 * fs * ty + rng.normal(0, noise_psi, days)
134
+ p_cv = setpoint_psi + r_cv / 100.0 * fs * ty + rng.normal(0, noise_psi, days)
135
+ out = bench_analysis(t, p_uq, t, p_cv, full_scale_psi=fs)
136
+ out["mode"] = ("SIMULATION_SELF_TEST: both legs synthesized from the "
137
+ "engine's own models - this verifies the analysis "
138
+ "arithmetic and the bench pipeline, NOT the physics; "
139
+ "no gauge was measured")
140
+ out["synth_inputs"] = {"days": days, "setpoint_psi": setpoint_psi,
141
+ "setpoint_temp_C": setpoint_temp_C,
142
+ "noise_psi": noise_psi, "seed": seed,
143
+ "conv_scale": conv_scale}
144
+ return out