gea-program 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. gea/BENCH_TEST_PROTOCOL.md +97 -0
  2. gea/IMPORT_RECORD.md +61 -0
  3. gea/__init__.py +160 -0
  4. gea/__main__.py +661 -0
  5. gea/acceptance_tests.py +1315 -0
  6. gea/accuracy_statement.py +181 -0
  7. gea/alarm_engine.py +307 -0
  8. gea/bench.py +144 -0
  9. gea/blind_harness.py +108 -0
  10. gea/case_study.py +229 -0
  11. gea/catalog/acex_lomonosov_age_depth_model.provenance.json +23 -0
  12. gea/catalog/acex_lomonosov_age_depth_model.txt +28 -0
  13. gea/catalog/agassiz77_canada_temperature.csv +68 -0
  14. gea/catalog/agassiz77_canada_temperature.provenance.json +17 -0
  15. gea/catalog/barbados_110_consolidation.provenance.json +24 -0
  16. gea/catalog/barbados_110_consolidation.txt +91 -0
  17. gea/catalog/bengal_u1452_grain_size.provenance.json +20 -0
  18. gea/catalog/bengal_u1452_grain_size.txt +252 -0
  19. gea/catalog/blake_164_methane_isotopes.provenance.json +22 -0
  20. gea/catalog/blake_164_methane_isotopes.txt +68 -0
  21. gea/catalog/chicxulub_m0077a_pwave_velocity.provenance.json +23 -0
  22. gea/catalog/chicxulub_m0077a_pwave_velocity.txt +735 -0
  23. gea/catalog/collingwood_1_28_ks_complete.las +128 -0
  24. gea/catalog/collingwood_1_28_ks_complete.provenance.json +17 -0
  25. gea/catalog/costa_rica_odp_friction_envelope.provenance.json +24 -0
  26. gea/catalog/costa_rica_odp_friction_envelope.txt +57 -0
  27. gea/catalog/dead_sea_5017_debrite_xrf_ms.provenance.json +21 -0
  28. gea/catalog/dead_sea_5017_debrite_xrf_ms.txt +94 -0
  29. gea/catalog/dsdp_504b_physical_properties.provenance.json +24 -0
  30. gea/catalog/dsdp_504b_physical_properties.txt +82 -0
  31. gea/catalog/dsdp_504b_sound_velocity.provenance.json +21 -0
  32. gea/catalog/dsdp_504b_sound_velocity.txt +81 -0
  33. gea/catalog/elgygytgyn_5011_turbidites.provenance.json +20 -0
  34. gea/catalog/elgygytgyn_5011_turbidites.txt +193 -0
  35. gea/catalog/epica_domec_co2_800kyr.provenance.json +21 -0
  36. gea/catalog/epica_domec_co2_800kyr.txt +265 -0
  37. gea/catalog/fram_909_organic_petrography.provenance.json +22 -0
  38. gea/catalog/fram_909_organic_petrography.txt +40 -0
  39. gea/catalog/gbr_325_coral_uth_ages.provenance.json +23 -0
  40. gea/catalog/gbr_325_coral_uth_ages.txt +71 -0
  41. gea/catalog/gisp2_greenland_temperature.csv +599 -0
  42. gea/catalog/gisp2_greenland_temperature.provenance.json +17 -0
  43. gea/catalog/gom_308_t2p_insitu.provenance.json +21 -0
  44. gea/catalog/gom_308_t2p_insitu.txt +40 -0
  45. gea/catalog/guaymas_385_dom_d13c.provenance.json +21 -0
  46. gea/catalog/guaymas_385_dom_d13c.txt +103 -0
  47. gea/catalog/hikurangi_u1520_friction_insitu.provenance.json +26 -0
  48. gea/catalog/hikurangi_u1520_friction_insitu.txt +74 -0
  49. gea/catalog/hydrate_ridge_204_ncr_hydrate.provenance.json +22 -0
  50. gea/catalog/hydrate_ridge_204_ncr_hydrate.txt +60 -0
  51. gea/catalog/iodp_u1324_pore_pressure.provenance.json +22 -0
  52. gea/catalog/iodp_u1324_pore_pressure.txt +42 -0
  53. gea/catalog/jfast_c0019_slow_slip_events.provenance.json +28 -0
  54. gea/catalog/jfast_c0019_slow_slip_events.txt +35 -0
  55. gea/catalog/kennetcook_2_p129_excerpt.las +139 -0
  56. gea/catalog/kennetcook_2_p129_excerpt.provenance.json +17 -0
  57. gea/catalog/ktb_hb_bhgm_density.dat +227 -0
  58. gea/catalog/ktb_hb_bhgm_density.provenance.json +22 -0
  59. gea/catalog/ktb_hb_complog_6020_excerpt.provenance.json +20 -0
  60. gea/catalog/ktb_hb_complog_6020_excerpt.txt +72 -0
  61. gea/catalog/ktb_hb_hlog246_temperature.dat +1636 -0
  62. gea/catalog/ktb_hb_hlog246_temperature.provenance.json +22 -0
  63. gea/catalog/ktb_hb_rockmech_compress.dat +33 -0
  64. gea/catalog/ktb_hb_rockmech_compress.provenance.json +22 -0
  65. gea/catalog/ktb_hb_tvd_0_9080_excerpt.dat +2817 -0
  66. gea/catalog/ktb_hb_tvd_0_9080_excerpt.provenance.json +20 -0
  67. gea/catalog/ktb_vb_rockmech_compress.dat +125 -0
  68. gea/catalog/ktb_vb_rockmech_compress.provenance.json +22 -0
  69. gea/catalog/ktb_vb_vlog251_temperature.dat +1127 -0
  70. gea/catalog/ktb_vb_vlog251_temperature.provenance.json +24 -0
  71. gea/catalog/l06_06_nl_survey.csv +201 -0
  72. gea/catalog/l06_06_nl_survey.provenance.json +17 -0
  73. gea/catalog/l07_01_nl_excerpt.las +90 -0
  74. gea/catalog/l07_01_nl_excerpt.provenance.json +17 -0
  75. gea/catalog/mariana_1200_serpentinite_geochem.provenance.json +21 -0
  76. gea/catalog/mariana_1200_serpentinite_geochem.txt +63 -0
  77. gea/catalog/med_160_sapropels.provenance.json +22 -0
  78. gea/catalog/med_160_sapropels.txt +43 -0
  79. gea/catalog/nankai_megasplay_shear_strength.provenance.json +26 -0
  80. gea/catalog/nankai_megasplay_shear_strength.txt +42 -0
  81. gea/catalog/odp_1027b_thermal_conductivity.provenance.json +21 -0
  82. gea/catalog/odp_1027b_thermal_conductivity.txt +52 -0
  83. gea/catalog/odp_1027c_cork_temperature.provenance.json +20 -0
  84. gea/catalog/odp_1027c_cork_temperature.txt +26 -0
  85. gea/catalog/odp_1165b_thermal_conductivity.provenance.json +22 -0
  86. gea/catalog/odp_1165b_thermal_conductivity.txt +102 -0
  87. gea/catalog/odp_1274a_mantle_peridotite_mad.provenance.json +27 -0
  88. gea/catalog/odp_1274a_mantle_peridotite_mad.txt +44 -0
  89. gea/catalog/odp_504b_dike_elastic_moduli.provenance.json +27 -0
  90. gea/catalog/odp_504b_dike_elastic_moduli.txt +85 -0
  91. gea/catalog/odp_504b_leg137_borehole_fluids.provenance.json +19 -0
  92. gea/catalog/odp_504b_leg137_borehole_fluids.txt +68 -0
  93. gea/catalog/odp_735b_gabbro_elastic_moduli.provenance.json +28 -0
  94. gea/catalog/odp_735b_gabbro_elastic_moduli.txt +127 -0
  95. gea/catalog/peru_201_sulfate_reduction.provenance.json +22 -0
  96. gea/catalog/peru_201_sulfate_reduction.txt +322 -0
  97. gea/catalog/scorpio_e1_sa_excerpt.las +113 -0
  98. gea/catalog/scorpio_e1_sa_excerpt.provenance.json +17 -0
  99. gea/catalog/sumatra_362_cohesion.provenance.json +21 -0
  100. gea/catalog/sumatra_362_cohesion.txt +38 -0
  101. gea/catalog/university_6_17_no1_tx_excerpt.las +119 -0
  102. gea/catalog/university_6_17_no1_tx_excerpt.provenance.json +17 -0
  103. gea/catalog/ursa_308_xrd_mineralogy.provenance.json +21 -0
  104. gea/catalog/ursa_308_xrd_mineralogy.txt +46 -0
  105. gea/catalog/volve_15_9_19_sr_excerpt.las +183 -0
  106. gea/catalog/volve_15_9_19_sr_excerpt.provenance.json +16 -0
  107. gea/catalog/volve_15_9_19a_core_excerpt.csv +88 -0
  108. gea/catalog/volve_15_9_19a_core_excerpt.provenance.json +18 -0
  109. gea/catalog/volve_f12_f14_production_excerpt.csv +167 -0
  110. gea/catalog/volve_f12_f14_production_excerpt.provenance.json +21 -0
  111. gea/catalog/walvis_208_petm_carbonate.provenance.json +21 -0
  112. gea/catalog/walvis_208_petm_carbonate.txt +268 -0
  113. gea/catalog/woodlark_1109_rock_eval.provenance.json +23 -0
  114. gea/catalog/woodlark_1109_rock_eval.txt +30 -0
  115. gea/cli.py +125 -0
  116. gea/client_reports.py +943 -0
  117. gea/config_versioning.py +133 -0
  118. gea/correlation.py +155 -0
  119. gea/dashboard.py +390 -0
  120. gea/deviation.py +70 -0
  121. gea/downhole_engine.py +395 -0
  122. gea/drift_monitor.py +310 -0
  123. gea/earth_model.py +230 -0
  124. gea/example_register_map.json +14 -0
  125. gea/fat_sat.py +68 -0
  126. gea/follower.py +98 -0
  127. gea/forward_model.py +139 -0
  128. gea/gamma.py +176 -0
  129. gea/gauge_specs.py +112 -0
  130. gea/gravity_reference.py +116 -0
  131. gea/inverse_engine.py +215 -0
  132. gea/matplotlib_demo.py +85 -0
  133. gea/modbus.py +229 -0
  134. gea/model_card.py +248 -0
  135. gea/operator_app.py +442 -0
  136. gea/ports.py +340 -0
  137. gea/profile_catalog.py +773 -0
  138. gea/project.py +213 -0
  139. gea/qt6_downhole_app.py +144 -0
  140. gea/quartz_hpht_extension.py +152 -0
  141. gea/reconciler.py +206 -0
  142. gea/rock_inventory.py +404 -0
  143. gea/sample_record.py +430 -0
  144. gea/sample_well_profile.csv +15 -0
  145. gea/sbom.py +116 -0
  146. gea/segy.py +181 -0
  147. gea/service_life.py +173 -0
  148. gea/shell.py +107 -0
  149. gea/sla_report.py +199 -0
  150. gea/store_forward.py +234 -0
  151. gea/strata_join.py +186 -0
  152. gea/survey_cmd.py +264 -0
  153. gea/survey_view.py +138 -0
  154. gea/telemetry.py +306 -0
  155. gea/tool_library.py +260 -0
  156. gea/well_assembler.py +457 -0
  157. gea/well_test_validation.py +369 -0
  158. gea_program-0.1.0.dist-info/METADATA +138 -0
  159. gea_program-0.1.0.dist-info/RECORD +163 -0
  160. gea_program-0.1.0.dist-info/WHEEL +5 -0
  161. gea_program-0.1.0.dist-info/entry_points.txt +2 -0
  162. gea_program-0.1.0.dist-info/licenses/LICENSE +373 -0
  163. gea_program-0.1.0.dist-info/top_level.txt +1 -0
gea/client_reports.py ADDED
@@ -0,0 +1,943 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """client_reports — the client-facing report family, in the outline and
5
+ vocabulary of a production-operations scope of work.
6
+
7
+ The engines in this package return dicts. This module turns them into the
8
+ reports a client engineer expects to open, numbered the way a scope of work
9
+ numbers them, and says nothing in the program's internal register. Every
10
+ number is recomputed at generation time from the same engines; every
11
+ threshold in force is printed; every flag names the rule that fired.
12
+
13
+ Reports (this module grows one report at a time; each is a function that
14
+ returns a `Document`):
15
+
16
+ gauge_drift_report(...) Gauge Drift & Reconciliation Report
17
+ (SOW 4.2.10 drift / re-fit; SLA 1.0 Model
18
+ Drift & Model Staleness; 4.2.2 data quality;
19
+ 4.2.1.2 tag catalogue; 4.2.1.4 configuration)
20
+
21
+ Rendering: `render_markdown(doc)`, `render_html(doc)`, `write(doc, out_dir)`.
22
+
23
+ Vocabulary gate: `forbidden_terms(text)` returns the internal-register words
24
+ found in a rendered report; the acceptance suite requires an empty list.
25
+
26
+ Headless-safe: numpy only.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import html as _html
32
+ import json
33
+ import os
34
+ import time
35
+ from dataclasses import dataclass, field
36
+ from datetime import datetime, timedelta, timezone
37
+ from typing import Dict, List, Optional, Sequence
38
+
39
+ from .sample_record import (TagCatalogue, records_from_stream, quality_summary,
40
+ write_records_csv, write_catalogue_csv, parse_utc)
41
+ from .accuracy_statement import Z90
42
+
43
+ PROGRAM_NAME = 'Downhole Gauge Monitoring'
44
+
45
+ # SLA clocks (business days) for a detected model drift / staleness event.
46
+ SLA_EVALUATE_EVERY_H = 24.0
47
+ SLA_NOTIFY_BD = 1
48
+ SLA_FALLBACK_BD = 2
49
+ SLA_REFIT_BD = 10
50
+
51
+ # The internal register never reaches a client report.
52
+ FORBIDDEN_TERMS = ('uqff', 'star-magic', 'star magic', 'primitive', 'doctrine',
53
+ 'pinned_awaiting', 'refus', 'aether', 'dpm', 'rule 7', 'honest',
54
+ 'landmark', 'lattice')
55
+
56
+ CLASS_KEY: Dict[str, dict] = {
57
+ 'IN_FAMILY': dict(
58
+ meaning='Residual within the gauge noise band; live data and model agree.',
59
+ status='NO DRIFT', action='ACCEPT - no action.'),
60
+ 'CALIBRATION_OFFSET': dict(
61
+ meaning='Constant bias above the noise band and within instrument scale.',
62
+ status='DRIFT DETECTED', action='RE-FIT - apply the offset correction and record it in the change log.'),
63
+ 'DRIFT_CONSISTENT': dict(
64
+ meaning='Trend inside the instrument aging envelope at station conditions.',
65
+ status='AGING WITHIN ENVELOPE', action='MONITOR - schedule recalibration per the maintenance plan.'),
66
+ 'TRANSIENTS': dict(
67
+ meaning='Clustered short excursions consistent with well events, not an instrument fault.',
68
+ status='NO DRIFT', action='REVIEW - confirm the excursions against the operations log.'),
69
+ 'UNEXPLAINED_OFFSET': dict(
70
+ meaning='Bias larger than a calibration correction can account for.',
71
+ status='DRIFT DETECTED', action='FALLBACK - hold the model output for this station; investigate gauge and completion; SLA clocks start.'),
72
+ 'UNEXPLAINED_TREND': dict(
73
+ meaning='Trend outside the instrument aging envelope; a process change (for example drawdown) or an instrument fault.',
74
+ status='DRIFT DETECTED', action='FALLBACK - hold the model output for this station; investigate process change versus instrument; SLA clocks start.'),
75
+ 'INSUFFICIENT_DATA': dict(
76
+ meaning='Fewer than 8 valid samples in the window.',
77
+ status='PENDING', action='PENDING - evaluate again when the window fills.'),
78
+ }
79
+
80
+
81
+ # ---------------------------------------------------------------------------
82
+ # Document model
83
+ # ---------------------------------------------------------------------------
84
+ @dataclass
85
+ class Table:
86
+ columns: List[str]
87
+ rows: List[List[object]]
88
+ caption: str = ''
89
+
90
+
91
+ @dataclass
92
+ class Section:
93
+ number: str
94
+ title: str
95
+ paragraphs: List[str] = field(default_factory=list)
96
+ tables: List[Table] = field(default_factory=list)
97
+
98
+
99
+ @dataclass
100
+ class Document:
101
+ title: str
102
+ report_id: str
103
+ front: List[List[str]] # key/value rows
104
+ sections: List[Section]
105
+ footer: str = ''
106
+ data: dict = field(default_factory=dict) # machine copy of what was rendered
107
+
108
+
109
+ def _fmt(v) -> str:
110
+ if v is None:
111
+ return '-'
112
+ if isinstance(v, float):
113
+ if v != v: # NaN
114
+ return '-'
115
+ if v == int(v) and abs(v) < 1e9:
116
+ return f'{int(v):,}'
117
+ return f'{v:,.2f}' if abs(v) < 1e6 else f'{v:.3e}'
118
+ return str(v)
119
+
120
+
121
+ def _business_days_after(start: datetime, n: int) -> datetime:
122
+ d = start
123
+ added = 0
124
+ while added < n:
125
+ d += timedelta(days=1)
126
+ if d.weekday() < 5:
127
+ added += 1
128
+ return d
129
+
130
+
131
+ def _utc_now() -> datetime:
132
+ return datetime.now(timezone.utc).replace(microsecond=0)
133
+
134
+
135
+ # ---------------------------------------------------------------------------
136
+ # Report 1 - Gauge Drift & Reconciliation
137
+ # ---------------------------------------------------------------------------
138
+ def gauge_drift_report(evaluation: dict, stream, catalogue: Optional[TagCatalogue] = None,
139
+ well_name: str = '', evaluated_at: Optional[datetime] = None,
140
+ program_version: str = '', gauge_spec=None,
141
+ roc_limits: Optional[Dict[str, float]] = None,
142
+ monitor_status: Optional[dict] = None) -> Document:
143
+ """Build the report from a reconciler evaluation dict and the live stream
144
+ it was evaluated on. Nothing is recomputed here except summaries; the
145
+ engine's numbers are the record. With `gauge_spec` the tag limits come
146
+ from the datasheet and the report cites it."""
147
+ now = evaluated_at or _utc_now()
148
+ if catalogue is None:
149
+ catalogue = TagCatalogue.from_stream(stream, gauge_spec=gauge_spec, roc_limits=roc_limits)
150
+ records = records_from_stream(stream, catalogue)
151
+ dq = quality_summary(records)
152
+ stations = evaluation.get('stations', [])
153
+ counts = evaluation.get('classification_counts', {})
154
+ n_st = len(stations)
155
+ drift_stations = [s for s in stations if CLASS_KEY.get(s['classification'], {}).get('status') == 'DRIFT DETECTED']
156
+ drift_detected = bool(drift_stations)
157
+ window_days = max((s.get('span_years', 0.0) or 0.0) for s in stations) * 365.25 if stations else 0.0
158
+ first = min(v['first_utc'] for v in dq.values()) if dq else '-'
159
+ last = max(v['last_utc'] for v in dq.values()) if dq else '-'
160
+ well = well_name or evaluation.get('stream', 'well')
161
+ report_id = f"GDR-{well.replace('/', '-').replace(' ', '_')[:24]}-{now.strftime('%Y%m%dT%H%M%SZ')}"
162
+ meta = dict(getattr(stream, 'meta', {}) or {})
163
+
164
+ # 1. Summary ---------------------------------------------------------------
165
+ parts = [f'{n_st} gauge station{"s" if n_st != 1 else ""} evaluated against the well model over a '
166
+ f'{window_days:.0f}-day window ({first} to {last}).']
167
+ if counts:
168
+ parts.append('Classification: ' + ', '.join(f'{k} x{v}' for k, v in sorted(counts.items())) + '.')
169
+ if drift_detected:
170
+ parts.append(f'MODEL DRIFT DETECTED at {len(drift_stations)} station'
171
+ f'{"s" if len(drift_stations) != 1 else ""}: '
172
+ + '; '.join(f"{s['channel']} ({s['classification']}, bias {_fmt(s.get('bias_psi'))} psi"
173
+ + (f", trend {_fmt(s.get('slope_psi_yr'))} psi/yr" if s.get('slope_psi_yr') is not None else '')
174
+ + ')' for s in drift_stations)
175
+ + '. SLA clocks for notification, fallback and re-fit start at the evaluation timestamp (section 5).')
176
+ else:
177
+ parts.append('No model drift detected; no SLA clock started.')
178
+ summary = Section('1', 'Summary', [' '.join(parts)])
179
+
180
+ # 2. Tag catalogue (SOW 4.2.1.2) --------------------------------------------
181
+ cat_rows = [[r['tag_id'], r['description'][:70], r['unit'], r['owner'], r['tag_class'],
182
+ f"{_fmt(r['eng_range_lo'])} to {_fmt(r['eng_range_hi'])}",
183
+ r['cadence_s'] or '-', r['source_layer'], r['source'][:50]] for r in catalogue.rows()]
184
+ tag_sec = Section('2', 'Tag catalogue and canonical data model (SOW 4.2.1.2)',
185
+ ['One definition per tag: owner, engineering unit, engineering range, cadence and '
186
+ 'source layer. Quality rules in force per tag are listed in section 6.'],
187
+ [Table(['Tag', 'Description', 'Unit', 'Owner', 'Class', 'Engineering range',
188
+ 'Cadence (s)', 'Source layer', 'Source'], cat_rows)])
189
+
190
+ # 3. Data quality (SOW 4.2.2) -----------------------------------------------
191
+ dq_rows = []
192
+ for tag, v in dq.items():
193
+ c = v['counts']
194
+ dq_rows.append([tag, v['n'], f"{v['pct_good']:.1f}", c['RANGE'], c['ROC'], c['FLATLINE'],
195
+ c['SPIKE'], c['STALE'], c['GAP'],
196
+ f"{v['longest_gap_samples']} ({v['longest_gap_s']/3600:.1f} h)" if v['longest_gap_samples'] else '0',
197
+ _fmt(v['latency_p95_s']) if v['latency_p95_s'] is not None else 'not stamped'])
198
+ dq_par = ['Per-sample quality flags after the range, rate-of-change, flatline, spike and staleness '
199
+ 'rules. Every flagged sample carries the rule and limit that fired (records CSV).']
200
+ ex = {tag: v['rule_examples'] for tag, v in dq.items() if v['rule_examples']}
201
+ if ex:
202
+ dq_par.append('Rules that fired: ' + '; '.join(
203
+ f"{tag}: " + ', '.join(f'{f} ({r})' for f, r in d.items()) for tag, d in ex.items()) + '.')
204
+ if meta.get('nan_days_dropped') not in (None, '0'):
205
+ dq_par.append(f"Source records without a value in the evaluated channel: {meta['nan_days_dropped']} "
206
+ f"(excluded from the evaluation window by the ingest adapter and counted here).")
207
+ dq_par.append('Timestamp latency is reported per source layer (field/edge, OT lake, DOF) when ingest '
208
+ 'timestamps are present; this source carries none, so latency is not stamped.')
209
+ dq_sec = Section('3', 'Data quality (SOW 4.2.2)', dq_par,
210
+ [Table(['Tag', 'n', 'GOOD %', 'RANGE', 'ROC', 'FLATLINE', 'SPIKE', 'STALE', 'GAP',
211
+ 'Longest gap', 'Latency p95 (s)'], dq_rows)])
212
+
213
+ # 4. Gauge drift evaluation (SOW 4.2.10) ------------------------------------
214
+ st_rows = []
215
+ for s in stations:
216
+ env = s.get('drift_envelope_psi_yr') or [None, None]
217
+ key = CLASS_KEY.get(s['classification'], {'status': '-', 'action': '-'})
218
+ st_rows.append([s['channel'], _fmt(s.get('md_ft')), _fmt(s.get('predicted_baseline_psi')), s.get('n'),
219
+ f"{(s.get('span_years') or 0)*365.25:.0f}", _fmt(s.get('bias_psi')),
220
+ _fmt(s.get('slope_psi_yr')) if s.get('trend_usable') else 'window too short',
221
+ f"{_fmt(min(env))} to {_fmt(max(env))}" if env[0] is not None else '-',
222
+ _fmt(s.get('noise_sigma_psi')), s.get('transient_count'),
223
+ s['classification'], key['status'], key['action']])
224
+ drift_sec = Section('4', 'Gauge drift evaluation and reconciliation (SOW 4.2.10)',
225
+ ['For each station the live pressure series is compared with the well model baseline at '
226
+ 'that measured depth. Residual = measured - baseline. Bias is the mean residual; drift rate '
227
+ 'is the fitted linear trend of the residual; the aging envelope is the expected instrument '
228
+ 'drift band at station pressure and temperature from the gauge library; noise is the '
229
+ 'robust 1-sigma of the detrended residual. Classification and action follow the key below.'],
230
+ [Table(['Tag', 'Station MD (ft)', 'Baseline (psi)', 'n', 'Window (days)', 'Bias (psi)',
231
+ 'Drift rate (psi/yr)', 'Aging envelope (psi/yr)', 'Noise 1-sigma (psi)', 'Transients',
232
+ 'Classification', 'Status', 'Action'], st_rows),
233
+ Table(['Classification', 'Meaning', 'Status', 'Action'],
234
+ [[k, v['meaning'], v['status'], v['action']] for k, v in CLASS_KEY.items()],
235
+ caption='Classification key')])
236
+
237
+ # 5. SLA - model drift and staleness (SLA 1.0 / drift SLA) ------------------
238
+ next_due = now + timedelta(hours=SLA_EVALUATE_EVERY_H)
239
+ ms = monitor_status or {}
240
+ stale = ms.get('staleness') or {}
241
+ if stale:
242
+ cad = stale.get('cadence_h', SLA_EVALUATE_EVERY_H)
243
+ st_txt = {'CURRENT': f"CURRENT - last evaluation {stale.get('age_h')} h ago, cadence {cad:.0f} h",
244
+ 'STALE': f"STALE - last evaluation {stale.get('age_h')} h ago, {stale.get('overdue_h')} h overdue against cadence {cad:.0f} h",
245
+ 'NEVER_EVALUATED': 'NEVER EVALUATED'}.get(stale.get('status'), str(stale.get('status')))
246
+ sla_rows = [['Evaluation timestamp (UTC)', stale.get('last_evaluation_utc') or now.strftime('%Y-%m-%dT%H:%M:%SZ')],
247
+ ['Evaluation ID', stale.get('last_evaluation_id') or '-'],
248
+ ['Evaluation cadence', f'every {cad:.0f} h'],
249
+ ['Next evaluation due (UTC)', stale.get('next_due_utc') or '-'],
250
+ ['Model staleness', st_txt],
251
+ ['Evaluations on record', str(ms.get('evaluation_count', 0))],
252
+ ['Model drift', 'DETECTED' if drift_detected else 'NOT DETECTED'],
253
+ ['Stations requiring action', ', '.join(s['channel'] for s in drift_stations) or 'none'],
254
+ ['Offset corrections in force (psi)', ', '.join(f'{k}: {v:+.2f}' for k, v in (ms.get('corrections_psi') or {}).items()) or 'none'],
255
+ ['Record integrity (sha256, first 16)', ', '.join(f'{k} {v}' for k, v in (ms.get('log_sha256') or {}).items()) or '-']]
256
+ else:
257
+ sla_rows = [['Evaluation timestamp (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
258
+ ['Evaluation cadence', f'every {SLA_EVALUATE_EVERY_H:.0f} h'],
259
+ ['Next evaluation due (UTC)', next_due.strftime('%Y-%m-%dT%H:%M:%SZ')],
260
+ ['Model staleness', 'CURRENT - evaluated within the cadence'],
261
+ ['Model drift', 'DETECTED' if drift_detected else 'NOT DETECTED'],
262
+ ['Stations requiring action', ', '.join(s['channel'] for s in drift_stations) or 'none']]
263
+ if drift_detected:
264
+ sla_rows += [['Notification due', _business_days_after(now, SLA_NOTIFY_BD).strftime('%Y-%m-%d') + f' ({SLA_NOTIFY_BD} business day)'],
265
+ ['Fallback due', _business_days_after(now, SLA_FALLBACK_BD).strftime('%Y-%m-%d') + f' ({SLA_FALLBACK_BD} business days)'],
266
+ ['Re-fit / redeploy due', _business_days_after(now, SLA_REFIT_BD).strftime('%Y-%m-%d') + f' ({SLA_REFIT_BD} business days)']]
267
+ sla_tables = [Table(['Item', 'Value'], sla_rows)]
268
+ hist = ms.get('history') or []
269
+ if hist:
270
+ sla_tables.append(Table(['Evaluation (UTC)', 'Evaluation ID', 'Station', 'Classification', 'Bias (psi)',
271
+ 'Drift rate (psi/yr)', 'Correction in force (psi)'],
272
+ [[h.get('timestamp_utc'), h.get('evaluation_id'), h.get('station'), h.get('classification'),
273
+ _fmt(h.get('bias_psi')), _fmt(h.get('slope_psi_yr')), _fmt(h.get('correction_in_force_psi'))]
274
+ for h in hist[-10:]], caption='Evaluation history (last 10 entries)'))
275
+ sla_sec = Section('5', 'Model drift and staleness status (SLA 1.0; drift and staleness SLA)',
276
+ ['Model Drift: the live residual at a station moves outside the band the well model and the '
277
+ 'instrument aging envelope account for. Model Staleness: the evaluation is older than the '
278
+ 'cadence. Clocks below start at the evaluation timestamp when drift is detected.'],
279
+ sla_tables)
280
+
281
+ # 6. Configuration in force (SOW 4.2.1.4) -----------------------------------
282
+ th = dict(evaluation.get('thresholds_disclosed', {}))
283
+ th.pop('note', None)
284
+ cfg_rows = [['Bias significance', f"{th.get('bias_n_sigma')} x noise / sqrt(n) + 1 psi floor"],
285
+ ['Transient gate', f"{th.get('transient_n_sigma')} x noise, single sample"],
286
+ ['Bias too large for calibration', f"{th.get('model_mismatch_psi')} psi"],
287
+ ['Aging envelope margin', f"{th.get('drift_envelope_margin')} x upper envelope"],
288
+ ['Minimum window for a trend', f"{(th.get('min_trend_span_years') or 0)*365.25:.0f} days"],
289
+ ['Minimum samples per station', '8']]
290
+ rule_rows = []
291
+ for r in catalogue.rows():
292
+ rule_rows.append([r['tag_id'], f"{_fmt(r['eng_range_lo'])} to {_fmt(r['eng_range_hi'])} {r['unit']}",
293
+ _fmt(r['roc_limit_per_s']) + '/s' if r['roc_limit_per_s'] != '' else 'off',
294
+ f"{r['flatline_min_samples']} samples" if r['flatline_min_samples'] else 'off',
295
+ f"{_fmt(r['spike_n_sigma'])} x MAD, window {r['spike_window']}" if r['spike_n_sigma'] != '' else 'off',
296
+ f"{_fmt(r['stale_after_s'])} s" if r['stale_after_s'] != '' else 'off',
297
+ r['limits_basis']])
298
+ cfg_par = ['Classification thresholds and per-tag quality rules applied by this evaluation. These are '
299
+ 'engineering settings under version control; a change is a change-log entry, never silent.']
300
+ if gauge_spec is not None:
301
+ cfg_par.append(f"Gauge datasheet in force: '{gauge_spec.name}' - {gauge_spec.source}")
302
+ cfg_sec = Section('6', 'Configuration in force (SOW 4.2.1.4)', cfg_par,
303
+ [Table(['Threshold', 'Value'], cfg_rows, caption='Classification thresholds'),
304
+ Table(['Tag', 'Engineering range', 'Rate-of-change limit', 'Flatline', 'Spike', 'Staleness', 'Basis'],
305
+ rule_rows, caption='Quality rules per tag')])
306
+
307
+ # 7. Change log ------------------------------------------------------------
308
+ cl_rows = []
309
+ for s in stations:
310
+ if s['classification'] == 'CALIBRATION_OFFSET':
311
+ cl_rows.append([now.strftime('%Y-%m-%dT%H:%M:%SZ'), s['channel'], 'PROPOSED offset correction',
312
+ f"-{_fmt(s.get('bias_psi'))} psi", 'awaiting approval'])
313
+ elif s['classification'] in ('UNEXPLAINED_OFFSET', 'UNEXPLAINED_TREND'):
314
+ cl_rows.append([now.strftime('%Y-%m-%dT%H:%M:%SZ'), s['channel'], 'PROPOSED fallback',
315
+ 'hold model output for this station', 'awaiting approval'])
316
+ if ms.get('change_log'):
317
+ cl_rows = [[e.get('timestamp_utc'), e.get('station'), e.get('type'),
318
+ (f"{e.get('coefficient')}: {_fmt(e.get('before'))} -> {_fmt(e.get('after'))} psi" if e.get('type') == 'REFIT_OFFSET'
319
+ else str(e.get('detail', ''))),
320
+ e.get('status') + (f" by {e.get('approver')} {e.get('decided_utc')}" if e.get('approver') else '')]
321
+ for e in ms['change_log'][-10:]]
322
+ cl_sec = Section('7', 'Change log entries proposed by this evaluation' if not ms.get('change_log') else 'Change log (last 10 entries)',
323
+ ['No configuration or model coefficient was changed by generating this report. Entries below '
324
+ 'are proposals for the approval workflow; an approved entry is applied and logged with the '
325
+ 'approver, timestamp and before/after values.' if cl_rows else
326
+ 'No configuration or model coefficient was changed by generating this report, and no change '
327
+ 'is proposed.'],
328
+ [Table(['Timestamp (UTC)', 'Tag', 'Change', 'Value', 'Status'], cl_rows)] if cl_rows else [])
329
+
330
+ # 8. Data provenance and limitations ---------------------------------------
331
+ prov = [f"Live data: {evaluation.get('stream', stream.name)}."]
332
+ for k in ('source_channel', 'source_unit', 'conversion', 'station_md_ft', 'start_date', 'end_date', 'cadence_s'):
333
+ if meta.get(k):
334
+ prov.append(f"{k.replace('_', ' ')}: {meta[k]}.")
335
+ w = evaluation.get('well', {})
336
+ prov.append(f"Well model: total depth {_fmt(w.get('td_ft'))} ft; profile '{w.get('profile')}'; "
337
+ f"deviation '{w.get('deviation')}'; gauge datasheet '{w.get('gauge_spec')}'.")
338
+ prov.append('Limitations: a trend is only classified when the window meets the minimum in section 6; '
339
+ 'the aging envelope is a model band, not a measurement of this gauge; station depths supplied '
340
+ 'by the caller are marked as such above and should be confirmed from completion records.')
341
+ prov_sec = Section('8', 'Data provenance and limitations', [' '.join(prov)])
342
+
343
+ front = [['Report ID', report_id],
344
+ ['Well', well],
345
+ ['Generated (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
346
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
347
+ ['Data window', f'{first} to {last}'],
348
+ ['Stations', str(n_st)],
349
+ ['Result', 'MODEL DRIFT DETECTED' if drift_detected else 'NO MODEL DRIFT DETECTED']]
350
+ doc = Document(title=f'{PROGRAM_NAME} - Gauge Drift and Reconciliation Report', report_id=report_id,
351
+ front=front, sections=[summary, tag_sec, dq_sec, drift_sec, sla_sec, cfg_sec, cl_sec, prov_sec],
352
+ footer='Every number in this report is recomputed from the source data at generation time. '
353
+ 'Classifications are advisory; the numbers beside them are the record.',
354
+ data={'evaluation': evaluation, 'data_quality': dq, 'catalogue': catalogue.rows(),
355
+ 'drift_detected': drift_detected, 'report_id': report_id,
356
+ 'evaluated_at_utc': now.strftime('%Y-%m-%dT%H:%M:%SZ')})
357
+ doc.data['_records'] = records
358
+ doc.data['_catalogue_obj'] = catalogue
359
+ return doc
360
+
361
+
362
+ # ---------------------------------------------------------------------------
363
+ # Report 2 - Accuracy Statement (SLA 4.0; SOW 4.2.10 MAPE)
364
+ # ---------------------------------------------------------------------------
365
+ BAND_TEXT = {'MEETS_TARGET': 'MEETS TARGET (>= 95 %)', 'BAND_2': 'BAND 2 (90-95 %)',
366
+ 'BAND_3': 'BAND 3 (85-90 %)', 'NOT_ACCEPTABLE': 'NOT ACCEPTABLE (< 85 %)'}
367
+
368
+
369
+ def accuracy_statement_report(backtest: dict, evaluated_at: Optional[datetime] = None,
370
+ program_version: str = '', scope: str = 'library back-test') -> Document:
371
+ """The Accuracy Statement from a `library_backtest()` (or any dict of the
372
+ same shape) - MAPE at the stated CI per predicted quantity per well, the
373
+ calibration check, and the band read at the conservative end of the CI."""
374
+ now = evaluated_at or _utc_now()
375
+ meth = backtest.get('method', {})
376
+ ci = meth.get('ci', 0.90)
377
+ ok = backtest.get('statements', [])
378
+ pend = backtest.get('pending', [])
379
+ bands = backtest.get('bands', {})
380
+ report_id = f"ACC-{scope.replace(' ', '_')[:20]}-{now.strftime('%Y%m%dT%H%M%SZ')}"
381
+ n_meet = bands.get('MEETS_TARGET', 0)
382
+ parts = [f'{len(ok)} predicted quantities scored on {sum(r["n"] for r in ok)} blind trials; '
383
+ f'{len(pend)} pending below the minimum of {meth.get("min_trials")} trials.']
384
+ parts.append('Bands at the conservative end of the ' + f'{ci*100:.0f} % CI: ' +
385
+ ', '.join(f'{BAND_TEXT.get(b, b)} x{n}' for b, n in sorted(bands.items(), key=lambda kv: -kv[1])) + '.')
386
+ if ok:
387
+ parts.append(f'Best MAPE {backtest.get("best_mape_pct"):.2f} %; worst {backtest.get("worst_mape_pct"):.2f} %; '
388
+ f'median calibration coverage at the CI level {backtest.get("median_coverage_at_ci")} (expected about {ci:.2f}).')
389
+ parts.append(f'{n_meet} of {len(ok)} quantities meet the 95 % accuracy target; the quantities that do not are listed '
390
+ 'with their numbers and are excluded from any accuracy claim until they do.')
391
+ summary = Section('1', 'Summary', [' '.join(parts)])
392
+
393
+ defs = Section('2', 'Definitions in force (SLA 4.0; SOW 4.2.10)',
394
+ ['Absolute percentage error per trial: |estimate - truth| / |truth| x 100. MAPE: the mean over trials. '
395
+ 'Accuracy: 100 - MAPE. Confidence interval: bootstrap percentile interval on MAPE '
396
+ f'({meth.get("bootstrap_resamples")} resamples, seed {meth.get("seed")}, reproducible). '
397
+ f'Coverage at CI: the fraction of trials whose truth lies inside the estimate\'s own +/- {Z90:.3f}-sigma band '
398
+ '(the two-sided normal quantile at the CI level), a calibration check on the stated uncertainty. '
399
+ 'Band: accuracy read against 95 / 90 / 85 % '
400
+ 'using the upper CI bound of MAPE, so a statement never claims a band the interval does not support. '
401
+ 'A quantity with fewer trials than the minimum is PENDING, never scored.'],
402
+ [Table(['Band', 'Accuracy (100 - MAPE at the upper CI bound)'],
403
+ [['MEETS TARGET', '>= 95 %'], ['BAND 2', '90 % to < 95 %'], ['BAND 3', '85 % to < 90 %'],
404
+ ['NOT ACCEPTABLE', '< 85 %']])])
405
+
406
+ rows = []
407
+ for r in ok:
408
+ rows.append([r['well'], r['given'], r['target'], r['n'], f"{r['mape_pct']:.2f}",
409
+ f"{r['mape_ci_lo_pct']:.2f} to {r['mape_ci_hi_pct']:.2f}",
410
+ f"{r['accuracy_pct']:.2f}", f"{r['accuracy_conservative_pct']:.2f}",
411
+ _fmt(r['coverage_at_ci']), BAND_TEXT.get(r['band'], r['band'])])
412
+ stmt = Section('3', 'Accuracy statement',
413
+ [f'Per predicted quantity, per well, over the back-test window of the library. CI level {ci*100:.0f} %.'],
414
+ [Table(['Well', 'Given', 'Predicted quantity', 'n', 'MAPE %', f'MAPE {ci*100:.0f} % CI',
415
+ 'Accuracy %', 'Accuracy % (conservative)', 'Coverage at CI', 'Band'], rows)])
416
+
417
+ prow = [[r['well'], r['given'], r['target'], r['n'], r.get('min_n'), 'PENDING - below minimum trials'] for r in pend]
418
+ pending = Section('4', 'Pending quantities (not scored)',
419
+ ['Quantities whose co-located trials are below the minimum are reported with their count and no number.'
420
+ if prow else 'None.'],
421
+ [Table(['Well', 'Given', 'Predicted quantity', 'n', 'Minimum', 'Status'], prow)] if prow else [])
422
+
423
+ method = Section('5', 'Back-test method (for inspection; SCC 5.0)',
424
+ ['Hold-out: ' + str(meth.get('hold_out')) + '. Estimator: ' + str(meth.get('estimator')) +
425
+ '. Every trial holds one co-located observation out, predicts it from the rest with the same estimator '
426
+ 'the product runs, and scores the prediction against the held-out truth. Nothing is tuned to this test. '
427
+ f'Resamples {meth.get("bootstrap_resamples")}, seed {meth.get("seed")}: the statement regenerates '
428
+ 'byte-identically from the same library, so it can be re-run by the client from source.'])
429
+
430
+ plan_rows = [['Go-live', 'Baseline accuracy statement on the client\'s own validated data, this method'],
431
+ ['Monthly', 'Statement regenerated; drift evaluation feeds re-fits; bands updated'],
432
+ ['Quarterly', 'Aggregated statement for the SLA review; bands read at the conservative CI end'],
433
+ ['Annually, years 1-5', 'Full re-statement with the year\'s validated tests as the back-test window']]
434
+ plan = Section('6', 'Accuracy measurement plan (five years)',
435
+ ['The statement is regenerated on the cadence below from the client\'s validated measurements, '
436
+ 'never from a stored snapshot. Each statement carries its window, n and CI.'],
437
+ [Table(['When', 'What'], plan_rows)])
438
+
439
+ prov = Section('7', 'Provenance and limitations',
440
+ [f'Scope: {scope}. The quantities scored here are the property estimators over the archived public '
441
+ 'library the program ships with; they are a statement about the estimator on that library, not a '
442
+ 'claim about a client well. Accuracy of a deployed model is measured on the client\'s own data by the '
443
+ 'same method, with the same definitions, before any band is claimed. Quantities with truth values at '
444
+ 'zero cannot carry a percentage error and are excluded from MAPE with the count disclosed.'])
445
+
446
+ front = [['Report ID', report_id], ['Scope', scope],
447
+ ['Generated (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
448
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
449
+ ['CI level', f'{ci*100:.0f} %'], ['Quantities scored / pending', f'{len(ok)} / {len(pend)}'],
450
+ ['Result', f'{n_meet} of {len(ok)} MEET TARGET' if ok else 'NO QUANTITY SCORED']]
451
+ machine = {k: v for k, v in backtest.items()}
452
+ return Document(title=f'{PROGRAM_NAME} - Accuracy Statement', report_id=report_id, front=front,
453
+ sections=[summary, defs, stmt, pending, method, plan, prov],
454
+ footer='Every number in this statement is recomputed from the source data at generation time; '
455
+ 'the bands are read at the conservative end of the confidence interval.',
456
+ data={'backtest': machine, 'report_id': report_id,
457
+ 'evaluated_at_utc': now.strftime('%Y-%m-%dT%H:%M:%SZ'), '_records': [], '_catalogue_obj': None})
458
+
459
+
460
+ # ---------------------------------------------------------------------------
461
+ # Report 3 - Well Test Validation (SOW 4.2.3.1)
462
+ # ---------------------------------------------------------------------------
463
+ def well_test_report(detection: dict, approvals=None, well_name: str = '',
464
+ evaluated_at: Optional[datetime] = None, program_version: str = '') -> Document:
465
+ """The Well Test Validation Report from a `WellTestValidator.detect()`
466
+ result and (optionally) an `ApprovalTrail`."""
467
+ now = evaluated_at or _utc_now()
468
+ c = detection.get('criteria', {})
469
+ tests = detection.get('tests', [])
470
+ rej = detection.get('rejected', [])
471
+ well = well_name or detection.get('stream', 'well')
472
+ report_id = f"WTV-{well.replace('/', '-').replace(' ', '_')[:24]}-{now.strftime('%Y%m%dT%H%M%SZ')}"
473
+ win = detection.get('window') or ['-', '-']
474
+ status_of = (lambda tid: approvals.status_of(tid)) if approvals is not None else (lambda tid: {'status': 'PENDING_LEVEL_1', 'decisions': []})
475
+ n_app = sum(1 for t in tests if status_of(t['test_id'])['status'] == 'APPROVED_ALL_LEVELS')
476
+
477
+ summary = Section('1', 'Summary', [
478
+ f"{detection.get('n_samples')} samples from {win[0]} to {win[1]}; {detection.get('eligible_samples')} eligible "
479
+ f"(on stream, complete, quality GOOD). {len(tests)} stable period{'s' if len(tests) != 1 else ''} detected and "
480
+ f"ACCEPTED as well tests covering {detection.get('samples_in_accepted_tests')} samples; {len(rej)} candidate "
481
+ f"period{'s' if len(rej) != 1 else ''} REJECTED with reason codes. {n_app} of {len(tests)} accepted tests "
482
+ f"approved at all levels; the rest await review. Virtual rates are the means over each accepted test and are "
483
+ f"released to allocation only after approval."])
484
+
485
+ crit_rows = [[k, _fmt(v)] for k, v in c.items() if not k.startswith('_') and k not in ('approval_levels', 'name', 'basis')]
486
+ crit_rows += [['approval levels', '; '.join(f"level {l['level']}: {l['role']}" for l in c.get('approval_levels', []))]]
487
+ map_rows = [[role, chan] for role, chan in (detection.get('channel_map') or {}).items()]
488
+ criteria = Section('2', 'Stability criteria in force (client-agreed logic; SOW 4.2.3.1)',
489
+ [f"Criteria: '{c.get('name')}'. Source: {c.get('_source')} (sha256 {c.get('_sha256')}). "
490
+ f"Basis: {c.get('basis')}. A change to this file is a change-log entry; the report always "
491
+ "prints the version it ran with."],
492
+ [Table(['Parameter', 'Value'], crit_rows, caption='Parameters'),
493
+ Table(['Role', 'Channel'], map_rows, caption='Channel roles')])
494
+
495
+ rate_roles = list(next(iter(tests))['virtual_rates'].keys()) if tests else []
496
+ _u = lambda u: (u or '').split(' ')[0]
497
+ unit_notes = sorted({u for t in tests for u in (v.get('unit', '') for v in t['statistics'].values()) if u and ' ' in u})
498
+ cols = ['Test', 'Start (UTC)', 'End (UTC)', 'n'] + [f'{r} (mean)' for r in rate_roles] + \
499
+ ['Downhole P (mean)', 'Wellhead P (mean)', 'Operating point', 'On stream (h, mean)', 'Approval']
500
+ trows = []
501
+ for t in tests:
502
+ st = t['statistics']
503
+ dh = st.get('pressure:downhole', {})
504
+ wh = st.get('pressure:wellhead', {})
505
+ op = '; '.join(f"{k.split(':')[1]} {v['mean']:g}" for k, v in st.items() if k.startswith('operating:'))
506
+ trows.append([t['test_id'], t['start_utc'][:10], t['end_utc'][:10], t['n']] +
507
+ [f"{t['virtual_rates'][r]:,.1f} {_u(st[f'rate:{r}']['unit'])}" for r in rate_roles] +
508
+ [f"{dh.get('mean', float('nan')):,.1f} {_u(dh.get('unit', ''))}", f"{wh.get('mean', float('nan')):,.1f} {_u(wh.get('unit', ''))}",
509
+ op or '-', _fmt(st.get('on_stream', {}).get('mean_hours')), status_of(t['test_id'])['status']])
510
+ accepted = Section('3', 'Accepted well tests',
511
+ ['Each accepted test is the longest window from its start that satisfies every criterion. '
512
+ 'Means are the test result; CV and trend per channel are in the machine JSON.'],
513
+ [Table(cols, trows)] if trows else [])
514
+
515
+ rrows = [[r['candidate_id'], r['start_utc'][:10], r['end_utc'][:10], r['n'], r.get('primary_reason') or ', '.join(r['reason_codes']),
516
+ r.get('detail', '')[:90]] for r in rej]
517
+ rejected = Section('4', 'Rejected candidate periods (reason codes)',
518
+ ['Every period not accepted is listed with the rule that failed and the number that failed it.'],
519
+ [Table(['Candidate', 'Start (UTC)', 'End (UTC)', 'n', 'Primary reason', 'Detail'], rrows)] if rrows else [])
520
+
521
+ arows = []
522
+ if approvals is not None:
523
+ for e in approvals.entries():
524
+ arows.append([e['timestamp_utc'], e['test_id'], f"L{e['level']} {e['role']}", e['approver'], e['decision'], e.get('note', '')])
525
+ trail = Section('5', 'Approval trail',
526
+ ['Multi-level approval with timestamps. A test is released when every level has approved; a '
527
+ 'rejection at any level holds it. Corrections are recorded as CORRECTED decisions with a note.'
528
+ if arows else 'No approvals recorded yet.'],
529
+ [Table(['Timestamp (UTC)', 'Test', 'Level', 'Approver', 'Decision', 'Note'], arows)] if arows else [])
530
+
531
+ prov = Section('6', 'Data provenance and limitations',
532
+ [f"Source stream: {detection.get('stream')}. Detection is deterministic from the stream and the criteria "
533
+ "file; re-running with the same inputs reproduces this report. Rates whose mean is below the negligible "
534
+ "fraction of the largest rate in the window are not stability criteria (disclosed in section 2). "
535
+ "Quality flags come from the record layer's rules when a tag catalogue is supplied."
536
+ + (' Units: ' + '; '.join(unit_notes) + '.' if unit_notes else '')])
537
+
538
+ front = [['Report ID', report_id], ['Well', well], ['Generated (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
539
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
540
+ ['Data window', f'{win[0]} to {win[1]}'],
541
+ ['Accepted / rejected', f'{len(tests)} / {len(rej)}'],
542
+ ['Result', f'{len(tests)} WELL TESTS ACCEPTED, {n_app} APPROVED' if tests else 'NO STABLE PERIOD FOUND']]
543
+ return Document(title=f'{PROGRAM_NAME} - Well Test Validation Report', report_id=report_id, front=front,
544
+ sections=[summary, criteria, accepted, rejected, trail, prov],
545
+ footer='Every number in this report is recomputed from the source data and the criteria file at '
546
+ 'generation time; reason codes name the rule and the value that failed it.',
547
+ data={'detection': detection, 'report_id': report_id,
548
+ 'approvals': (approvals.entries() if approvals is not None else []),
549
+ 'evaluated_at_utc': now.strftime('%Y-%m-%dT%H:%M:%SZ'), '_records': [], '_catalogue_obj': None})
550
+
551
+
552
+ # ---------------------------------------------------------------------------
553
+ # Report 4 - Alarm & Event (SOW 4.2.4; ISA-18.2 / IEC 62682 practice)
554
+ # ---------------------------------------------------------------------------
555
+ def alarm_event_report(engine, well_name: str = '', evaluated_at: Optional[datetime] = None,
556
+ program_version: str = '', last_events: int = 60) -> Document:
557
+ now = evaluated_at or _utc_now()
558
+ k = engine.kpis()
559
+ defs = list(engine.defs.values())
560
+ active = engine.active()
561
+ well = well_name or 'well'
562
+ report_id = f"ALM-{well.replace('/', '-').replace(' ', '_')[:24]}-{now.strftime('%Y%m%dT%H%M%SZ')}"
563
+ n_act = k.get('n_activations', 0)
564
+ parts = [f"{len(defs)} alarm definitions in force on {len(engine.by_tag)} tags; {n_act} activations "
565
+ + (f"over {k.get('window_hours')} h ({k.get('per_day')} per day, average {k.get('avg_per_10min_per_position')} "
566
+ f"per 10 min per operator position, peak {k.get('peak_per_10min_per_position')})." if n_act else 'in the window.')]
567
+ if n_act:
568
+ parts.append(f"{k.get('flood_10min_bins')} ten-minute flood periods ({k.get('pct_time_in_flood')} % of the window); "
569
+ f"{len(k.get('standing_alarms_over_24h', []))} standing; {len(k.get('chattering_alarms', []))} chattering; "
570
+ f"{len(k.get('active_unacknowledged', []))} active unacknowledged at window end.")
571
+ top = k.get('top_alarms', [])[:3]
572
+ if top:
573
+ parts.append('Most frequent: ' + ', '.join(f"{t['alarm_id']} ({t['activations']}, {t['pct_of_total']} %)" for t in top) + '.')
574
+ if k.get('avg_per_10min_per_position', 0) > 2 or k.get('flood_10min_bins', 0):
575
+ parts.append('The rate exceeds the manageable target; the definitions table shows which alarms carry the load, '
576
+ 'and the usual remedy is to move journal-type conditions (sample quality) out of the operator alarm list.')
577
+ summary = Section('1', 'Summary', [' '.join(parts)])
578
+
579
+ drows = [[d.alarm_id, d.tag_id, d.kind, d.priority, _fmt(d.setpoint), _fmt(d.deadband), _fmt(d.on_delay_s),
580
+ 'yes' if d.enabled else 'no', d.basis] for d in defs]
581
+ defsec = Section('2', 'Alarm definitions in force (SOW 4.2.4)',
582
+ ['Each alarm names its tag, kind, setpoint, return-to-normal deadband, on-delay, priority and the '
583
+ 'basis of the setpoint. Over-range alarms derive from the tag catalogue; process setpoints are '
584
+ 'client settings loaded from the definitions file.'],
585
+ [Table(['Alarm', 'Tag', 'Kind', 'Priority', 'Setpoint', 'Deadband', 'On-delay (s)', 'Enabled', 'Basis'], drows)])
586
+
587
+ arows = [[a['alarm_id'], a['tag_id'], a['priority'], a['state'], a['active_since_utc'] or '-', _fmt(a['last_value'])] for a in active]
588
+ actsec = Section('3', 'Active alarms at window end',
589
+ ['State per ISA-18.2: ACTIVE_UNACKED awaits operator acknowledgement; ACTIVE_ACKED is acknowledged and still in alarm.'
590
+ if arows else 'No alarm active at window end.'],
591
+ [Table(['Alarm', 'Tag', 'Priority', 'State', 'Active since (UTC)', 'Last value'], arows)] if arows else [])
592
+
593
+ evs = engine.events[-last_events:]
594
+ erows = [[e['timestamp_utc'], e['alarm_id'], e['priority'], e['event'], _fmt(e['value']), e.get('operator', ''), e.get('note', '')[:60]] for e in evs]
595
+ evsec = Section('4', f'Event log (last {len(erows)} of {len(engine.events)} events)',
596
+ ['One line per transition; the full log is the JSON lines file named in section 6.'],
597
+ [Table(['Timestamp (UTC)', 'Alarm', 'Priority', 'Event', 'Value', 'Operator', 'Note'], erows)] if erows else [])
598
+
599
+ tg = k.get('targets', {})
600
+ krows = [['Activations per day', _fmt(k.get('per_day')), tg.get('per_day', '')],
601
+ ['Average per 10 min per operator position', _fmt(k.get('avg_per_10min_per_position')), tg.get('avg_per_10min', '')],
602
+ ['Peak per 10 min per operator position', _fmt(k.get('peak_per_10min_per_position')), tg.get('flood', '')],
603
+ ['Ten-minute flood periods / % time in flood', f"{k.get('flood_10min_bins', 0)} / {k.get('pct_time_in_flood', 0)} %", tg.get('flood', '')],
604
+ ['Standing alarms', ', '.join(k.get('standing_alarms_over_24h', [])) or 'none', tg.get('standing', '')],
605
+ ['Chattering alarms', ', '.join(k.get('chattering_alarms', [])) or 'none', tg.get('chattering', '')],
606
+ ['Priority distribution (%)', ', '.join(f'{p} {v}' for p, v in (k.get('priority_distribution_pct') or {}).items()), tg.get('priority_split', '')]] if n_act else \
607
+ [['Activations', '0', 'no activations in the window']]
608
+
609
+ trows = [[t['alarm_id'], t['activations'], t['pct_of_total']] for t in k.get('top_alarms', [])]
610
+ kpisec = Section('5', 'Alarm-management KPIs against targets',
611
+ [tg.get('basis', 'targets as commonly stated in alarm-management practice') + '.'],
612
+ [Table(['KPI', 'Value', 'Target'], krows)] + ([Table(['Alarm', 'Activations', '% of total'], trows, caption='Most frequent alarms')] if trows else []))
613
+
614
+ prov = Section('6', 'Provenance', [f"Operator positions: {engine.operator_positions}. Event log: "
615
+ f"{engine.log_path or 'in memory for this run'}. Window: "
616
+ f"{k.get('window', ['-', '-'])[0] if k.get('window') else '-'} to "
617
+ f"{k.get('window', ['-', '-'])[1] if k.get('window') else '-'}. The state machine, "
618
+ "deadband and on-delay are applied per alarm in time order; shelved alarms are "
619
+ "suppressed and logged."])
620
+ front = [['Report ID', report_id], ['Well', well], ['Generated (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
621
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
622
+ ['Definitions / activations', f'{len(defs)} / {n_act}'],
623
+ ['Result', (f"{len(active)} ACTIVE, {len(k.get('active_unacknowledged', []))} UNACKNOWLEDGED" if active else 'NO ACTIVE ALARM')]]
624
+ return Document(title=f'{PROGRAM_NAME} - Alarm and Event Report', report_id=report_id, front=front,
625
+ sections=[summary, defsec, actsec, evsec, kpisec, prov],
626
+ footer='Every KPI is recomputed from the event log at generation time; targets are printed beside values, never in place of them.',
627
+ data={'kpis': k, 'definitions': [d.row() for d in defs], 'active': active, 'events': engine.events,
628
+ 'report_id': report_id, 'evaluated_at_utc': now.strftime('%Y-%m-%dT%H:%M:%SZ'),
629
+ '_records': [], '_catalogue_obj': None})
630
+
631
+
632
+ # ---------------------------------------------------------------------------
633
+ # Report 5 - Model Card (SCC 5.0 inspection; M3 explainability)
634
+ # ---------------------------------------------------------------------------
635
+ def model_card_report(card, evaluated_at: Optional[datetime] = None) -> Document:
636
+ now = evaluated_at or _utc_now()
637
+ report_id = f"MC-{card.model_id}-{card.version}"
638
+ secs = [
639
+ Section('1', 'Intended use', [card.intended_use]),
640
+ Section('2', 'Out-of-scope uses', [card.out_of_scope]),
641
+ Section('3', 'Inputs and feature definitions (with sources)', [],
642
+ [Table(['Input', 'Unit / definition', 'Source'], [[i['name'], i['unit'], i['source']] for i in card.inputs])]),
643
+ Section('4', 'Settings and hyperparameters', [],
644
+ [Table(['Setting', 'Value', 'Basis'], [[x['setting'], x['value'], x['basis']] for x in card.settings])]),
645
+ Section('5', 'Calibration and training data (provenance)', [],
646
+ [Table(['Dataset', 'Records', 'Database', 'URL', 'Licence', 'Fetch date'],
647
+ [[d['dataset'], d['records'], d['database'], d['url'], d['licence'], d['fetch_date']] for d in card.calibration_data])]),
648
+ Section('6', 'Evaluation, validation and back-test results (recomputed)', [],
649
+ [Table(['Metric', 'Result', 'Method'], [[e['metric'], e['value'], e['method']] for e in card.evaluation])]),
650
+ Section('7', 'Known limitations', [' '.join(f'({i+1}) {l}' for i, l in enumerate(card.limitations))]),
651
+ Section('8', 'Re-fit and retraining history', ['No re-fit on record for this model.'] if not card.refit_history else [],
652
+ [Table(['Decided (UTC)', 'Station', 'Coefficient', 'Before', 'After', 'Status', 'Approver'],
653
+ [[r['timestamp_utc'], r['station'], r['coefficient'], _fmt(r['before']), _fmt(r['after']), r['status'], r['approver']] for r in card.refit_history])] if card.refit_history else []),
654
+ Section('9', 'Components for inspection', ['Each component is named by role with the sha256 (first 16 hex) of the source file that implements it in this build; the delivered source matches these hashes.'],
655
+ [Table(['Component', 'sha256'], [[c['role'], c['sha256']] for c in card.components])]),
656
+ ]
657
+ front = [['Model ID', card.model_id], ['Version', f'{PROGRAM_NAME} build {card.version}'], ['Owner', card.owner],
658
+ ['Generated (UTC)', card.generated_utc or now.strftime('%Y-%m-%dT%H:%M:%SZ')],
659
+ ['Inputs / datasets / evaluations', f'{len(card.inputs)} / {len(card.calibration_data)} / {len(card.evaluation)}']]
660
+ return Document(title=f'Model Card - {card.title}', report_id=report_id, front=front, sections=secs,
661
+ footer='Generated from the live objects of this build; every evaluation is recomputed at generation time.',
662
+ data={'card': card.to_dict(), 'report_id': report_id, '_records': [], '_catalogue_obj': None})
663
+
664
+
665
+ # ---------------------------------------------------------------------------
666
+ # Report 6 - Data Resilience: store-and-forward (SOW 4.2.1.6; SLA 5.0 remote site)
667
+ # ---------------------------------------------------------------------------
668
+ def data_resilience_report(sim: dict, site_name: str = '', evaluated_at: Optional[datetime] = None,
669
+ program_version: str = '') -> Document:
670
+ import numpy as _np
671
+ now = evaluated_at or _utc_now()
672
+ st = sim.get('stats', {})
673
+ cfg = sim.get('config', {})
674
+ gr = sim.get('gap_report', {})
675
+ outs = sim.get('outages', [])
676
+ site = site_name or 'site'
677
+ report_id = f"SF-{site.replace('/', '-').replace(' ', '_')[:24]}-{now.strftime('%Y%m%dT%H%M%SZ')}"
678
+ delivered = sim.get('delivered', [])
679
+ lat_all = _np.array([d.latency_s() for d in delivered if d.latency_s() is not None], dtype=float)
680
+ lat_live = _np.array([d.latency_s() for d in delivered if d.source_layer == 'OT_LAKE' and d.latency_s() is not None], dtype=float)
681
+ n_exp = sum(v['expected'] for v in gr.values())
682
+ n_del = sum(v['delivered'] for v in gr.values())
683
+ n_missing = sum(v['not_delivered'] for v in gr.values())
684
+ dup = sum(v['duplicates_in_delivery'] for v in gr.values())
685
+ chrono = all(v['chronological_replay'] for v in gr.values()) if gr else True
686
+ total_out_h = sum(o['duration_h'] for o in outs)
687
+ window_h = None
688
+ if delivered:
689
+ ts = sorted(parse_utc(d.timestamp_utc) for d in delivered)
690
+ window_h = (ts[-1] - ts[0]).total_seconds() / 3600.0
691
+ avail = round(100.0 * (1 - total_out_h / window_h), 3) if window_h else None
692
+ pct2 = lambda a: (round(100.0 * float(_np.mean(a <= 120)), 2) if len(a) else None)
693
+ pct5 = lambda a: (round(100.0 * float(_np.mean(a <= 300)), 2) if len(a) else None)
694
+
695
+ summary = Section('1', 'Summary', [
696
+ f"{n_exp} samples across {len(gr)} tags over {window_h:.1f} h with {len(outs)} link outage{'s' if len(outs) != 1 else ''} "
697
+ f"totalling {total_out_h:.2f} h. {st.get('live', 0)} delivered live, {st.get('buffered', 0)} buffered at the edge, "
698
+ f"{st.get('replayed', 0)} replayed in chronological order at {cfg.get('replay_rate_per_s')} records/s, "
699
+ f"{st.get('dropped_over_capacity', 0)} dropped over the {cfg.get('capacity_hours')} h capacity, "
700
+ f"{st.get('duplicates_suppressed', 0)} duplicates suppressed at the edge and {dup} in the delivered stream. "
701
+ f"{n_del} of {n_exp} delivered ({n_missing} not delivered). Peak buffer occupancy "
702
+ f"{100.0 * st.get('peak_backlog', 0) / max(cfg.get('capacity_records', 1), 1):.1f} % of capacity. "
703
+ f"Chronological replay {'verified' if chrono else 'VIOLATED'}. Remote-site link availability {avail} % over the window."])
704
+
705
+ cfg_rows = [['Buffer capacity', f"{cfg.get('capacity_hours')} h ({cfg.get('capacity_records')} records at {cfg.get('cadence_s')} s cadence x {len(gr)} tags)", 'SOW 4.2.1.6: 72 h store-and-forward'],
706
+ ['Replay rate', f"{cfg.get('replay_rate_per_s')} records/s, metered per second", 'SOW 4.2.1.6: chronological, rate-controlled replay'],
707
+ ['Edge processing latency (modelled)', f"{cfg.get('edge_latency_s')} s", 'sample timestamp to arrival at the edge'],
708
+ ['Duplicate suppression', 'by (tag, timestamp) at the edge and verified in the delivered stream', 'SLA 5.0: no duplication']]
709
+ cfgsec = Section('2', 'Buffer configuration (SOW 4.2.1.6)', [], [Table(['Item', 'Value', 'Requirement'], cfg_rows)])
710
+
711
+ orows = [[o['start_utc'], o['end_utc'], f"{o['duration_h']:.2f}"] for o in outs]
712
+ erows = [[e['timestamp_utc'], e['link'], e['backlog']] for e in sim.get('link_events', [])]
713
+ outsec = Section('3', 'Link outages and events', [],
714
+ [Table(['Outage start (UTC)', 'Outage end (UTC)', 'Duration (h)'], orows, caption='Outage windows'),
715
+ Table(['Timestamp (UTC)', 'Link', 'Backlog at event'], erows, caption='Link events')])
716
+
717
+ srows = [['Delivered live', st.get('live', 0)], ['Buffered at edge', st.get('buffered', 0)], ['Replayed', st.get('replayed', 0)],
718
+ ['Dropped over capacity', st.get('dropped_over_capacity', 0)], ['Duplicates suppressed (edge)', st.get('duplicates_suppressed', 0)],
719
+ ['Duplicates in delivered stream', dup], ['Peak backlog (records)', st.get('peak_backlog', 0)],
720
+ ['Final backlog', sim.get('final_backlog', 0)], ['Replay slots used', sim.get('replay_slots', 0)]]
721
+ stsec = Section('4', 'Delivery statistics', [], [Table(['Item', 'Value'], srows)])
722
+
723
+ lrows = [['All delivered records', _fmt(float(_np.percentile(lat_all, 50))) if len(lat_all) else '-', _fmt(float(_np.percentile(lat_all, 95))) if len(lat_all) else '-',
724
+ _fmt(float(_np.percentile(lat_all, 99))) if len(lat_all) else '-', _fmt(float(lat_all.max())) if len(lat_all) else '-', pct2(lat_all), pct5(lat_all)],
725
+ ['Live delivery only (outage replay excluded)', _fmt(float(_np.percentile(lat_live, 50))) if len(lat_live) else '-', _fmt(float(_np.percentile(lat_live, 95))) if len(lat_live) else '-',
726
+ _fmt(float(_np.percentile(lat_live, 99))) if len(lat_live) else '-', _fmt(float(lat_live.max())) if len(lat_live) else '-', pct2(lat_live), pct5(lat_live)]]
727
+ latsec = Section('5', 'End-to-end latency, field edge to OT lake (SLA 1.0 definitions)',
728
+ ['Latency is measured per record from the sample timestamp to the ingest timestamp at the receiving layer. '
729
+ 'The SLA target is 95 % within 2 min and 99 % within 5 min. Replayed backlog carries the outage duration as '
730
+ 'latency by definition; SLA 7.0 excludes third-party telecom outages with a documented root cause, so both '
731
+ 'rows are printed and the client applies the exclusion.'],
732
+ [Table(['Population', 'p50 (s)', 'p95 (s)', 'p99 (s)', 'max (s)', '% within 2 min', '% within 5 min'], lrows)])
733
+
734
+ grows = [[t, v['expected'], v['delivered_live'], v['delivered_by_replay'], v['not_delivered'], v['source_gaps_in_data'],
735
+ v['duplicates_in_delivery'], 'yes' if v['chronological_replay'] else 'NO', _fmt(v['latency_p95_s']), _fmt(v['latency_max_s'])]
736
+ for t, v in gr.items()]
737
+ gapsec = Section('6', 'Gap report per tag',
738
+ ['"Not delivered" are samples lost at the edge (over capacity); "source gaps" are samples the historian itself '
739
+ 'had no value for (flag GAP) and are delivered as such, never invented.'],
740
+ [Table(['Tag', 'Expected', 'Live', 'Replayed', 'Not delivered', 'Source gaps', 'Duplicates', 'Chronological', 'Latency p95 (s)', 'Latency max (s)'], grows)])
741
+
742
+ prov = Section('7', 'Provenance', [f"Site: {site}. The simulation drives the buffer with the records' own timestamps as the clock and the "
743
+ "outage windows as the link schedule; the same buffer class serves a live edge. Results regenerate identically from the same input."])
744
+ front = [['Report ID', report_id], ['Site', site], ['Generated (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
745
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
746
+ ['Samples / delivered / lost', f'{n_exp} / {n_del} / {n_missing}'],
747
+ ['Result', ('ALL SAMPLES DELIVERED, NO DUPLICATES, CHRONOLOGICAL' if n_missing == 0 and dup == 0 and chrono else f'{n_missing} LOST, {dup} DUPLICATES' + ('' if chrono else ', ORDER VIOLATED'))]]
748
+ machine = {k: v for k, v in sim.items() if k != 'delivered'}
749
+ return Document(title=f'{PROGRAM_NAME} - Data Resilience Report (store-and-forward)', report_id=report_id, front=front,
750
+ sections=[summary, cfgsec, outsec, stsec, latsec, gapsec, prov],
751
+ footer='Every statistic is recomputed from the delivered stream at generation time.',
752
+ data={'simulation': machine, 'report_id': report_id, 'evaluated_at_utc': now.strftime('%Y-%m-%dT%H:%M:%SZ'),
753
+ '_records': delivered, '_catalogue_obj': None})
754
+
755
+
756
+ # ---------------------------------------------------------------------------
757
+ # Report 7 - Monthly SLA (SLA 2.0 / 5.0)
758
+ # ---------------------------------------------------------------------------
759
+ def monthly_sla_report(m: dict, site_name: str = '', evaluated_at: Optional[datetime] = None, program_version: str = '',
760
+ config_summary: Optional[List[dict]] = None, sbom: Optional[dict] = None) -> Document:
761
+ now = evaluated_at or _utc_now()
762
+ site = site_name or 'site'
763
+ report_id = f"SLA-{m['month']}-{site.replace('/', '-').replace(' ', '_')[:20]}"
764
+ sc = m.get('status_counts', {})
765
+ lines = m.get('lines', [])
766
+ summary = Section('1', 'Summary', [
767
+ f"SLA measurement for {m['month']} ({m['window'][0][:10]} to {m['window'][1][:10]}): {len(lines)} lines - "
768
+ + ', '.join(f'{k} {v}' for k, v in sorted(sc.items())) + '. '
769
+ + ('Lines marked NOT MEASURED had no record supplied and are not assumed met. ' if sc.get('NOT MEASURED') else '')
770
+ + 'Penalty bands, if any, are read by the client from the NOT MET lines against the contract; this report states measurements only.'])
771
+ areas = []
772
+ seen = set()
773
+ for l in lines:
774
+ if l['area'] not in seen:
775
+ seen.add(l['area']); areas.append(l['area'])
776
+ tables = [Table(['Metric', 'Measured', 'Target', 'Status', 'Note'],
777
+ [[l['metric'], l['measured'], l['target'], l['status'], l['note']] for l in lines if l['area'] == ar], caption=ar) for ar in areas]
778
+ meas = Section('2', 'SLA lines (SLA 2.0 measurement; 5.0 post-implementation; drift and staleness SLA)', [], tables)
779
+ crows = [[c['name'], c['versions'], c['latest'], c['sha256'], c['last_change_utc'], c['last_author'], c['rollbacks']] for c in (config_summary or [])]
780
+ csec = Section('3', 'Change and continuity (SOW 4.2.1.4, 4.2.24)',
781
+ ['Configuration store state at month end; every version carries author, note, sha256 and a key-level diff.' if crows else 'No configuration store supplied.'],
782
+ [Table(['Configuration', 'Versions', 'Latest', 'sha256', 'Last change (UTC)', 'Author', 'Rollbacks'], crows)] if crows else [])
783
+ srows = [[c['name'], c['version'], c['supplier'], c['licence'], c['hash'], c['relationship']] for c in (sbom or {}).get('components', [])]
784
+ ssec = Section('4', 'Software bill of materials (SCC 16.0)',
785
+ [f"Generated {sbom.get('generated_utc')} on {sbom.get('platform')}; {sbom.get('n_components')} components; format: {sbom.get('format')}." if sbom else 'No SBOM supplied.'],
786
+ [Table(['Component', 'Version', 'Supplier', 'Licence', 'Hash', 'Relationship'], srows)] if srows else [])
787
+ irows = [[k, v or '-'] for k, v in m.get('inputs', {}).items()]
788
+ prov = Section('5', 'Inputs and provenance', ['Records read for this measurement. A missing input yields NOT MEASURED lines above.'],
789
+ [Table(['Input', 'Path'], irows)])
790
+ front = [['Report ID', report_id], ['Site', site], ['Month', m['month']], ['Generated (UTC)', now.strftime('%Y-%m-%dT%H:%M:%SZ')],
791
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
792
+ ['Result', f"{sc.get('MET', 0)} MET / {sc.get('NOT MET', 0)} NOT MET / {sc.get('NOT MEASURED', 0)} NOT MEASURED"]]
793
+ return Document(title=f'{PROGRAM_NAME} - Monthly SLA Report', report_id=report_id, front=front, sections=[summary, meas, csec, ssec, prov],
794
+ footer='Every line is measured from the named records at generation time; nothing unmeasured is reported as met.',
795
+ data={'measurement': m, 'config_summary': config_summary or [], 'sbom': sbom or {}, 'report_id': report_id,
796
+ 'evaluated_at_utc': now.strftime('%Y-%m-%dT%H:%M:%SZ'), '_records': [], '_catalogue_obj': None})
797
+
798
+
799
+ # ---------------------------------------------------------------------------
800
+ # Report 8 - FAT / SAT protocol (SOW 4.2.20 / 4.2.21)
801
+ # ---------------------------------------------------------------------------
802
+ def fat_sat_report(proto: dict, site_name: str = '', program_version: str = '') -> Document:
803
+ kind = proto['kind']
804
+ clause = '4.2.20' if kind == 'FAT' else '4.2.21'
805
+ title = 'Factory Acceptance Test Protocol' if kind == 'FAT' else 'Site Acceptance Test Protocol'
806
+ report_id = f"{kind}-{proto['started_utc'].replace(':', '').replace('-', '')}"
807
+ summary = Section('1', 'Summary', [
808
+ f"{proto['n_steps']} numbered steps across sections {', '.join(proto['sections'])}: {proto['n_pass']} PASS, {proto['n_fail']} FAIL. "
809
+ f"Result: {proto['result']}. Run {proto['started_utc']} to {proto['finished_utc']}. "
810
+ f"{proto['internal_checks_excluded']} internal build checks ran alongside and are excluded from this client protocol; "
811
+ "no check was altered."])
812
+ secs = [summary]
813
+ n = 2
814
+ if proto.get('environment'):
815
+ env = proto['environment']
816
+ secs.append(Section(str(n), 'Installed environment (SAT)', [f"Platform {env.get('platform')}; Python {env.get('python')}; SBOM generated {env.get('generated_utc')}."],
817
+ [Table(['Component', 'Version', 'Hash', 'Relationship'], [[c['name'], c['version'], c['hash'], c['relationship']] for c in env.get('components', [])])]))
818
+ n += 1
819
+ for key in proto['sections']:
820
+ rows = [[r['step'], r['check'], r['expected'], r['actual'], r['witness'] or '________'] for r in proto['rows'] if r['section'] == key]
821
+ title_s = next((t for k, _, t in __import__('gea.fat_sat', fromlist=['CLIENT_SECTIONS']).CLIENT_SECTIONS if k == key), key)
822
+ secs.append(Section(str(n), f'Section {key} - {title_s}', [], [Table(['Step', 'Check', 'Expected', 'Actual', 'Witness'], rows)]))
823
+ n += 1
824
+ secs.append(Section(str(n), 'Signatures', [],
825
+ [Table(['Role', 'Name', 'Signature', 'Date'], [['Test engineer (supplier)', '________________', '________________', '__________'],
826
+ ['Witness (client)', '________________', '________________', '__________'],
827
+ ['Approver (client)', '________________', '________________', '__________']])]))
828
+ front = [['Protocol ID', report_id], ['Type', f'{title} (SOW {clause})'], ['Site', site_name or ('factory' if kind == 'FAT' else 'site')],
829
+ ['Program', f'{PROGRAM_NAME}' + (f' build {program_version}' if program_version else '')],
830
+ ['Run (UTC)', f"{proto['started_utc']} to {proto['finished_utc']}"], ['Steps', f"{proto['n_steps']} ({proto['n_pass']} pass / {proto['n_fail']} fail)"],
831
+ ['Result', proto['result']]]
832
+ return Document(title=f'{PROGRAM_NAME} - {title}', report_id=report_id, front=front, sections=secs,
833
+ footer='Each step is executed by the program against its own outputs at run time; the witness column is signed on paper or in the approval workflow.',
834
+ data={'protocol': {k: v for k, v in proto.items() if k != 'environment'}, 'environment': proto.get('environment'),
835
+ 'report_id': report_id, '_records': [], '_catalogue_obj': None})
836
+
837
+
838
+ # ---------------------------------------------------------------------------
839
+ # Renderers
840
+ # ---------------------------------------------------------------------------
841
+ def render_markdown(doc: Document) -> str:
842
+ L: List[str] = [f'# {doc.title}', '']
843
+ L.append('| | |')
844
+ L.append('|---|---|')
845
+ for k, v in doc.front:
846
+ L.append(f'| **{k}** | {v} |')
847
+ L.append('')
848
+ for s in doc.sections:
849
+ L.append(f'## {s.number}. {s.title}')
850
+ L.append('')
851
+ for p in s.paragraphs:
852
+ L.append(p)
853
+ L.append('')
854
+ for t in s.tables:
855
+ if t.caption:
856
+ L.append(f'*{t.caption}*')
857
+ L.append('')
858
+ L.append('| ' + ' | '.join(t.columns) + ' |')
859
+ L.append('|' + '---|' * len(t.columns))
860
+ for r in t.rows:
861
+ L.append('| ' + ' | '.join(_fmt(c).replace('|', '/') for c in r) + ' |')
862
+ L.append('')
863
+ if doc.footer:
864
+ L.append('---')
865
+ L.append(f'*{doc.footer}*')
866
+ L.append('')
867
+ return '\n'.join(L)
868
+
869
+
870
+ _CSS = ('body{font-family:Segoe UI,Arial,sans-serif;margin:2rem auto;max-width:1200px;color:#1a1a1a;'
871
+ 'background:#fff;padding:0 16px}h1{font-size:1.5rem}h2{font-size:1.1rem;margin-top:2rem;'
872
+ 'border-bottom:1px solid #ccc;padding-bottom:.2rem}table{border-collapse:collapse;margin:.5rem 0 1rem;'
873
+ 'font-size:.85rem;width:100%}th,td{border:1px solid #d0d0d0;padding:.3rem .5rem;text-align:left;'
874
+ 'vertical-align:top}th{background:#f2f2f2}caption{text-align:left;font-style:italic;padding:.2rem 0}'
875
+ '.front td:first-child{font-weight:600;width:14rem}.result-bad{color:#a40000;font-weight:700}'
876
+ '.result-ok{color:#0a6b1f;font-weight:700}footer{margin-top:2rem;font-size:.8rem;color:#555}')
877
+
878
+
879
+ def render_html(doc: Document) -> str:
880
+ e = _html.escape
881
+ H: List[str] = ['<!doctype html>', '<html lang="en"><head><meta charset="utf-8">',
882
+ f'<title>{e(doc.title)}</title>', f'<style>{_CSS}</style></head><body>',
883
+ f'<h1>{e(doc.title)}</h1>', '<table class="front">']
884
+ for k, v in doc.front:
885
+ cls = ''
886
+ if k == 'Result':
887
+ cls = ' class="result-bad"' if 'DETECTED' in v and 'NO ' not in v else ' class="result-ok"'
888
+ H.append(f'<tr><td>{e(k)}</td><td{cls}>{e(v)}</td></tr>')
889
+ H.append('</table>')
890
+ for s in doc.sections:
891
+ H.append(f'<h2>{e(s.number)}. {e(s.title)}</h2>')
892
+ for p in s.paragraphs:
893
+ H.append(f'<p>{e(p)}</p>')
894
+ for t in s.tables:
895
+ H.append('<table>')
896
+ if t.caption:
897
+ H.append(f'<caption>{e(t.caption)}</caption>')
898
+ H.append('<tr>' + ''.join(f'<th>{e(c)}</th>' for c in t.columns) + '</tr>')
899
+ for r in t.rows:
900
+ H.append('<tr>' + ''.join(f'<td>{e(_fmt(c))}</td>' for c in r) + '</tr>')
901
+ H.append('</table>')
902
+ if doc.footer:
903
+ H.append(f'<footer>{e(doc.footer)}</footer>')
904
+ H.append('</body></html>')
905
+ return '\n'.join(H)
906
+
907
+
908
+ def forbidden_terms(text: str) -> List[str]:
909
+ low = text.lower()
910
+ return [t for t in FORBIDDEN_TERMS if t in low]
911
+
912
+
913
+ def write(doc: Document, out_dir: str, basename: str = 'gauge_drift_report') -> dict:
914
+ """Write markdown, HTML, the machine JSON, the records CSV and the tag
915
+ catalogue CSV. Returns the paths. Raises if the rendered report contains
916
+ any internal-register term."""
917
+ os.makedirs(out_dir, exist_ok=True)
918
+ md = render_markdown(doc)
919
+ ht = render_html(doc)
920
+ bad = forbidden_terms(md) + forbidden_terms(ht)
921
+ if bad:
922
+ raise ValueError(f'client report contains internal-register terms: {sorted(set(bad))}')
923
+ paths = {'markdown': os.path.join(out_dir, basename + '.md'),
924
+ 'html': os.path.join(out_dir, basename + '.html'),
925
+ 'json': os.path.join(out_dir, basename + '.json'),
926
+ 'records_csv': os.path.join(out_dir, basename + '_records.csv'),
927
+ 'catalogue_csv': os.path.join(out_dir, basename + '_tags.csv')}
928
+ with open(paths['markdown'], 'w', encoding='utf-8', newline='') as f:
929
+ f.write(md)
930
+ with open(paths['html'], 'w', encoding='utf-8', newline='') as f:
931
+ f.write(ht)
932
+ machine = {k: v for k, v in doc.data.items() if not k.startswith('_')}
933
+ with open(paths['json'], 'w', encoding='utf-8') as f:
934
+ json.dump(machine, f, indent=1, default=str)
935
+ if doc.data.get('_records'):
936
+ write_records_csv(doc.data['_records'], paths['records_csv'])
937
+ else:
938
+ paths.pop('records_csv')
939
+ if doc.data.get('_catalogue_obj') is not None:
940
+ write_catalogue_csv(doc.data['_catalogue_obj'], paths['catalogue_csv'])
941
+ else:
942
+ paths.pop('catalogue_csv')
943
+ return paths