gea-program 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. gea/BENCH_TEST_PROTOCOL.md +97 -0
  2. gea/IMPORT_RECORD.md +61 -0
  3. gea/__init__.py +160 -0
  4. gea/__main__.py +661 -0
  5. gea/acceptance_tests.py +1315 -0
  6. gea/accuracy_statement.py +181 -0
  7. gea/alarm_engine.py +307 -0
  8. gea/bench.py +144 -0
  9. gea/blind_harness.py +108 -0
  10. gea/case_study.py +229 -0
  11. gea/catalog/acex_lomonosov_age_depth_model.provenance.json +23 -0
  12. gea/catalog/acex_lomonosov_age_depth_model.txt +28 -0
  13. gea/catalog/agassiz77_canada_temperature.csv +68 -0
  14. gea/catalog/agassiz77_canada_temperature.provenance.json +17 -0
  15. gea/catalog/barbados_110_consolidation.provenance.json +24 -0
  16. gea/catalog/barbados_110_consolidation.txt +91 -0
  17. gea/catalog/bengal_u1452_grain_size.provenance.json +20 -0
  18. gea/catalog/bengal_u1452_grain_size.txt +252 -0
  19. gea/catalog/blake_164_methane_isotopes.provenance.json +22 -0
  20. gea/catalog/blake_164_methane_isotopes.txt +68 -0
  21. gea/catalog/chicxulub_m0077a_pwave_velocity.provenance.json +23 -0
  22. gea/catalog/chicxulub_m0077a_pwave_velocity.txt +735 -0
  23. gea/catalog/collingwood_1_28_ks_complete.las +128 -0
  24. gea/catalog/collingwood_1_28_ks_complete.provenance.json +17 -0
  25. gea/catalog/costa_rica_odp_friction_envelope.provenance.json +24 -0
  26. gea/catalog/costa_rica_odp_friction_envelope.txt +57 -0
  27. gea/catalog/dead_sea_5017_debrite_xrf_ms.provenance.json +21 -0
  28. gea/catalog/dead_sea_5017_debrite_xrf_ms.txt +94 -0
  29. gea/catalog/dsdp_504b_physical_properties.provenance.json +24 -0
  30. gea/catalog/dsdp_504b_physical_properties.txt +82 -0
  31. gea/catalog/dsdp_504b_sound_velocity.provenance.json +21 -0
  32. gea/catalog/dsdp_504b_sound_velocity.txt +81 -0
  33. gea/catalog/elgygytgyn_5011_turbidites.provenance.json +20 -0
  34. gea/catalog/elgygytgyn_5011_turbidites.txt +193 -0
  35. gea/catalog/epica_domec_co2_800kyr.provenance.json +21 -0
  36. gea/catalog/epica_domec_co2_800kyr.txt +265 -0
  37. gea/catalog/fram_909_organic_petrography.provenance.json +22 -0
  38. gea/catalog/fram_909_organic_petrography.txt +40 -0
  39. gea/catalog/gbr_325_coral_uth_ages.provenance.json +23 -0
  40. gea/catalog/gbr_325_coral_uth_ages.txt +71 -0
  41. gea/catalog/gisp2_greenland_temperature.csv +599 -0
  42. gea/catalog/gisp2_greenland_temperature.provenance.json +17 -0
  43. gea/catalog/gom_308_t2p_insitu.provenance.json +21 -0
  44. gea/catalog/gom_308_t2p_insitu.txt +40 -0
  45. gea/catalog/guaymas_385_dom_d13c.provenance.json +21 -0
  46. gea/catalog/guaymas_385_dom_d13c.txt +103 -0
  47. gea/catalog/hikurangi_u1520_friction_insitu.provenance.json +26 -0
  48. gea/catalog/hikurangi_u1520_friction_insitu.txt +74 -0
  49. gea/catalog/hydrate_ridge_204_ncr_hydrate.provenance.json +22 -0
  50. gea/catalog/hydrate_ridge_204_ncr_hydrate.txt +60 -0
  51. gea/catalog/iodp_u1324_pore_pressure.provenance.json +22 -0
  52. gea/catalog/iodp_u1324_pore_pressure.txt +42 -0
  53. gea/catalog/jfast_c0019_slow_slip_events.provenance.json +28 -0
  54. gea/catalog/jfast_c0019_slow_slip_events.txt +35 -0
  55. gea/catalog/kennetcook_2_p129_excerpt.las +139 -0
  56. gea/catalog/kennetcook_2_p129_excerpt.provenance.json +17 -0
  57. gea/catalog/ktb_hb_bhgm_density.dat +227 -0
  58. gea/catalog/ktb_hb_bhgm_density.provenance.json +22 -0
  59. gea/catalog/ktb_hb_complog_6020_excerpt.provenance.json +20 -0
  60. gea/catalog/ktb_hb_complog_6020_excerpt.txt +72 -0
  61. gea/catalog/ktb_hb_hlog246_temperature.dat +1636 -0
  62. gea/catalog/ktb_hb_hlog246_temperature.provenance.json +22 -0
  63. gea/catalog/ktb_hb_rockmech_compress.dat +33 -0
  64. gea/catalog/ktb_hb_rockmech_compress.provenance.json +22 -0
  65. gea/catalog/ktb_hb_tvd_0_9080_excerpt.dat +2817 -0
  66. gea/catalog/ktb_hb_tvd_0_9080_excerpt.provenance.json +20 -0
  67. gea/catalog/ktb_vb_rockmech_compress.dat +125 -0
  68. gea/catalog/ktb_vb_rockmech_compress.provenance.json +22 -0
  69. gea/catalog/ktb_vb_vlog251_temperature.dat +1127 -0
  70. gea/catalog/ktb_vb_vlog251_temperature.provenance.json +24 -0
  71. gea/catalog/l06_06_nl_survey.csv +201 -0
  72. gea/catalog/l06_06_nl_survey.provenance.json +17 -0
  73. gea/catalog/l07_01_nl_excerpt.las +90 -0
  74. gea/catalog/l07_01_nl_excerpt.provenance.json +17 -0
  75. gea/catalog/mariana_1200_serpentinite_geochem.provenance.json +21 -0
  76. gea/catalog/mariana_1200_serpentinite_geochem.txt +63 -0
  77. gea/catalog/med_160_sapropels.provenance.json +22 -0
  78. gea/catalog/med_160_sapropels.txt +43 -0
  79. gea/catalog/nankai_megasplay_shear_strength.provenance.json +26 -0
  80. gea/catalog/nankai_megasplay_shear_strength.txt +42 -0
  81. gea/catalog/odp_1027b_thermal_conductivity.provenance.json +21 -0
  82. gea/catalog/odp_1027b_thermal_conductivity.txt +52 -0
  83. gea/catalog/odp_1027c_cork_temperature.provenance.json +20 -0
  84. gea/catalog/odp_1027c_cork_temperature.txt +26 -0
  85. gea/catalog/odp_1165b_thermal_conductivity.provenance.json +22 -0
  86. gea/catalog/odp_1165b_thermal_conductivity.txt +102 -0
  87. gea/catalog/odp_1274a_mantle_peridotite_mad.provenance.json +27 -0
  88. gea/catalog/odp_1274a_mantle_peridotite_mad.txt +44 -0
  89. gea/catalog/odp_504b_dike_elastic_moduli.provenance.json +27 -0
  90. gea/catalog/odp_504b_dike_elastic_moduli.txt +85 -0
  91. gea/catalog/odp_504b_leg137_borehole_fluids.provenance.json +19 -0
  92. gea/catalog/odp_504b_leg137_borehole_fluids.txt +68 -0
  93. gea/catalog/odp_735b_gabbro_elastic_moduli.provenance.json +28 -0
  94. gea/catalog/odp_735b_gabbro_elastic_moduli.txt +127 -0
  95. gea/catalog/peru_201_sulfate_reduction.provenance.json +22 -0
  96. gea/catalog/peru_201_sulfate_reduction.txt +322 -0
  97. gea/catalog/scorpio_e1_sa_excerpt.las +113 -0
  98. gea/catalog/scorpio_e1_sa_excerpt.provenance.json +17 -0
  99. gea/catalog/sumatra_362_cohesion.provenance.json +21 -0
  100. gea/catalog/sumatra_362_cohesion.txt +38 -0
  101. gea/catalog/university_6_17_no1_tx_excerpt.las +119 -0
  102. gea/catalog/university_6_17_no1_tx_excerpt.provenance.json +17 -0
  103. gea/catalog/ursa_308_xrd_mineralogy.provenance.json +21 -0
  104. gea/catalog/ursa_308_xrd_mineralogy.txt +46 -0
  105. gea/catalog/volve_15_9_19_sr_excerpt.las +183 -0
  106. gea/catalog/volve_15_9_19_sr_excerpt.provenance.json +16 -0
  107. gea/catalog/volve_15_9_19a_core_excerpt.csv +88 -0
  108. gea/catalog/volve_15_9_19a_core_excerpt.provenance.json +18 -0
  109. gea/catalog/volve_f12_f14_production_excerpt.csv +167 -0
  110. gea/catalog/volve_f12_f14_production_excerpt.provenance.json +21 -0
  111. gea/catalog/walvis_208_petm_carbonate.provenance.json +21 -0
  112. gea/catalog/walvis_208_petm_carbonate.txt +268 -0
  113. gea/catalog/woodlark_1109_rock_eval.provenance.json +23 -0
  114. gea/catalog/woodlark_1109_rock_eval.txt +30 -0
  115. gea/cli.py +125 -0
  116. gea/client_reports.py +943 -0
  117. gea/config_versioning.py +133 -0
  118. gea/correlation.py +155 -0
  119. gea/dashboard.py +390 -0
  120. gea/deviation.py +70 -0
  121. gea/downhole_engine.py +395 -0
  122. gea/drift_monitor.py +310 -0
  123. gea/earth_model.py +230 -0
  124. gea/example_register_map.json +14 -0
  125. gea/fat_sat.py +68 -0
  126. gea/follower.py +98 -0
  127. gea/forward_model.py +139 -0
  128. gea/gamma.py +176 -0
  129. gea/gauge_specs.py +112 -0
  130. gea/gravity_reference.py +116 -0
  131. gea/inverse_engine.py +215 -0
  132. gea/matplotlib_demo.py +85 -0
  133. gea/modbus.py +229 -0
  134. gea/model_card.py +248 -0
  135. gea/operator_app.py +442 -0
  136. gea/ports.py +340 -0
  137. gea/profile_catalog.py +773 -0
  138. gea/project.py +213 -0
  139. gea/qt6_downhole_app.py +144 -0
  140. gea/quartz_hpht_extension.py +152 -0
  141. gea/reconciler.py +206 -0
  142. gea/rock_inventory.py +404 -0
  143. gea/sample_record.py +430 -0
  144. gea/sample_well_profile.csv +15 -0
  145. gea/sbom.py +116 -0
  146. gea/segy.py +181 -0
  147. gea/service_life.py +173 -0
  148. gea/shell.py +107 -0
  149. gea/sla_report.py +199 -0
  150. gea/store_forward.py +234 -0
  151. gea/strata_join.py +186 -0
  152. gea/survey_cmd.py +264 -0
  153. gea/survey_view.py +138 -0
  154. gea/telemetry.py +306 -0
  155. gea/tool_library.py +260 -0
  156. gea/well_assembler.py +457 -0
  157. gea/well_test_validation.py +369 -0
  158. gea_program-0.1.0.dist-info/METADATA +138 -0
  159. gea_program-0.1.0.dist-info/RECORD +163 -0
  160. gea_program-0.1.0.dist-info/WHEEL +5 -0
  161. gea_program-0.1.0.dist-info/entry_points.txt +2 -0
  162. gea_program-0.1.0.dist-info/licenses/LICENSE +373 -0
  163. gea_program-0.1.0.dist-info/top_level.txt +1 -0
gea/profile_catalog.py ADDED
@@ -0,0 +1,773 @@
1
+ # This Source Code Form is subject to the terms of the Mozilla Public
2
+ # License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ # file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ """profile_catalog — the well-profile catalogue (v1.12.0 extension).
5
+
6
+ Daniel GO 2026-08-24: build a catalogue of well profiles from public
7
+ geophysical databases, so the closed stream can run on REAL wells instead of
8
+ one synthetic sample. Three parts:
9
+
10
+ 1. `PROFILE_SOURCES` — the machine-readable table of public databases
11
+ (what each offers, direct entry points, license, access barriers stated
12
+ honestly: several serve only ZIPs or need registration, which this
13
+ environment cannot fetch — those are documented pull-it-yourself paths).
14
+ 2. `CATALOG` — shipped entries. Every entry has a MANDATORY provenance
15
+ sidecar (.provenance.json) naming the source database, well, URL,
16
+ license, fetch date, and coverage. First real entry:
17
+ **Equinor Volve well 15/9-19 SR** (verbatim excerpt, CC/Equinor open
18
+ licence, disclosed coverage) — real third-party well-log data flowing
19
+ the las2 port end-to-end.
20
+ 3. `las_to_profile()` — the converter: a LAS LiveStream becomes the
21
+ engine's profile CSV (depth_ft,pressure_psi,temp_F). Where the log
22
+ carries no temperature/pressure curves (most composites do not), the
23
+ converter fills them from DECLARED gradients and stamps the output
24
+ `derivation: DERIVED_GRADIENTS` — a profile built from a real
25
+ trajectory with derived conditions is useful and honest ONLY when
26
+ labeled (Rule 7); measured-curve conversion is used automatically when
27
+ the curves exist.
28
+
29
+ Headless-safe: numpy + stdlib.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import json
35
+ from dataclasses import dataclass
36
+ from pathlib import Path
37
+ from typing import Dict, Optional
38
+
39
+ import numpy as np
40
+
41
+ from .ports import LiveStream, read_las
42
+
43
+ _CATALOG_DIR = Path(__file__).parent / "catalog"
44
+
45
+ # Temperature/pressure curve mnemonics accepted as MEASURED (LAS conventions)
46
+ _TEMP_MNEMONICS = ('TEMP', 'TEMPERATURE', 'BHT', 'WTEP', 'MRT', 'DTEMP', 'TMP')
47
+ _PRES_MNEMONICS = ('PRES', 'PRESSURE', 'WPRE', 'BHP', 'PFOR')
48
+
49
+ M_TO_FT = 3.28084
50
+
51
+
52
+ # ---------------------------------------------------------------------------
53
+ # 1) The public-source table (honest access notes)
54
+ # ---------------------------------------------------------------------------
55
+ PROFILE_SOURCES: Dict[str, dict] = {
56
+ 'kgs': {
57
+ 'name': 'Kansas Geological Survey LAS database',
58
+ 'url': 'https://www.kgs.ku.edu/Magellan/Logs/',
59
+ 'offers': '21,000+ digital wireline logs (LAS), free, no registration',
60
+ 'license': 'public state archive',
61
+ 'access': 'individual downloads are ZIPPED via the search app; yearly bulk ZIPs; download and unzip locally, then ingest via las2'},
62
+ 'volve': {
63
+ 'name': 'Equinor Volve open dataset',
64
+ 'url': 'https://www.equinor.com/energy/volve-data-sharing',
65
+ 'offers': 'complete real North Sea field: logs, surveys, production (~40,000 files)',
66
+ 'license': 'Equinor Open Data Licence (attribution)',
67
+ 'access': 'registration required for the full archive; some files publicly redistributed (see catalogue entry volve_15_9_19_sr_excerpt)'},
68
+ 'gdr_forge': {
69
+ 'name': 'DOE Geothermal Data Repository - Utah FORGE',
70
+ 'url': 'https://gdr.openei.org/submissions/1326',
71
+ 'offers': 'REAL downhole T/P logs (wells 58-32, 56-32, 78-32; June 2021 update), drilling data, surveys; DOI 10.15121/1812334',
72
+ 'license': 'CC BY 4.0',
73
+ 'access': 'T/P logs served as ZIPs (server marks all files octet-stream); download and unzip locally, then ingest the contained .las/.csv'},
74
+ 'nlog': {
75
+ 'name': 'NLOG (Netherlands Oil and Gas portal)',
76
+ 'url': 'https://www.nlog.nl/en',
77
+ 'offers': 'thousands of onshore/offshore wells: logs, deviation, production',
78
+ 'license': 'open by mandate',
79
+ 'access': 'per-well downloads; formats vary'},
80
+ 'state_regulators': {
81
+ 'name': 'US state regulators (TX RRC, ND NDIC, OK OCC, CO ECMC, WY OGCC)',
82
+ 'url': 'https://www.rrc.texas.gov/ (and peers)',
83
+ 'offers': 'well files: directional surveys, pressure tests, BHT reports',
84
+ 'license': 'public regulatory archives',
85
+ 'access': 'per-state portals; mostly PDF/scans plus some digital data'},
86
+ 'offshore_national': {
87
+ 'name': 'BOEM/BSEE (US offshore), UK NSTA NDR, Australia NOPIMS',
88
+ 'url': 'https://www.data.boem.gov/ ; https://ndr.nstauthority.co.uk/ ; https://nopims.disr.gov.au/',
89
+ 'offers': 'national open repositories: surveys, logs, completions',
90
+ 'license': 'open national archives',
91
+ 'access': 'portal downloads; registration varies'},
92
+ }
93
+
94
+
95
+ # ---------------------------------------------------------------------------
96
+ # 2) The shipped catalogue (provenance mandatory)
97
+ # ---------------------------------------------------------------------------
98
+ def read_temperature_csv(path) -> LiveStream:
99
+ """Ingest a temperature-profile CSV (header `d,t`: depth in metres,
100
+ temperature in degC — the GEUS ice-borehole database format) as a
101
+ depth-indexed LiveStream with a TEMP channel. The catalogue's non-LAS
102
+ entry path (v1.16.0, driven by the GISP2 prize well)."""
103
+ import csv as _csv
104
+ p = Path(path)
105
+ d, t = [], []
106
+ with p.open(newline='', encoding='utf-8') as f:
107
+ for row in _csv.DictReader(f):
108
+ d.append(float(row['d']))
109
+ t.append(float(row['t']))
110
+ from .ports import StreamChannel
111
+ return LiveStream(name=p.stem, source_format='temperature_csv',
112
+ index_kind='depth', index=np.array(d, dtype=float),
113
+ channels={'TEMP': StreamChannel(name='TEMP', unit='DEGC',
114
+ values=np.array(t, dtype=float))},
115
+ meta={'format': 'GEUS d,t temperature profile'})
116
+
117
+
118
+ def read_survey_csv(path):
119
+ """Ingest a deviation-survey CSV (NLOG-style long headers: Depth /
120
+ TrueVertical Depth, with inclination/azimuth/offsets alongside) as a
121
+ DeviationSurvey - MD->TVD taken DIRECTLY from the measured columns, no
122
+ minimum-curvature reconstruction needed. The catalogue's survey-entry
123
+ path (v1.19.0, driven by the real L06-06 trajectory)."""
124
+ import csv as _csv
125
+ from .deviation import DeviationSurvey
126
+ p = Path(path)
127
+ md, tvd = [], []
128
+ with p.open(newline='', encoding='utf-8') as f:
129
+ reader = _csv.DictReader(f)
130
+ md_col = next(c for c in reader.fieldnames if c.strip().lower() in ('depth', 'md', 'md_ft'))
131
+ tvd_col = next(c for c in reader.fieldnames
132
+ if 'truevertical' in c.strip().lower().replace(' ', '')
133
+ or c.strip().lower() in ('tvd', 'tvd_ft'))
134
+ for row in reader:
135
+ md.append(float(row[md_col]))
136
+ tvd.append(float(row[tvd_col]))
137
+ return DeviationSurvey(md_ft=md, tvd_ft=tvd, name=p.stem)
138
+
139
+
140
+ def read_core_csv(path) -> LiveStream:
141
+ """Ingest a conventional core-analysis CSV (Volve-style header:
142
+ DEPTH,OrigDepth,CORE_NO,SAMPLE,...) as a depth-indexed LiveStream whose
143
+ channels are the numeric lab columns (permeability/porosity/saturations/
144
+ grain density; blanks -> NaN). Laboratory ground truth alongside logs
145
+ (v1.20.0, driven by the real 15/9-19 A core data)."""
146
+ import csv as _csv
147
+ from .ports import StreamChannel
148
+ p = Path(path)
149
+ with p.open(newline='', encoding='utf-8') as f:
150
+ reader = _csv.DictReader(f)
151
+ cols = [c for c in reader.fieldnames if c != 'DEPTH']
152
+ depth, data = [], {c: [] for c in cols}
153
+ for row in reader:
154
+ depth.append(float(row['DEPTH']))
155
+ for c in cols:
156
+ v = (row.get(c) or '').strip()
157
+ data[c].append(float(v) if v else np.nan)
158
+ return LiveStream(name=p.stem, source_format='core_csv',
159
+ index_kind='depth', index=np.array(depth, dtype=float),
160
+ channels={c: StreamChannel(name=c, unit='', values=np.array(data[c]))
161
+ for c in cols},
162
+ meta={'format': 'conventional core analysis (units per provenance)'})
163
+
164
+
165
+ def read_production_csv(path) -> LiveStream:
166
+ """Ingest a daily production-history CSV (Volve-style header:
167
+ DATEPRD,WELL_BORE_CODE,...) as a TIME-indexed LiveStream - the
168
+ catalogue's first time-indexed kind (v1.21.0, driven by the real
169
+ 15/9-F-12/F-14 daily records). Index = elapsed seconds from the first
170
+ date (86400 s cadence); channels are the per-well numeric operational
171
+ columns, namespaced COL[well]; blanks and absent dates -> NaN. Units are
172
+ NOT in the header - carried per the provenance data dictionary."""
173
+ import csv as _csv
174
+ from datetime import date as _date
175
+ from .ports import StreamChannel
176
+ p = Path(path)
177
+ numeric = ('ON_STREAM_HRS', 'AVG_DOWNHOLE_PRESSURE', 'AVG_DOWNHOLE_TEMPERATURE',
178
+ 'AVG_DP_TUBING', 'AVG_ANNULUS_PRESS', 'AVG_CHOKE_SIZE_P', 'AVG_WHP_P',
179
+ 'AVG_WHT_P', 'DP_CHOKE_SIZE', 'BORE_OIL_VOL', 'BORE_GAS_VOL',
180
+ 'BORE_WAT_VOL', 'BORE_WI_VOL')
181
+ units = {'ON_STREAM_HRS': 'h', 'AVG_DOWNHOLE_PRESSURE': 'bar',
182
+ 'AVG_DOWNHOLE_TEMPERATURE': 'degC', 'AVG_DP_TUBING': 'bar',
183
+ 'AVG_ANNULUS_PRESS': 'bar', 'AVG_CHOKE_SIZE_P': 'pct',
184
+ 'AVG_WHP_P': 'bar', 'AVG_WHT_P': 'degC', 'DP_CHOKE_SIZE': 'bar',
185
+ 'BORE_OIL_VOL': 'Sm3', 'BORE_GAS_VOL': 'Sm3', 'BORE_WAT_VOL': 'Sm3',
186
+ 'BORE_WI_VOL': 'Sm3'}
187
+ rows = []
188
+ with p.open(newline='', encoding='utf-8') as f:
189
+ for row in _csv.DictReader(f):
190
+ rows.append(row)
191
+ if not rows:
192
+ raise ValueError(f"production CSV {p.name}: no records")
193
+ dates = sorted({r['DATEPRD'] for r in rows})
194
+ wells = []
195
+ for r in rows:
196
+ w = r['NPD_WELL_BORE_NAME']
197
+ if w not in wells:
198
+ wells.append(w)
199
+ d0 = _date.fromisoformat(dates[0])
200
+ idx = np.array([( _date.fromisoformat(d) - d0).days * 86400.0 for d in dates])
201
+ pos = {d: i for i, d in enumerate(dates)}
202
+ channels = {}
203
+ for w in wells:
204
+ grids = {c: np.full(len(dates), np.nan) for c in numeric}
205
+ for r in rows:
206
+ if r['NPD_WELL_BORE_NAME'] != w:
207
+ continue
208
+ i = pos[r['DATEPRD']]
209
+ for c in numeric:
210
+ v = (r.get(c) or '').strip()
211
+ if v:
212
+ grids[c][i] = float(v)
213
+ for c in numeric:
214
+ channels[f"{c}[{w}]"] = StreamChannel(
215
+ name=f"{c}[{w}]", unit=units[c] + ' (per provenance dictionary)',
216
+ values=grids[c])
217
+ return LiveStream(name=p.stem, source_format='production_csv',
218
+ index_kind='time_s', index=idx, channels=channels,
219
+ meta={'format': 'daily production history (per-well, namespaced channels)',
220
+ 'start_date': dates[0], 'end_date': dates[-1],
221
+ 'wells': ';'.join(wells), 'cadence_s': '86400',
222
+ 'units_note': 'units interpretive per provenance data dictionary, not in-file'})
223
+
224
+
225
+ def read_ktb_dat(path) -> LiveStream:
226
+ """Ingest a KTB Information System temperature-log file ('!'-comment
227
+ header + space-separated DEPT TMP3 HTEN MRES rows) as a depth-indexed
228
+ LiveStream - the catalogue's first HOT-regime temperature dialect
229
+ (v1.22.0, driven by the real KTB-HB hlog246). Header lines are carried
230
+ into meta (well name, log date, time-since-circulation fields - the
231
+ disturbed-log disclosure lives in the data itself)."""
232
+ import re as _re
233
+ p = Path(path)
234
+ hdr, rows, cols = [], [], []
235
+ col_re = _re.compile(r'^!\s+\d+\s+"(\w+)\s[^"]*"\s+F\d*\s+(\S+)')
236
+ with p.open(encoding='utf-8') as f:
237
+ for line in f:
238
+ line = line.rstrip('\n')
239
+ if not line.strip():
240
+ continue
241
+ if line.lstrip().startswith('!'):
242
+ hdr.append(line)
243
+ m = col_re.match(line.strip())
244
+ if m:
245
+ cols.append((m.group(1), m.group(2)))
246
+ continue
247
+ parts = line.split()
248
+ if cols and len(parts) == len(cols):
249
+ try:
250
+ rows.append([float(x) for x in parts])
251
+ except ValueError:
252
+ continue
253
+ if not cols:
254
+ raise ValueError(f"KTB log {p.name}: no column-definition block in header")
255
+ if not rows:
256
+ raise ValueError(f"KTB log {p.name}: no data rows")
257
+ import numpy as _np
258
+ arr = _np.array(rows, dtype=float)
259
+ meta = {'format': 'KTB Information System temperature log (disturbed mud-temperature log, per provenance)'}
260
+ val_re = {'well': _re.compile(r'"WN\s+UNAL\s+.*?\s{3,}(\S[^"]*)"'),
261
+ 'log_date': _re.compile(r'"DATE\s+UNAL\s+.*?\s{3,}(\S[^"]*)"'),
262
+ 'time_logger_at_bottom': _re.compile(r'"TLAB\s+UNAL\s+Time Logger At Bottom\s{3,}(\S[^"]*)"'),
263
+ 'time_circulation_stopped': _re.compile(r'"TCS\s+UNAL\s+Time Circulation Stopped\s{3,}(\S[^"]*)"')}
264
+ for h in hdr:
265
+ for key, rx in val_re.items():
266
+ m = rx.search(h)
267
+ if m and key not in meta:
268
+ meta[key] = m.group(1).strip()
269
+ from .ports import StreamChannel
270
+ return LiveStream(name=p.stem, source_format='ktb_dat',
271
+ index_kind='depth', index=arr[:, 0],
272
+ channels={n: StreamChannel(name=n, unit=u, values=arr[:, i + 1])
273
+ for i, (n, u) in enumerate(cols[1:], start=0)},
274
+ meta=meta)
275
+
276
+
277
+ def read_ktb_table(path) -> LiveStream:
278
+ """Ingest a KTB Information System TYPED table ('!'-header declaring
279
+ F/C/I columns, e.g. the rock-mechanics compressive-strength tables) as a
280
+ depth-indexed LiveStream (v1.26.0, driven by the real VB core-strength
281
+ file). Numeric (F/I) columns become channels; C-typed string columns
282
+ stay verbatim in the file (ROCK TYPE is carried as per-sample quality
283
+ on the strength channel). Rendering-collapsed tabs make some short rows
284
+ ambiguous: a trailing decimal token is assigned by DECLARED TYPE (an I2
285
+ dip cannot hold a decimal); a trailing integer token that could be
286
+ either column is REFUSED - NaN + a quality flag with the raw token."""
287
+ import re as _re
288
+ p = Path(path)
289
+ col_re = _re.compile(r'^!\s+\d+\s+"([^"]+)"\s+([FCI])\d*\s*(\S*)')
290
+ tok_re = _re.compile(r'"[^"]*"|\S+')
291
+ cols, rows = [], []
292
+ with p.open(encoding='utf-8') as f:
293
+ for line in f:
294
+ line = line.rstrip('\n')
295
+ if not line.strip():
296
+ continue
297
+ if line.lstrip().startswith('!'):
298
+ m = col_re.match(line.strip())
299
+ if m:
300
+ cols.append((m.group(1).replace(' ', '_'), m.group(2), m.group(3)))
301
+ continue
302
+ toks = tok_re.findall(line)
303
+ if len(toks) >= 6:
304
+ rows.append(toks)
305
+ if not cols or not rows:
306
+ raise ValueError(f"KTB table {p.name}: no typed column block or no rows")
307
+ import numpy as _np
308
+ n = len(rows)
309
+ names = [c[0] for c in cols]
310
+ num_idx = [i for i, c in enumerate(cols) if c[1] in ('F', 'I')]
311
+ grids = {names[i]: _np.full(n, _np.nan) for i in num_idx if i > 0}
312
+ depth = _np.full(n, _np.nan)
313
+ rock = ['' for _ in range(n)]
314
+ flags = ['' for _ in range(n)]
315
+ rock_col = next((i for i, c in enumerate(cols) if 'ROCK' in c[0]), None)
316
+ for r, toks in enumerate(rows):
317
+ if len(toks) == len(cols):
318
+ assign = list(enumerate(toks))
319
+ else:
320
+ assign = list(enumerate(toks[:6]))
321
+ trail = toks[6:]
322
+ if len(trail) == 1:
323
+ if '.' in trail[0]:
324
+ assign.append((6, trail[0]))
325
+ else:
326
+ flags[r] = f"AMBIGUOUS_TRAILING:{trail[0]}"
327
+ for ci, tok in assign:
328
+ name, typ = cols[ci][0], cols[ci][1]
329
+ if ci == 0:
330
+ depth[r] = float(tok)
331
+ elif typ in ('F', 'I'):
332
+ v = tok.strip().strip('"')
333
+ if v:
334
+ grids[name][r] = float(v)
335
+ elif ci == rock_col:
336
+ rock[r] = tok.strip('"')
337
+ from .ports import StreamChannel
338
+ channels = {}
339
+ for i in num_idx:
340
+ if i == 0:
341
+ continue
342
+ name, unit = names[i], cols[i][2]
343
+ q = rock if 'STRENGTH' in name else (flags if name == 'E_MODUL' else None)
344
+ channels[name] = StreamChannel(name=name, unit=unit, values=grids[name],
345
+ quality=list(q) if q else None)
346
+ return LiveStream(name=p.stem, source_format='ktb_table',
347
+ index_kind='depth', index=depth, channels=channels,
348
+ meta={'format': 'KTB typed table (rock mechanics); string columns verbatim in file',
349
+ 'ambiguous_rows_refused': str(sum(1 for x in flags if x))})
350
+
351
+
352
+ def read_operator_table(path) -> LiveStream:
353
+ """Ingest a verbatim OPERATOR TABLE TRANSCRIPTION (v1.72.0): field-data
354
+ tables recovered from operator report screenshots/exports, transcribed
355
+ cell-for-cell. Format: /* OPERATOR TABLE TRANSCRIPTION */ header
356
+ (Key:<TAB>Value lines incl. IndexKind: depth|ordinal) then a TSV table.
357
+ Rows whose index cell is non-numeric (section markers like 'Curve',
358
+ 'Lateral') are carried verbatim into meta['marker_rows'] with their
359
+ position. Text columns ride in meta as row-aligned lists; numeric columns
360
+ become channels; nothing is recomputed at ingest."""
361
+ p = Path(path)
362
+ txt = p.read_text(encoding='utf-8')
363
+ head, _, body = txt.partition('*/')
364
+ meta = {}
365
+ for line in head.splitlines():
366
+ if ':\t' in line:
367
+ k, _, v = line.partition(':\t')
368
+ meta[k.strip('/* ').strip().lower()] = v.strip()
369
+ lines = [l for l in body.strip('\n').split('\n') if l]
370
+ cols = lines[0].split('\t')
371
+ raw = [l.split('\t') for l in lines[1:]]
372
+ markers, rows = [], []
373
+ ordinal = meta.get('indexkind') == 'ordinal'
374
+ for i, r in enumerate(raw):
375
+ if ordinal:
376
+ rows.append(r) # ordinal tables: col 0 may be text (timestamps)
377
+ continue
378
+ try:
379
+ float(r[0])
380
+ rows.append(r)
381
+ except ValueError:
382
+ markers.append((i, '\t'.join(r).strip()))
383
+ if markers:
384
+ meta['marker_rows'] = '; '.join('row %d: %s' % m for m in markers)
385
+ from .ports import StreamChannel
386
+ index_kind = meta.get('indexkind', 'depth')
387
+ if ordinal:
388
+ index = np.arange(1, len(rows) + 1, dtype=float)
389
+ start = 0
390
+ else:
391
+ index = np.array([float(r[0]) for r in rows], dtype=float)
392
+ start = 1
393
+ channels = {}
394
+ for c in range(start, len(cols)):
395
+ vals, numeric = [], 0
396
+ for r in rows:
397
+ cell = r[c].strip() if c < len(r) else ''
398
+ try:
399
+ vals.append(float(cell))
400
+ numeric += 1
401
+ except ValueError:
402
+ vals.append(float('nan'))
403
+ if numeric:
404
+ channels[cols[c]] = StreamChannel(name=cols[c], unit='',
405
+ values=np.array(vals, dtype=float))
406
+ else:
407
+ meta['textcol_' + cols[c]] = [r[c].strip() if c < len(r) else ''
408
+ for r in rows]
409
+ return LiveStream(name=p.stem, source_format='operator_table',
410
+ index_kind=('depth' if index_kind != 'ordinal' else 'ordinal'),
411
+ index=index, channels=channels, meta=meta)
412
+
413
+
414
+ def read_drift_xls(path) -> LiveStream:
415
+ """Ingest a directional-drilling drift/survey XLS export (v1.71.0, driven
416
+ by the first OPERATOR-tier entry: the Retama Ranch #403H 183-station
417
+ survey). Header row names MD / Inclination / Azimuth / TVD / NS / EW
418
+ (vendor exports interleave blank columns; they are skipped). Depth index =
419
+ MD [ft]; every named numeric column becomes a channel. Verbatim: cells are
420
+ read as exported, nothing is recomputed or smoothed at ingest."""
421
+ try:
422
+ import xlrd as _xlrd
423
+ except ImportError as _e:
424
+ raise ImportError(
425
+ "read_drift_xls needs the optional third-party module 'xlrd' "
426
+ "(pip install xlrd). No SHIPPED catalogue entry requires it - "
427
+ "operator drift surveys are stored in the dependency-free "
428
+ "operator-table format since the v0.406.0 ship-rehearsal catch; "
429
+ "this reader exists for ingesting NEW vendor .xls drops only."
430
+ ) from _e
431
+ p = Path(path)
432
+ wb = _xlrd.open_workbook(str(p))
433
+ sh = wb.sheet_by_index(0)
434
+ header = [str(sh.cell_value(0, c)).strip() for c in range(sh.ncols)]
435
+ cols = [(c, h) for c, h in enumerate(header) if h]
436
+ md_c = next(c for c, h in cols if h.lower().startswith('md'))
437
+ from .ports import StreamChannel
438
+ md, rows = [], []
439
+ for r in range(1, sh.nrows):
440
+ try:
441
+ md.append(float(sh.cell_value(r, md_c)))
442
+ except (TypeError, ValueError):
443
+ continue
444
+ rows.append(r)
445
+ channels = {}
446
+ for c, h in cols:
447
+ if c == md_c:
448
+ continue
449
+ vals = []
450
+ for r in rows:
451
+ try:
452
+ vals.append(float(sh.cell_value(r, c)))
453
+ except (TypeError, ValueError):
454
+ vals.append(float('nan'))
455
+ channels[h] = StreamChannel(name=h, unit='', values=np.array(vals, dtype=float))
456
+ return LiveStream(name=p.stem, source_format='drift_xls', index_kind='depth',
457
+ index=np.array(md, dtype=float), channels=channels,
458
+ meta={'sheet': sh.name, 'stations': str(len(md))})
459
+
460
+
461
+ def read_iodp_table(path) -> LiveStream:
462
+ """Ingest a verbatim IODP Proceedings data-report table transcription
463
+ (v1.70.0, driven by Exp 308 Table T2 - the in situ temperature AND
464
+ pressure penetrometer results that made U1324 the catalogue's first
465
+ measured-T+P site). File format: a /* IODP TABLE TRANSCRIPTION */ header
466
+ (citation, source URL, license, verbatim table notes) then a tab-separated
467
+ table whose cells are carried verbatim. Numeric columns become channels;
468
+ cells that are blank or hold the T2P dual-port 'a; b' pairs become NaN in
469
+ the channel (the verbatim cell stays in the file - the reader never
470
+ repairs); the Hole column rides row-aligned in meta['hole'] so assemblies
471
+ can filter one site out of a multi-site table without touching the
472
+ archive. Depth index = the first column whose name contains 'mbsf'."""
473
+ p = Path(path)
474
+ txt = p.read_text(encoding='utf-8')
475
+ head, _, body = txt.partition('*/')
476
+ meta = {}
477
+ for line in head.splitlines():
478
+ if ':\t' in line:
479
+ k, _, v = line.partition(':\t')
480
+ meta[k.strip('/* ').strip().lower()] = v.strip()
481
+ lines = [l for l in body.strip('\n').split('\n') if l]
482
+ cols = lines[0].split('\t')
483
+ rows = [l.split('\t') for l in lines[1:]]
484
+ depth_i = next(i for i, c in enumerate(cols) if 'mbsf' in c.lower())
485
+ hole_i = next((i for i, c in enumerate(cols) if c.strip().lower() == 'hole'), None)
486
+ from .ports import StreamChannel
487
+ channels = {}
488
+ for i, c in enumerate(cols):
489
+ if i in (depth_i, hole_i):
490
+ continue
491
+ vals = []
492
+ numeric = 0
493
+ for r in rows:
494
+ cell = r[i].strip() if i < len(r) else ''
495
+ try:
496
+ vals.append(float(cell))
497
+ numeric += 1
498
+ except ValueError:
499
+ vals.append(float('nan'))
500
+ if numeric:
501
+ channels[c] = StreamChannel(name=c, unit='', values=np.array(vals, dtype=float))
502
+ if hole_i is not None:
503
+ meta['hole'] = [r[hole_i].strip() for r in rows]
504
+ return LiveStream(name=p.stem, source_format='iodp_table', index_kind='depth',
505
+ index=np.array([float(r[depth_i]) for r in rows], dtype=float),
506
+ channels=channels, meta=meta)
507
+
508
+
509
+ def read_pangaea_txt(path) -> LiveStream:
510
+ """Ingest a PANGAEA machine-readable textfile export (self-describing
511
+ '/* DATA DESCRIPTION */' header + tab-separated matrix) as a
512
+ depth-indexed LiveStream (v1.28.0, driven by the real ODP 504B borehole
513
+ -fluid dataset). The header's citation, license and coordinates go to
514
+ meta; numeric columns become channels (units parsed from '[...]');
515
+ short rows pad to NaN; the first non-numeric column rides as per-sample
516
+ quality on the first channel."""
517
+ import re as _re
518
+ p = Path(path)
519
+ text = p.open(encoding='utf-8').read()
520
+ if not text.startswith('/* DATA DESCRIPTION'):
521
+ raise ValueError(f"{p.name}: not a PANGAEA textfile export")
522
+ head, _, body = text.partition('*/')
523
+ meta = {'format': 'PANGAEA textfile export (self-describing header verbatim in file)'}
524
+ m = _re.search(r'Citation:\t([^\n]+)', head)
525
+ if m:
526
+ meta['citation'] = m.group(1).strip().rstrip(',')
527
+ m = _re.search(r'License:\t([^\n]+)', head)
528
+ if m:
529
+ meta['license'] = m.group(1).strip()
530
+ m = _re.search(r'LATITUDE:\s*(-?[\d.]+)\s*\*\s*LONGITUDE:\s*(-?[\d.]+)', head)
531
+ if m:
532
+ meta['latitude'], meta['longitude'] = m.group(1), m.group(2)
533
+ m = _re.search(r'ELEVATION:\s*(-?[\d.]+)', head)
534
+ if m:
535
+ meta['elevation_m'] = m.group(1)
536
+ lines = [l for l in body.split('\n') if l.strip()]
537
+ headers = lines[0].split('\t')
538
+ rows = [l.split('\t') for l in lines[1:]]
539
+ import numpy as _np
540
+ depth_i = next((i for i, h in enumerate(headers) if h.lower().startswith('depth')), None)
541
+ index_kind = 'depth'
542
+ def val(r, i):
543
+ v = r[i].strip() if i < len(r) else ''
544
+ try:
545
+ return float(v) if v else _np.nan
546
+ except ValueError:
547
+ return None
548
+ if depth_i is None:
549
+ index_kind = 'ordinal'
550
+ depth = _np.arange(1, len(rows) + 1, dtype=float)
551
+ else:
552
+ depth = _np.array([val(r, depth_i) for r in rows], dtype=float)
553
+ from .ports import StreamChannel
554
+ channels = {}
555
+ label_col = None
556
+ for i, h in enumerate(headers):
557
+ if i == depth_i:
558
+ continue
559
+ vals = [val(r, i) for r in rows]
560
+ if any(v is None for v in vals):
561
+ if label_col is None:
562
+ label_col = [r[i].strip() if i < len(r) else '' for r in rows]
563
+ continue
564
+ mu = _re.search(r'\[([^\]]+)\]', h)
565
+ name = _re.sub(r'\s*\[[^\]]+\]', '', h).strip()
566
+ base, _n2 = name, 2
567
+ while name in channels:
568
+ name = f"{base} ({_n2})" # v1.50.0: PANGAEA tables may repeat
569
+ _n2 += 1 # bare names (k, a, b1...); silent
570
+ channels[name] = StreamChannel( # overwrite would lose channels
571
+ name=name, unit=(mu.group(1) if mu else ''),
572
+ values=_np.array(vals, dtype=float))
573
+ if label_col and channels:
574
+ first = next(iter(channels))
575
+ channels[first].quality = label_col
576
+ return LiveStream(name=p.stem, source_format='pangaea_txt',
577
+ index_kind=index_kind, index=depth, channels=channels, meta=meta)
578
+
579
+
580
+ def _csv_kind(path: Path) -> str:
581
+ """Distinguish catalogue CSV kinds by header: 'd,t' = temperature profile;
582
+ survey headers = deviation survey; DEPTH,OrigDepth,CORE_NO = core analysis;
583
+ DATEPRD,WELL_BORE_CODE = daily production history (time-indexed)."""
584
+ with path.open(encoding='utf-8') as f:
585
+ header = f.readline().strip().lower()
586
+ if header.startswith('d,t'):
587
+ return 'temperature'
588
+ if header.startswith('depth,origdepth,core_no'):
589
+ return 'core'
590
+ if header.startswith('dateprd,well_bore_code'):
591
+ return 'production'
592
+ if 'depth' in header and ('truevertical' in header.replace(' ', '') or 'tvd' in header):
593
+ return 'survey'
594
+ raise ValueError(f"catalogue CSV {path.name}: unrecognized header kind")
595
+
596
+
597
+ @dataclass(frozen=True)
598
+ class CatalogEntry:
599
+ name: str
600
+ las_path: Path
601
+ provenance: dict
602
+
603
+ def stream(self) -> LiveStream:
604
+ if self.las_path.suffix.lower() == '.txt':
605
+ with self.las_path.open(encoding='utf-8') as _f:
606
+ _first = _f.readline()
607
+ if _first.startswith('/* IODP TABLE TRANSCRIPTION'):
608
+ return read_iodp_table(self.las_path)
609
+ if _first.startswith('/* OPERATOR TABLE TRANSCRIPTION'):
610
+ return read_operator_table(self.las_path)
611
+ return read_pangaea_txt(self.las_path)
612
+ if self.las_path.suffix.lower() == '.xls':
613
+ return read_drift_xls(self.las_path)
614
+ if self.las_path.suffix.lower() == '.dat':
615
+ import re as _re
616
+ with self.las_path.open(encoding='utf-8') as _f:
617
+ _head = _f.read(4000)
618
+ if _re.search(r'^!\s+\d+\s+"[^"]*"\s+C\d*', _head, _re.M):
619
+ return read_ktb_table(self.las_path)
620
+ return read_ktb_dat(self.las_path)
621
+ if self.las_path.suffix.lower() == '.csv':
622
+ kind = _csv_kind(self.las_path)
623
+ if kind == 'survey':
624
+ raise ValueError(f"{self.name} is a DEVIATION SURVEY entry - "
625
+ "use .survey() (it is a trajectory, not a log stream)")
626
+ if kind == 'core':
627
+ return read_core_csv(self.las_path)
628
+ if kind == 'production':
629
+ return read_production_csv(self.las_path)
630
+ return read_temperature_csv(self.las_path)
631
+ return read_las(self.las_path)
632
+
633
+ def survey(self):
634
+ if self.las_path.suffix.lower() == '.dat':
635
+ st = read_ktb_dat(self.las_path)
636
+ if 'TVD' not in st.channels:
637
+ raise ValueError(f"{self.name} is not a trajectory entry (no TVD channel)")
638
+ from .deviation import DeviationSurvey
639
+ return DeviationSurvey(md_ft=[float(x) for x in st.index],
640
+ tvd_ft=[float(x) for x in st.channels['TVD'].values],
641
+ name=self.name)
642
+ if self.las_path.suffix.lower() != '.csv' or _csv_kind(self.las_path) != 'survey':
643
+ raise ValueError(f"{self.name} is not a survey entry")
644
+ return read_survey_csv(self.las_path)
645
+
646
+
647
+ _OPERATOR_DIR = _CATALOG_DIR.parent / 'catalog_operator'
648
+ # v1.71.0 OPERATOR TIER: field data supplied by the operator/user, loaded with
649
+ # the SAME sidecar discipline as the public catalogue but PRIVATE by
650
+ # construction - the directory is .gitignore'd, never listed in pyproject
651
+ # data-files (gate-enforced), and therefore never ships in the wheel or
652
+ # reaches PyPI/GitHub. Entries carry provenance['tier']='operator'. Machines
653
+ # without the directory (CI, other installs) simply load zero operator
654
+ # entries; nothing in the gate or acceptance suite REQUIRES their presence.
655
+
656
+
657
+ def _load_catalog() -> Dict[str, CatalogEntry]:
658
+ out: Dict[str, CatalogEntry] = {}
659
+ scan = [(_CATALOG_DIR, 'public'), (_OPERATOR_DIR, 'operator')]
660
+ for cat_dir, tier in scan:
661
+ if not cat_dir.is_dir():
662
+ continue
663
+ _load_catalog_dir(out, cat_dir, tier)
664
+ return out
665
+
666
+
667
+ def _load_catalog_dir(out, cat_dir, tier) -> None:
668
+ files = sorted(list(cat_dir.glob('*.las')) + list(cat_dir.glob('*.dat'))
669
+ + list(cat_dir.glob('*.txt')) + list(cat_dir.glob('*.xls'))
670
+ + [p for p in cat_dir.glob('*.csv') if not p.name.endswith('.provenance.json')])
671
+ for las in files:
672
+ prov_path = las.with_suffix('.provenance.json')
673
+ if not prov_path.exists():
674
+ raise ValueError(f"catalogue entry {las.name} has NO provenance sidecar - "
675
+ "an uncited catalogue entry is not a catalogue entry (Rule 7)")
676
+ with prov_path.open(encoding='utf-8') as f:
677
+ prov = json.load(f)
678
+ for req in ('source_database', 'source_url', 'license', 'fetch_date', 'coverage'):
679
+ if not prov.get(req):
680
+ raise ValueError(f"catalogue entry {las.name}: provenance missing '{req}'")
681
+ prov['tier'] = tier
682
+ out[las.stem] = CatalogEntry(name=las.stem, las_path=las, provenance=prov)
683
+
684
+
685
+ CATALOG: Dict[str, CatalogEntry] = _load_catalog()
686
+
687
+
688
+ # ---------------------------------------------------------------------------
689
+ # 3) The converter
690
+ # ---------------------------------------------------------------------------
691
+ def las_to_profile(stream_or_path, out_csv=None,
692
+ depth_unit: str = 'm',
693
+ surface_temp_F: float = 75.0, # anchor: template surface ambient
694
+ temp_gradient_F_per_ft: float = 0.018, # anchor: template geothermal gradient
695
+ surface_pressure_psi: float = 14.7, # anchor: 1 atm
696
+ pressure_gradient_psi_per_ft: float = 0.465 # anchor: industry hydrostatic
697
+ ) -> dict:
698
+ """Convert a LAS stream (or path) to the engine's profile CSV format
699
+ (depth_ft,pressure_psi,temp_F).
700
+
701
+ MEASURED path: when the LAS carries temperature/pressure curves (matched
702
+ on standard mnemonics) they are used, unit-converted, and the result is
703
+ stamped `derivation: MEASURED_CURVES`.
704
+
705
+ DERIVED path (most composite logs): no T/P curves exist - the REAL depth
706
+ stations are kept and conditions are filled from the DECLARED gradients,
707
+ stamped `derivation: DERIVED_GRADIENTS`. Useful for geometry-true
708
+ simulation; honest only because it says so (Rule 7).
709
+ """
710
+ stream = stream_or_path if isinstance(stream_or_path, LiveStream) else read_las(stream_or_path)
711
+ if stream.index_kind != 'depth':
712
+ raise ValueError("las_to_profile needs a depth-indexed stream")
713
+ depths = np.asarray(stream.index, dtype=float)
714
+ ok = ~np.isnan(depths)
715
+ depths_ft = depths[ok] * (M_TO_FT if depth_unit.lower().startswith('m') else 1.0)
716
+
717
+ temp_ch = next((c for m in _TEMP_MNEMONICS for c in stream.channels if c.upper().startswith(m)), None)
718
+ pres_ch = next((c for m in _PRES_MNEMONICS for c in stream.channels if c.upper().startswith(m)), None)
719
+
720
+ # Real header anchors (v1.13.0, from the Kennetcook #2 catalogue well):
721
+ # a measured BHT + TD in the LAS ~P section gives a REAL two-point thermal
722
+ # profile - stronger than pure gradients, weaker than a full curve, and
723
+ # labeled as exactly that.
724
+ bht_F = td_ft = None
725
+ if stream.meta.get('BHT'):
726
+ try:
727
+ bht = float(stream.meta['BHT'])
728
+ bht_F = bht * 9.0 / 5.0 + 32.0 if stream.meta.get('BHT_UNIT', '').upper().startswith('DEGC') else bht
729
+ td_m = float(stream.meta.get('TDL') or stream.meta.get('TDD') or 0.0)
730
+ td_ft = td_m * (M_TO_FT if depth_unit.lower().startswith('m') else 1.0) or None
731
+ except (ValueError, TypeError):
732
+ bht_F = td_ft = None
733
+
734
+ if temp_ch or pres_ch:
735
+ derivation = 'MEASURED_CURVES'
736
+ t_vals = stream.channels[temp_ch].values[ok] if temp_ch else None
737
+ p_vals = stream.channels[pres_ch].values[ok] if pres_ch else None
738
+ if t_vals is not None and stream.channels[temp_ch].unit.upper().startswith('DEGC'):
739
+ t_vals = t_vals * 9.0 / 5.0 + 32.0
740
+ temp_F = (t_vals if t_vals is not None
741
+ else surface_temp_F + depths_ft * temp_gradient_F_per_ft)
742
+ pres_psi = (p_vals if p_vals is not None
743
+ else surface_pressure_psi + depths_ft * pressure_gradient_psi_per_ft)
744
+ elif bht_F is not None and td_ft:
745
+ derivation = 'DERIVED_FROM_MEASURED_BHT'
746
+ temp_F = surface_temp_F + (bht_F - surface_temp_F) * (depths_ft / td_ft)
747
+ pres_psi = surface_pressure_psi + depths_ft * pressure_gradient_psi_per_ft
748
+ else:
749
+ derivation = 'DERIVED_GRADIENTS'
750
+ temp_F = surface_temp_F + depths_ft * temp_gradient_F_per_ft
751
+ pres_psi = surface_pressure_psi + depths_ft * pressure_gradient_psi_per_ft
752
+
753
+ rows = [(float(d), float(p), float(t)) for d, p, t in zip(depths_ft, pres_psi, temp_F)
754
+ if not (np.isnan(p) or np.isnan(t))]
755
+ result = {
756
+ 'stations': len(rows),
757
+ 'depth_range_ft': [round(rows[0][0], 1), round(rows[-1][0], 1)] if rows else None,
758
+ 'derivation': derivation,
759
+ 'temperature_curve_used': temp_ch,
760
+ 'pressure_curve_used': pres_ch,
761
+ 'source_stream': stream.name,
762
+ 'note': ('REAL depth stations; T/P filled from DECLARED gradients - labeled, not measured'
763
+ if derivation == 'DERIVED_GRADIENTS' else
764
+ 'measured curves converted; gaps dropped'),
765
+ }
766
+ if out_csv is not None:
767
+ p = Path(out_csv)
768
+ with p.open('w', encoding='utf-8', newline='') as f:
769
+ f.write('depth_ft,pressure_psi,temp_F\n')
770
+ for d, pr, t in rows:
771
+ f.write(f'{d:.1f},{pr:.1f},{t:.1f}\n')
772
+ result['csv'] = str(p)
773
+ return result