gea-program 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gea/BENCH_TEST_PROTOCOL.md +97 -0
- gea/IMPORT_RECORD.md +61 -0
- gea/__init__.py +160 -0
- gea/__main__.py +661 -0
- gea/acceptance_tests.py +1315 -0
- gea/accuracy_statement.py +181 -0
- gea/alarm_engine.py +307 -0
- gea/bench.py +144 -0
- gea/blind_harness.py +108 -0
- gea/case_study.py +229 -0
- gea/catalog/acex_lomonosov_age_depth_model.provenance.json +23 -0
- gea/catalog/acex_lomonosov_age_depth_model.txt +28 -0
- gea/catalog/agassiz77_canada_temperature.csv +68 -0
- gea/catalog/agassiz77_canada_temperature.provenance.json +17 -0
- gea/catalog/barbados_110_consolidation.provenance.json +24 -0
- gea/catalog/barbados_110_consolidation.txt +91 -0
- gea/catalog/bengal_u1452_grain_size.provenance.json +20 -0
- gea/catalog/bengal_u1452_grain_size.txt +252 -0
- gea/catalog/blake_164_methane_isotopes.provenance.json +22 -0
- gea/catalog/blake_164_methane_isotopes.txt +68 -0
- gea/catalog/chicxulub_m0077a_pwave_velocity.provenance.json +23 -0
- gea/catalog/chicxulub_m0077a_pwave_velocity.txt +735 -0
- gea/catalog/collingwood_1_28_ks_complete.las +128 -0
- gea/catalog/collingwood_1_28_ks_complete.provenance.json +17 -0
- gea/catalog/costa_rica_odp_friction_envelope.provenance.json +24 -0
- gea/catalog/costa_rica_odp_friction_envelope.txt +57 -0
- gea/catalog/dead_sea_5017_debrite_xrf_ms.provenance.json +21 -0
- gea/catalog/dead_sea_5017_debrite_xrf_ms.txt +94 -0
- gea/catalog/dsdp_504b_physical_properties.provenance.json +24 -0
- gea/catalog/dsdp_504b_physical_properties.txt +82 -0
- gea/catalog/dsdp_504b_sound_velocity.provenance.json +21 -0
- gea/catalog/dsdp_504b_sound_velocity.txt +81 -0
- gea/catalog/elgygytgyn_5011_turbidites.provenance.json +20 -0
- gea/catalog/elgygytgyn_5011_turbidites.txt +193 -0
- gea/catalog/epica_domec_co2_800kyr.provenance.json +21 -0
- gea/catalog/epica_domec_co2_800kyr.txt +265 -0
- gea/catalog/fram_909_organic_petrography.provenance.json +22 -0
- gea/catalog/fram_909_organic_petrography.txt +40 -0
- gea/catalog/gbr_325_coral_uth_ages.provenance.json +23 -0
- gea/catalog/gbr_325_coral_uth_ages.txt +71 -0
- gea/catalog/gisp2_greenland_temperature.csv +599 -0
- gea/catalog/gisp2_greenland_temperature.provenance.json +17 -0
- gea/catalog/gom_308_t2p_insitu.provenance.json +21 -0
- gea/catalog/gom_308_t2p_insitu.txt +40 -0
- gea/catalog/guaymas_385_dom_d13c.provenance.json +21 -0
- gea/catalog/guaymas_385_dom_d13c.txt +103 -0
- gea/catalog/hikurangi_u1520_friction_insitu.provenance.json +26 -0
- gea/catalog/hikurangi_u1520_friction_insitu.txt +74 -0
- gea/catalog/hydrate_ridge_204_ncr_hydrate.provenance.json +22 -0
- gea/catalog/hydrate_ridge_204_ncr_hydrate.txt +60 -0
- gea/catalog/iodp_u1324_pore_pressure.provenance.json +22 -0
- gea/catalog/iodp_u1324_pore_pressure.txt +42 -0
- gea/catalog/jfast_c0019_slow_slip_events.provenance.json +28 -0
- gea/catalog/jfast_c0019_slow_slip_events.txt +35 -0
- gea/catalog/kennetcook_2_p129_excerpt.las +139 -0
- gea/catalog/kennetcook_2_p129_excerpt.provenance.json +17 -0
- gea/catalog/ktb_hb_bhgm_density.dat +227 -0
- gea/catalog/ktb_hb_bhgm_density.provenance.json +22 -0
- gea/catalog/ktb_hb_complog_6020_excerpt.provenance.json +20 -0
- gea/catalog/ktb_hb_complog_6020_excerpt.txt +72 -0
- gea/catalog/ktb_hb_hlog246_temperature.dat +1636 -0
- gea/catalog/ktb_hb_hlog246_temperature.provenance.json +22 -0
- gea/catalog/ktb_hb_rockmech_compress.dat +33 -0
- gea/catalog/ktb_hb_rockmech_compress.provenance.json +22 -0
- gea/catalog/ktb_hb_tvd_0_9080_excerpt.dat +2817 -0
- gea/catalog/ktb_hb_tvd_0_9080_excerpt.provenance.json +20 -0
- gea/catalog/ktb_vb_rockmech_compress.dat +125 -0
- gea/catalog/ktb_vb_rockmech_compress.provenance.json +22 -0
- gea/catalog/ktb_vb_vlog251_temperature.dat +1127 -0
- gea/catalog/ktb_vb_vlog251_temperature.provenance.json +24 -0
- gea/catalog/l06_06_nl_survey.csv +201 -0
- gea/catalog/l06_06_nl_survey.provenance.json +17 -0
- gea/catalog/l07_01_nl_excerpt.las +90 -0
- gea/catalog/l07_01_nl_excerpt.provenance.json +17 -0
- gea/catalog/mariana_1200_serpentinite_geochem.provenance.json +21 -0
- gea/catalog/mariana_1200_serpentinite_geochem.txt +63 -0
- gea/catalog/med_160_sapropels.provenance.json +22 -0
- gea/catalog/med_160_sapropels.txt +43 -0
- gea/catalog/nankai_megasplay_shear_strength.provenance.json +26 -0
- gea/catalog/nankai_megasplay_shear_strength.txt +42 -0
- gea/catalog/odp_1027b_thermal_conductivity.provenance.json +21 -0
- gea/catalog/odp_1027b_thermal_conductivity.txt +52 -0
- gea/catalog/odp_1027c_cork_temperature.provenance.json +20 -0
- gea/catalog/odp_1027c_cork_temperature.txt +26 -0
- gea/catalog/odp_1165b_thermal_conductivity.provenance.json +22 -0
- gea/catalog/odp_1165b_thermal_conductivity.txt +102 -0
- gea/catalog/odp_1274a_mantle_peridotite_mad.provenance.json +27 -0
- gea/catalog/odp_1274a_mantle_peridotite_mad.txt +44 -0
- gea/catalog/odp_504b_dike_elastic_moduli.provenance.json +27 -0
- gea/catalog/odp_504b_dike_elastic_moduli.txt +85 -0
- gea/catalog/odp_504b_leg137_borehole_fluids.provenance.json +19 -0
- gea/catalog/odp_504b_leg137_borehole_fluids.txt +68 -0
- gea/catalog/odp_735b_gabbro_elastic_moduli.provenance.json +28 -0
- gea/catalog/odp_735b_gabbro_elastic_moduli.txt +127 -0
- gea/catalog/peru_201_sulfate_reduction.provenance.json +22 -0
- gea/catalog/peru_201_sulfate_reduction.txt +322 -0
- gea/catalog/scorpio_e1_sa_excerpt.las +113 -0
- gea/catalog/scorpio_e1_sa_excerpt.provenance.json +17 -0
- gea/catalog/sumatra_362_cohesion.provenance.json +21 -0
- gea/catalog/sumatra_362_cohesion.txt +38 -0
- gea/catalog/university_6_17_no1_tx_excerpt.las +119 -0
- gea/catalog/university_6_17_no1_tx_excerpt.provenance.json +17 -0
- gea/catalog/ursa_308_xrd_mineralogy.provenance.json +21 -0
- gea/catalog/ursa_308_xrd_mineralogy.txt +46 -0
- gea/catalog/volve_15_9_19_sr_excerpt.las +183 -0
- gea/catalog/volve_15_9_19_sr_excerpt.provenance.json +16 -0
- gea/catalog/volve_15_9_19a_core_excerpt.csv +88 -0
- gea/catalog/volve_15_9_19a_core_excerpt.provenance.json +18 -0
- gea/catalog/volve_f12_f14_production_excerpt.csv +167 -0
- gea/catalog/volve_f12_f14_production_excerpt.provenance.json +21 -0
- gea/catalog/walvis_208_petm_carbonate.provenance.json +21 -0
- gea/catalog/walvis_208_petm_carbonate.txt +268 -0
- gea/catalog/woodlark_1109_rock_eval.provenance.json +23 -0
- gea/catalog/woodlark_1109_rock_eval.txt +30 -0
- gea/cli.py +125 -0
- gea/client_reports.py +943 -0
- gea/config_versioning.py +133 -0
- gea/correlation.py +155 -0
- gea/dashboard.py +390 -0
- gea/deviation.py +70 -0
- gea/downhole_engine.py +395 -0
- gea/drift_monitor.py +310 -0
- gea/earth_model.py +230 -0
- gea/example_register_map.json +14 -0
- gea/fat_sat.py +68 -0
- gea/follower.py +98 -0
- gea/forward_model.py +139 -0
- gea/gamma.py +176 -0
- gea/gauge_specs.py +112 -0
- gea/gravity_reference.py +116 -0
- gea/inverse_engine.py +215 -0
- gea/matplotlib_demo.py +85 -0
- gea/modbus.py +229 -0
- gea/model_card.py +248 -0
- gea/operator_app.py +442 -0
- gea/ports.py +340 -0
- gea/profile_catalog.py +773 -0
- gea/project.py +213 -0
- gea/qt6_downhole_app.py +144 -0
- gea/quartz_hpht_extension.py +152 -0
- gea/reconciler.py +206 -0
- gea/rock_inventory.py +404 -0
- gea/sample_record.py +430 -0
- gea/sample_well_profile.csv +15 -0
- gea/sbom.py +116 -0
- gea/segy.py +181 -0
- gea/service_life.py +173 -0
- gea/shell.py +107 -0
- gea/sla_report.py +199 -0
- gea/store_forward.py +234 -0
- gea/strata_join.py +186 -0
- gea/survey_cmd.py +264 -0
- gea/survey_view.py +138 -0
- gea/telemetry.py +306 -0
- gea/tool_library.py +260 -0
- gea/well_assembler.py +457 -0
- gea/well_test_validation.py +369 -0
- gea_program-0.1.0.dist-info/METADATA +138 -0
- gea_program-0.1.0.dist-info/RECORD +163 -0
- gea_program-0.1.0.dist-info/WHEEL +5 -0
- gea_program-0.1.0.dist-info/entry_points.txt +2 -0
- gea_program-0.1.0.dist-info/licenses/LICENSE +373 -0
- gea_program-0.1.0.dist-info/top_level.txt +1 -0
gea/profile_catalog.py
ADDED
|
@@ -0,0 +1,773 @@
|
|
|
1
|
+
# This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
# file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
"""profile_catalog — the well-profile catalogue (v1.12.0 extension).
|
|
5
|
+
|
|
6
|
+
Daniel GO 2026-08-24: build a catalogue of well profiles from public
|
|
7
|
+
geophysical databases, so the closed stream can run on REAL wells instead of
|
|
8
|
+
one synthetic sample. Three parts:
|
|
9
|
+
|
|
10
|
+
1. `PROFILE_SOURCES` — the machine-readable table of public databases
|
|
11
|
+
(what each offers, direct entry points, license, access barriers stated
|
|
12
|
+
honestly: several serve only ZIPs or need registration, which this
|
|
13
|
+
environment cannot fetch — those are documented pull-it-yourself paths).
|
|
14
|
+
2. `CATALOG` — shipped entries. Every entry has a MANDATORY provenance
|
|
15
|
+
sidecar (.provenance.json) naming the source database, well, URL,
|
|
16
|
+
license, fetch date, and coverage. First real entry:
|
|
17
|
+
**Equinor Volve well 15/9-19 SR** (verbatim excerpt, CC/Equinor open
|
|
18
|
+
licence, disclosed coverage) — real third-party well-log data flowing
|
|
19
|
+
the las2 port end-to-end.
|
|
20
|
+
3. `las_to_profile()` — the converter: a LAS LiveStream becomes the
|
|
21
|
+
engine's profile CSV (depth_ft,pressure_psi,temp_F). Where the log
|
|
22
|
+
carries no temperature/pressure curves (most composites do not), the
|
|
23
|
+
converter fills them from DECLARED gradients and stamps the output
|
|
24
|
+
`derivation: DERIVED_GRADIENTS` — a profile built from a real
|
|
25
|
+
trajectory with derived conditions is useful and honest ONLY when
|
|
26
|
+
labeled (Rule 7); measured-curve conversion is used automatically when
|
|
27
|
+
the curves exist.
|
|
28
|
+
|
|
29
|
+
Headless-safe: numpy + stdlib.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import json
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
from typing import Dict, Optional
|
|
38
|
+
|
|
39
|
+
import numpy as np
|
|
40
|
+
|
|
41
|
+
from .ports import LiveStream, read_las
|
|
42
|
+
|
|
43
|
+
_CATALOG_DIR = Path(__file__).parent / "catalog"
|
|
44
|
+
|
|
45
|
+
# Temperature/pressure curve mnemonics accepted as MEASURED (LAS conventions)
|
|
46
|
+
_TEMP_MNEMONICS = ('TEMP', 'TEMPERATURE', 'BHT', 'WTEP', 'MRT', 'DTEMP', 'TMP')
|
|
47
|
+
_PRES_MNEMONICS = ('PRES', 'PRESSURE', 'WPRE', 'BHP', 'PFOR')
|
|
48
|
+
|
|
49
|
+
M_TO_FT = 3.28084
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# ---------------------------------------------------------------------------
|
|
53
|
+
# 1) The public-source table (honest access notes)
|
|
54
|
+
# ---------------------------------------------------------------------------
|
|
55
|
+
PROFILE_SOURCES: Dict[str, dict] = {
|
|
56
|
+
'kgs': {
|
|
57
|
+
'name': 'Kansas Geological Survey LAS database',
|
|
58
|
+
'url': 'https://www.kgs.ku.edu/Magellan/Logs/',
|
|
59
|
+
'offers': '21,000+ digital wireline logs (LAS), free, no registration',
|
|
60
|
+
'license': 'public state archive',
|
|
61
|
+
'access': 'individual downloads are ZIPPED via the search app; yearly bulk ZIPs; download and unzip locally, then ingest via las2'},
|
|
62
|
+
'volve': {
|
|
63
|
+
'name': 'Equinor Volve open dataset',
|
|
64
|
+
'url': 'https://www.equinor.com/energy/volve-data-sharing',
|
|
65
|
+
'offers': 'complete real North Sea field: logs, surveys, production (~40,000 files)',
|
|
66
|
+
'license': 'Equinor Open Data Licence (attribution)',
|
|
67
|
+
'access': 'registration required for the full archive; some files publicly redistributed (see catalogue entry volve_15_9_19_sr_excerpt)'},
|
|
68
|
+
'gdr_forge': {
|
|
69
|
+
'name': 'DOE Geothermal Data Repository - Utah FORGE',
|
|
70
|
+
'url': 'https://gdr.openei.org/submissions/1326',
|
|
71
|
+
'offers': 'REAL downhole T/P logs (wells 58-32, 56-32, 78-32; June 2021 update), drilling data, surveys; DOI 10.15121/1812334',
|
|
72
|
+
'license': 'CC BY 4.0',
|
|
73
|
+
'access': 'T/P logs served as ZIPs (server marks all files octet-stream); download and unzip locally, then ingest the contained .las/.csv'},
|
|
74
|
+
'nlog': {
|
|
75
|
+
'name': 'NLOG (Netherlands Oil and Gas portal)',
|
|
76
|
+
'url': 'https://www.nlog.nl/en',
|
|
77
|
+
'offers': 'thousands of onshore/offshore wells: logs, deviation, production',
|
|
78
|
+
'license': 'open by mandate',
|
|
79
|
+
'access': 'per-well downloads; formats vary'},
|
|
80
|
+
'state_regulators': {
|
|
81
|
+
'name': 'US state regulators (TX RRC, ND NDIC, OK OCC, CO ECMC, WY OGCC)',
|
|
82
|
+
'url': 'https://www.rrc.texas.gov/ (and peers)',
|
|
83
|
+
'offers': 'well files: directional surveys, pressure tests, BHT reports',
|
|
84
|
+
'license': 'public regulatory archives',
|
|
85
|
+
'access': 'per-state portals; mostly PDF/scans plus some digital data'},
|
|
86
|
+
'offshore_national': {
|
|
87
|
+
'name': 'BOEM/BSEE (US offshore), UK NSTA NDR, Australia NOPIMS',
|
|
88
|
+
'url': 'https://www.data.boem.gov/ ; https://ndr.nstauthority.co.uk/ ; https://nopims.disr.gov.au/',
|
|
89
|
+
'offers': 'national open repositories: surveys, logs, completions',
|
|
90
|
+
'license': 'open national archives',
|
|
91
|
+
'access': 'portal downloads; registration varies'},
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# ---------------------------------------------------------------------------
|
|
96
|
+
# 2) The shipped catalogue (provenance mandatory)
|
|
97
|
+
# ---------------------------------------------------------------------------
|
|
98
|
+
def read_temperature_csv(path) -> LiveStream:
|
|
99
|
+
"""Ingest a temperature-profile CSV (header `d,t`: depth in metres,
|
|
100
|
+
temperature in degC — the GEUS ice-borehole database format) as a
|
|
101
|
+
depth-indexed LiveStream with a TEMP channel. The catalogue's non-LAS
|
|
102
|
+
entry path (v1.16.0, driven by the GISP2 prize well)."""
|
|
103
|
+
import csv as _csv
|
|
104
|
+
p = Path(path)
|
|
105
|
+
d, t = [], []
|
|
106
|
+
with p.open(newline='', encoding='utf-8') as f:
|
|
107
|
+
for row in _csv.DictReader(f):
|
|
108
|
+
d.append(float(row['d']))
|
|
109
|
+
t.append(float(row['t']))
|
|
110
|
+
from .ports import StreamChannel
|
|
111
|
+
return LiveStream(name=p.stem, source_format='temperature_csv',
|
|
112
|
+
index_kind='depth', index=np.array(d, dtype=float),
|
|
113
|
+
channels={'TEMP': StreamChannel(name='TEMP', unit='DEGC',
|
|
114
|
+
values=np.array(t, dtype=float))},
|
|
115
|
+
meta={'format': 'GEUS d,t temperature profile'})
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def read_survey_csv(path):
|
|
119
|
+
"""Ingest a deviation-survey CSV (NLOG-style long headers: Depth /
|
|
120
|
+
TrueVertical Depth, with inclination/azimuth/offsets alongside) as a
|
|
121
|
+
DeviationSurvey - MD->TVD taken DIRECTLY from the measured columns, no
|
|
122
|
+
minimum-curvature reconstruction needed. The catalogue's survey-entry
|
|
123
|
+
path (v1.19.0, driven by the real L06-06 trajectory)."""
|
|
124
|
+
import csv as _csv
|
|
125
|
+
from .deviation import DeviationSurvey
|
|
126
|
+
p = Path(path)
|
|
127
|
+
md, tvd = [], []
|
|
128
|
+
with p.open(newline='', encoding='utf-8') as f:
|
|
129
|
+
reader = _csv.DictReader(f)
|
|
130
|
+
md_col = next(c for c in reader.fieldnames if c.strip().lower() in ('depth', 'md', 'md_ft'))
|
|
131
|
+
tvd_col = next(c for c in reader.fieldnames
|
|
132
|
+
if 'truevertical' in c.strip().lower().replace(' ', '')
|
|
133
|
+
or c.strip().lower() in ('tvd', 'tvd_ft'))
|
|
134
|
+
for row in reader:
|
|
135
|
+
md.append(float(row[md_col]))
|
|
136
|
+
tvd.append(float(row[tvd_col]))
|
|
137
|
+
return DeviationSurvey(md_ft=md, tvd_ft=tvd, name=p.stem)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def read_core_csv(path) -> LiveStream:
|
|
141
|
+
"""Ingest a conventional core-analysis CSV (Volve-style header:
|
|
142
|
+
DEPTH,OrigDepth,CORE_NO,SAMPLE,...) as a depth-indexed LiveStream whose
|
|
143
|
+
channels are the numeric lab columns (permeability/porosity/saturations/
|
|
144
|
+
grain density; blanks -> NaN). Laboratory ground truth alongside logs
|
|
145
|
+
(v1.20.0, driven by the real 15/9-19 A core data)."""
|
|
146
|
+
import csv as _csv
|
|
147
|
+
from .ports import StreamChannel
|
|
148
|
+
p = Path(path)
|
|
149
|
+
with p.open(newline='', encoding='utf-8') as f:
|
|
150
|
+
reader = _csv.DictReader(f)
|
|
151
|
+
cols = [c for c in reader.fieldnames if c != 'DEPTH']
|
|
152
|
+
depth, data = [], {c: [] for c in cols}
|
|
153
|
+
for row in reader:
|
|
154
|
+
depth.append(float(row['DEPTH']))
|
|
155
|
+
for c in cols:
|
|
156
|
+
v = (row.get(c) or '').strip()
|
|
157
|
+
data[c].append(float(v) if v else np.nan)
|
|
158
|
+
return LiveStream(name=p.stem, source_format='core_csv',
|
|
159
|
+
index_kind='depth', index=np.array(depth, dtype=float),
|
|
160
|
+
channels={c: StreamChannel(name=c, unit='', values=np.array(data[c]))
|
|
161
|
+
for c in cols},
|
|
162
|
+
meta={'format': 'conventional core analysis (units per provenance)'})
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def read_production_csv(path) -> LiveStream:
|
|
166
|
+
"""Ingest a daily production-history CSV (Volve-style header:
|
|
167
|
+
DATEPRD,WELL_BORE_CODE,...) as a TIME-indexed LiveStream - the
|
|
168
|
+
catalogue's first time-indexed kind (v1.21.0, driven by the real
|
|
169
|
+
15/9-F-12/F-14 daily records). Index = elapsed seconds from the first
|
|
170
|
+
date (86400 s cadence); channels are the per-well numeric operational
|
|
171
|
+
columns, namespaced COL[well]; blanks and absent dates -> NaN. Units are
|
|
172
|
+
NOT in the header - carried per the provenance data dictionary."""
|
|
173
|
+
import csv as _csv
|
|
174
|
+
from datetime import date as _date
|
|
175
|
+
from .ports import StreamChannel
|
|
176
|
+
p = Path(path)
|
|
177
|
+
numeric = ('ON_STREAM_HRS', 'AVG_DOWNHOLE_PRESSURE', 'AVG_DOWNHOLE_TEMPERATURE',
|
|
178
|
+
'AVG_DP_TUBING', 'AVG_ANNULUS_PRESS', 'AVG_CHOKE_SIZE_P', 'AVG_WHP_P',
|
|
179
|
+
'AVG_WHT_P', 'DP_CHOKE_SIZE', 'BORE_OIL_VOL', 'BORE_GAS_VOL',
|
|
180
|
+
'BORE_WAT_VOL', 'BORE_WI_VOL')
|
|
181
|
+
units = {'ON_STREAM_HRS': 'h', 'AVG_DOWNHOLE_PRESSURE': 'bar',
|
|
182
|
+
'AVG_DOWNHOLE_TEMPERATURE': 'degC', 'AVG_DP_TUBING': 'bar',
|
|
183
|
+
'AVG_ANNULUS_PRESS': 'bar', 'AVG_CHOKE_SIZE_P': 'pct',
|
|
184
|
+
'AVG_WHP_P': 'bar', 'AVG_WHT_P': 'degC', 'DP_CHOKE_SIZE': 'bar',
|
|
185
|
+
'BORE_OIL_VOL': 'Sm3', 'BORE_GAS_VOL': 'Sm3', 'BORE_WAT_VOL': 'Sm3',
|
|
186
|
+
'BORE_WI_VOL': 'Sm3'}
|
|
187
|
+
rows = []
|
|
188
|
+
with p.open(newline='', encoding='utf-8') as f:
|
|
189
|
+
for row in _csv.DictReader(f):
|
|
190
|
+
rows.append(row)
|
|
191
|
+
if not rows:
|
|
192
|
+
raise ValueError(f"production CSV {p.name}: no records")
|
|
193
|
+
dates = sorted({r['DATEPRD'] for r in rows})
|
|
194
|
+
wells = []
|
|
195
|
+
for r in rows:
|
|
196
|
+
w = r['NPD_WELL_BORE_NAME']
|
|
197
|
+
if w not in wells:
|
|
198
|
+
wells.append(w)
|
|
199
|
+
d0 = _date.fromisoformat(dates[0])
|
|
200
|
+
idx = np.array([( _date.fromisoformat(d) - d0).days * 86400.0 for d in dates])
|
|
201
|
+
pos = {d: i for i, d in enumerate(dates)}
|
|
202
|
+
channels = {}
|
|
203
|
+
for w in wells:
|
|
204
|
+
grids = {c: np.full(len(dates), np.nan) for c in numeric}
|
|
205
|
+
for r in rows:
|
|
206
|
+
if r['NPD_WELL_BORE_NAME'] != w:
|
|
207
|
+
continue
|
|
208
|
+
i = pos[r['DATEPRD']]
|
|
209
|
+
for c in numeric:
|
|
210
|
+
v = (r.get(c) or '').strip()
|
|
211
|
+
if v:
|
|
212
|
+
grids[c][i] = float(v)
|
|
213
|
+
for c in numeric:
|
|
214
|
+
channels[f"{c}[{w}]"] = StreamChannel(
|
|
215
|
+
name=f"{c}[{w}]", unit=units[c] + ' (per provenance dictionary)',
|
|
216
|
+
values=grids[c])
|
|
217
|
+
return LiveStream(name=p.stem, source_format='production_csv',
|
|
218
|
+
index_kind='time_s', index=idx, channels=channels,
|
|
219
|
+
meta={'format': 'daily production history (per-well, namespaced channels)',
|
|
220
|
+
'start_date': dates[0], 'end_date': dates[-1],
|
|
221
|
+
'wells': ';'.join(wells), 'cadence_s': '86400',
|
|
222
|
+
'units_note': 'units interpretive per provenance data dictionary, not in-file'})
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def read_ktb_dat(path) -> LiveStream:
|
|
226
|
+
"""Ingest a KTB Information System temperature-log file ('!'-comment
|
|
227
|
+
header + space-separated DEPT TMP3 HTEN MRES rows) as a depth-indexed
|
|
228
|
+
LiveStream - the catalogue's first HOT-regime temperature dialect
|
|
229
|
+
(v1.22.0, driven by the real KTB-HB hlog246). Header lines are carried
|
|
230
|
+
into meta (well name, log date, time-since-circulation fields - the
|
|
231
|
+
disturbed-log disclosure lives in the data itself)."""
|
|
232
|
+
import re as _re
|
|
233
|
+
p = Path(path)
|
|
234
|
+
hdr, rows, cols = [], [], []
|
|
235
|
+
col_re = _re.compile(r'^!\s+\d+\s+"(\w+)\s[^"]*"\s+F\d*\s+(\S+)')
|
|
236
|
+
with p.open(encoding='utf-8') as f:
|
|
237
|
+
for line in f:
|
|
238
|
+
line = line.rstrip('\n')
|
|
239
|
+
if not line.strip():
|
|
240
|
+
continue
|
|
241
|
+
if line.lstrip().startswith('!'):
|
|
242
|
+
hdr.append(line)
|
|
243
|
+
m = col_re.match(line.strip())
|
|
244
|
+
if m:
|
|
245
|
+
cols.append((m.group(1), m.group(2)))
|
|
246
|
+
continue
|
|
247
|
+
parts = line.split()
|
|
248
|
+
if cols and len(parts) == len(cols):
|
|
249
|
+
try:
|
|
250
|
+
rows.append([float(x) for x in parts])
|
|
251
|
+
except ValueError:
|
|
252
|
+
continue
|
|
253
|
+
if not cols:
|
|
254
|
+
raise ValueError(f"KTB log {p.name}: no column-definition block in header")
|
|
255
|
+
if not rows:
|
|
256
|
+
raise ValueError(f"KTB log {p.name}: no data rows")
|
|
257
|
+
import numpy as _np
|
|
258
|
+
arr = _np.array(rows, dtype=float)
|
|
259
|
+
meta = {'format': 'KTB Information System temperature log (disturbed mud-temperature log, per provenance)'}
|
|
260
|
+
val_re = {'well': _re.compile(r'"WN\s+UNAL\s+.*?\s{3,}(\S[^"]*)"'),
|
|
261
|
+
'log_date': _re.compile(r'"DATE\s+UNAL\s+.*?\s{3,}(\S[^"]*)"'),
|
|
262
|
+
'time_logger_at_bottom': _re.compile(r'"TLAB\s+UNAL\s+Time Logger At Bottom\s{3,}(\S[^"]*)"'),
|
|
263
|
+
'time_circulation_stopped': _re.compile(r'"TCS\s+UNAL\s+Time Circulation Stopped\s{3,}(\S[^"]*)"')}
|
|
264
|
+
for h in hdr:
|
|
265
|
+
for key, rx in val_re.items():
|
|
266
|
+
m = rx.search(h)
|
|
267
|
+
if m and key not in meta:
|
|
268
|
+
meta[key] = m.group(1).strip()
|
|
269
|
+
from .ports import StreamChannel
|
|
270
|
+
return LiveStream(name=p.stem, source_format='ktb_dat',
|
|
271
|
+
index_kind='depth', index=arr[:, 0],
|
|
272
|
+
channels={n: StreamChannel(name=n, unit=u, values=arr[:, i + 1])
|
|
273
|
+
for i, (n, u) in enumerate(cols[1:], start=0)},
|
|
274
|
+
meta=meta)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def read_ktb_table(path) -> LiveStream:
|
|
278
|
+
"""Ingest a KTB Information System TYPED table ('!'-header declaring
|
|
279
|
+
F/C/I columns, e.g. the rock-mechanics compressive-strength tables) as a
|
|
280
|
+
depth-indexed LiveStream (v1.26.0, driven by the real VB core-strength
|
|
281
|
+
file). Numeric (F/I) columns become channels; C-typed string columns
|
|
282
|
+
stay verbatim in the file (ROCK TYPE is carried as per-sample quality
|
|
283
|
+
on the strength channel). Rendering-collapsed tabs make some short rows
|
|
284
|
+
ambiguous: a trailing decimal token is assigned by DECLARED TYPE (an I2
|
|
285
|
+
dip cannot hold a decimal); a trailing integer token that could be
|
|
286
|
+
either column is REFUSED - NaN + a quality flag with the raw token."""
|
|
287
|
+
import re as _re
|
|
288
|
+
p = Path(path)
|
|
289
|
+
col_re = _re.compile(r'^!\s+\d+\s+"([^"]+)"\s+([FCI])\d*\s*(\S*)')
|
|
290
|
+
tok_re = _re.compile(r'"[^"]*"|\S+')
|
|
291
|
+
cols, rows = [], []
|
|
292
|
+
with p.open(encoding='utf-8') as f:
|
|
293
|
+
for line in f:
|
|
294
|
+
line = line.rstrip('\n')
|
|
295
|
+
if not line.strip():
|
|
296
|
+
continue
|
|
297
|
+
if line.lstrip().startswith('!'):
|
|
298
|
+
m = col_re.match(line.strip())
|
|
299
|
+
if m:
|
|
300
|
+
cols.append((m.group(1).replace(' ', '_'), m.group(2), m.group(3)))
|
|
301
|
+
continue
|
|
302
|
+
toks = tok_re.findall(line)
|
|
303
|
+
if len(toks) >= 6:
|
|
304
|
+
rows.append(toks)
|
|
305
|
+
if not cols or not rows:
|
|
306
|
+
raise ValueError(f"KTB table {p.name}: no typed column block or no rows")
|
|
307
|
+
import numpy as _np
|
|
308
|
+
n = len(rows)
|
|
309
|
+
names = [c[0] for c in cols]
|
|
310
|
+
num_idx = [i for i, c in enumerate(cols) if c[1] in ('F', 'I')]
|
|
311
|
+
grids = {names[i]: _np.full(n, _np.nan) for i in num_idx if i > 0}
|
|
312
|
+
depth = _np.full(n, _np.nan)
|
|
313
|
+
rock = ['' for _ in range(n)]
|
|
314
|
+
flags = ['' for _ in range(n)]
|
|
315
|
+
rock_col = next((i for i, c in enumerate(cols) if 'ROCK' in c[0]), None)
|
|
316
|
+
for r, toks in enumerate(rows):
|
|
317
|
+
if len(toks) == len(cols):
|
|
318
|
+
assign = list(enumerate(toks))
|
|
319
|
+
else:
|
|
320
|
+
assign = list(enumerate(toks[:6]))
|
|
321
|
+
trail = toks[6:]
|
|
322
|
+
if len(trail) == 1:
|
|
323
|
+
if '.' in trail[0]:
|
|
324
|
+
assign.append((6, trail[0]))
|
|
325
|
+
else:
|
|
326
|
+
flags[r] = f"AMBIGUOUS_TRAILING:{trail[0]}"
|
|
327
|
+
for ci, tok in assign:
|
|
328
|
+
name, typ = cols[ci][0], cols[ci][1]
|
|
329
|
+
if ci == 0:
|
|
330
|
+
depth[r] = float(tok)
|
|
331
|
+
elif typ in ('F', 'I'):
|
|
332
|
+
v = tok.strip().strip('"')
|
|
333
|
+
if v:
|
|
334
|
+
grids[name][r] = float(v)
|
|
335
|
+
elif ci == rock_col:
|
|
336
|
+
rock[r] = tok.strip('"')
|
|
337
|
+
from .ports import StreamChannel
|
|
338
|
+
channels = {}
|
|
339
|
+
for i in num_idx:
|
|
340
|
+
if i == 0:
|
|
341
|
+
continue
|
|
342
|
+
name, unit = names[i], cols[i][2]
|
|
343
|
+
q = rock if 'STRENGTH' in name else (flags if name == 'E_MODUL' else None)
|
|
344
|
+
channels[name] = StreamChannel(name=name, unit=unit, values=grids[name],
|
|
345
|
+
quality=list(q) if q else None)
|
|
346
|
+
return LiveStream(name=p.stem, source_format='ktb_table',
|
|
347
|
+
index_kind='depth', index=depth, channels=channels,
|
|
348
|
+
meta={'format': 'KTB typed table (rock mechanics); string columns verbatim in file',
|
|
349
|
+
'ambiguous_rows_refused': str(sum(1 for x in flags if x))})
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def read_operator_table(path) -> LiveStream:
|
|
353
|
+
"""Ingest a verbatim OPERATOR TABLE TRANSCRIPTION (v1.72.0): field-data
|
|
354
|
+
tables recovered from operator report screenshots/exports, transcribed
|
|
355
|
+
cell-for-cell. Format: /* OPERATOR TABLE TRANSCRIPTION */ header
|
|
356
|
+
(Key:<TAB>Value lines incl. IndexKind: depth|ordinal) then a TSV table.
|
|
357
|
+
Rows whose index cell is non-numeric (section markers like 'Curve',
|
|
358
|
+
'Lateral') are carried verbatim into meta['marker_rows'] with their
|
|
359
|
+
position. Text columns ride in meta as row-aligned lists; numeric columns
|
|
360
|
+
become channels; nothing is recomputed at ingest."""
|
|
361
|
+
p = Path(path)
|
|
362
|
+
txt = p.read_text(encoding='utf-8')
|
|
363
|
+
head, _, body = txt.partition('*/')
|
|
364
|
+
meta = {}
|
|
365
|
+
for line in head.splitlines():
|
|
366
|
+
if ':\t' in line:
|
|
367
|
+
k, _, v = line.partition(':\t')
|
|
368
|
+
meta[k.strip('/* ').strip().lower()] = v.strip()
|
|
369
|
+
lines = [l for l in body.strip('\n').split('\n') if l]
|
|
370
|
+
cols = lines[0].split('\t')
|
|
371
|
+
raw = [l.split('\t') for l in lines[1:]]
|
|
372
|
+
markers, rows = [], []
|
|
373
|
+
ordinal = meta.get('indexkind') == 'ordinal'
|
|
374
|
+
for i, r in enumerate(raw):
|
|
375
|
+
if ordinal:
|
|
376
|
+
rows.append(r) # ordinal tables: col 0 may be text (timestamps)
|
|
377
|
+
continue
|
|
378
|
+
try:
|
|
379
|
+
float(r[0])
|
|
380
|
+
rows.append(r)
|
|
381
|
+
except ValueError:
|
|
382
|
+
markers.append((i, '\t'.join(r).strip()))
|
|
383
|
+
if markers:
|
|
384
|
+
meta['marker_rows'] = '; '.join('row %d: %s' % m for m in markers)
|
|
385
|
+
from .ports import StreamChannel
|
|
386
|
+
index_kind = meta.get('indexkind', 'depth')
|
|
387
|
+
if ordinal:
|
|
388
|
+
index = np.arange(1, len(rows) + 1, dtype=float)
|
|
389
|
+
start = 0
|
|
390
|
+
else:
|
|
391
|
+
index = np.array([float(r[0]) for r in rows], dtype=float)
|
|
392
|
+
start = 1
|
|
393
|
+
channels = {}
|
|
394
|
+
for c in range(start, len(cols)):
|
|
395
|
+
vals, numeric = [], 0
|
|
396
|
+
for r in rows:
|
|
397
|
+
cell = r[c].strip() if c < len(r) else ''
|
|
398
|
+
try:
|
|
399
|
+
vals.append(float(cell))
|
|
400
|
+
numeric += 1
|
|
401
|
+
except ValueError:
|
|
402
|
+
vals.append(float('nan'))
|
|
403
|
+
if numeric:
|
|
404
|
+
channels[cols[c]] = StreamChannel(name=cols[c], unit='',
|
|
405
|
+
values=np.array(vals, dtype=float))
|
|
406
|
+
else:
|
|
407
|
+
meta['textcol_' + cols[c]] = [r[c].strip() if c < len(r) else ''
|
|
408
|
+
for r in rows]
|
|
409
|
+
return LiveStream(name=p.stem, source_format='operator_table',
|
|
410
|
+
index_kind=('depth' if index_kind != 'ordinal' else 'ordinal'),
|
|
411
|
+
index=index, channels=channels, meta=meta)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def read_drift_xls(path) -> LiveStream:
|
|
415
|
+
"""Ingest a directional-drilling drift/survey XLS export (v1.71.0, driven
|
|
416
|
+
by the first OPERATOR-tier entry: the Retama Ranch #403H 183-station
|
|
417
|
+
survey). Header row names MD / Inclination / Azimuth / TVD / NS / EW
|
|
418
|
+
(vendor exports interleave blank columns; they are skipped). Depth index =
|
|
419
|
+
MD [ft]; every named numeric column becomes a channel. Verbatim: cells are
|
|
420
|
+
read as exported, nothing is recomputed or smoothed at ingest."""
|
|
421
|
+
try:
|
|
422
|
+
import xlrd as _xlrd
|
|
423
|
+
except ImportError as _e:
|
|
424
|
+
raise ImportError(
|
|
425
|
+
"read_drift_xls needs the optional third-party module 'xlrd' "
|
|
426
|
+
"(pip install xlrd). No SHIPPED catalogue entry requires it - "
|
|
427
|
+
"operator drift surveys are stored in the dependency-free "
|
|
428
|
+
"operator-table format since the v0.406.0 ship-rehearsal catch; "
|
|
429
|
+
"this reader exists for ingesting NEW vendor .xls drops only."
|
|
430
|
+
) from _e
|
|
431
|
+
p = Path(path)
|
|
432
|
+
wb = _xlrd.open_workbook(str(p))
|
|
433
|
+
sh = wb.sheet_by_index(0)
|
|
434
|
+
header = [str(sh.cell_value(0, c)).strip() for c in range(sh.ncols)]
|
|
435
|
+
cols = [(c, h) for c, h in enumerate(header) if h]
|
|
436
|
+
md_c = next(c for c, h in cols if h.lower().startswith('md'))
|
|
437
|
+
from .ports import StreamChannel
|
|
438
|
+
md, rows = [], []
|
|
439
|
+
for r in range(1, sh.nrows):
|
|
440
|
+
try:
|
|
441
|
+
md.append(float(sh.cell_value(r, md_c)))
|
|
442
|
+
except (TypeError, ValueError):
|
|
443
|
+
continue
|
|
444
|
+
rows.append(r)
|
|
445
|
+
channels = {}
|
|
446
|
+
for c, h in cols:
|
|
447
|
+
if c == md_c:
|
|
448
|
+
continue
|
|
449
|
+
vals = []
|
|
450
|
+
for r in rows:
|
|
451
|
+
try:
|
|
452
|
+
vals.append(float(sh.cell_value(r, c)))
|
|
453
|
+
except (TypeError, ValueError):
|
|
454
|
+
vals.append(float('nan'))
|
|
455
|
+
channels[h] = StreamChannel(name=h, unit='', values=np.array(vals, dtype=float))
|
|
456
|
+
return LiveStream(name=p.stem, source_format='drift_xls', index_kind='depth',
|
|
457
|
+
index=np.array(md, dtype=float), channels=channels,
|
|
458
|
+
meta={'sheet': sh.name, 'stations': str(len(md))})
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def read_iodp_table(path) -> LiveStream:
|
|
462
|
+
"""Ingest a verbatim IODP Proceedings data-report table transcription
|
|
463
|
+
(v1.70.0, driven by Exp 308 Table T2 - the in situ temperature AND
|
|
464
|
+
pressure penetrometer results that made U1324 the catalogue's first
|
|
465
|
+
measured-T+P site). File format: a /* IODP TABLE TRANSCRIPTION */ header
|
|
466
|
+
(citation, source URL, license, verbatim table notes) then a tab-separated
|
|
467
|
+
table whose cells are carried verbatim. Numeric columns become channels;
|
|
468
|
+
cells that are blank or hold the T2P dual-port 'a; b' pairs become NaN in
|
|
469
|
+
the channel (the verbatim cell stays in the file - the reader never
|
|
470
|
+
repairs); the Hole column rides row-aligned in meta['hole'] so assemblies
|
|
471
|
+
can filter one site out of a multi-site table without touching the
|
|
472
|
+
archive. Depth index = the first column whose name contains 'mbsf'."""
|
|
473
|
+
p = Path(path)
|
|
474
|
+
txt = p.read_text(encoding='utf-8')
|
|
475
|
+
head, _, body = txt.partition('*/')
|
|
476
|
+
meta = {}
|
|
477
|
+
for line in head.splitlines():
|
|
478
|
+
if ':\t' in line:
|
|
479
|
+
k, _, v = line.partition(':\t')
|
|
480
|
+
meta[k.strip('/* ').strip().lower()] = v.strip()
|
|
481
|
+
lines = [l for l in body.strip('\n').split('\n') if l]
|
|
482
|
+
cols = lines[0].split('\t')
|
|
483
|
+
rows = [l.split('\t') for l in lines[1:]]
|
|
484
|
+
depth_i = next(i for i, c in enumerate(cols) if 'mbsf' in c.lower())
|
|
485
|
+
hole_i = next((i for i, c in enumerate(cols) if c.strip().lower() == 'hole'), None)
|
|
486
|
+
from .ports import StreamChannel
|
|
487
|
+
channels = {}
|
|
488
|
+
for i, c in enumerate(cols):
|
|
489
|
+
if i in (depth_i, hole_i):
|
|
490
|
+
continue
|
|
491
|
+
vals = []
|
|
492
|
+
numeric = 0
|
|
493
|
+
for r in rows:
|
|
494
|
+
cell = r[i].strip() if i < len(r) else ''
|
|
495
|
+
try:
|
|
496
|
+
vals.append(float(cell))
|
|
497
|
+
numeric += 1
|
|
498
|
+
except ValueError:
|
|
499
|
+
vals.append(float('nan'))
|
|
500
|
+
if numeric:
|
|
501
|
+
channels[c] = StreamChannel(name=c, unit='', values=np.array(vals, dtype=float))
|
|
502
|
+
if hole_i is not None:
|
|
503
|
+
meta['hole'] = [r[hole_i].strip() for r in rows]
|
|
504
|
+
return LiveStream(name=p.stem, source_format='iodp_table', index_kind='depth',
|
|
505
|
+
index=np.array([float(r[depth_i]) for r in rows], dtype=float),
|
|
506
|
+
channels=channels, meta=meta)
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def read_pangaea_txt(path) -> LiveStream:
|
|
510
|
+
"""Ingest a PANGAEA machine-readable textfile export (self-describing
|
|
511
|
+
'/* DATA DESCRIPTION */' header + tab-separated matrix) as a
|
|
512
|
+
depth-indexed LiveStream (v1.28.0, driven by the real ODP 504B borehole
|
|
513
|
+
-fluid dataset). The header's citation, license and coordinates go to
|
|
514
|
+
meta; numeric columns become channels (units parsed from '[...]');
|
|
515
|
+
short rows pad to NaN; the first non-numeric column rides as per-sample
|
|
516
|
+
quality on the first channel."""
|
|
517
|
+
import re as _re
|
|
518
|
+
p = Path(path)
|
|
519
|
+
text = p.open(encoding='utf-8').read()
|
|
520
|
+
if not text.startswith('/* DATA DESCRIPTION'):
|
|
521
|
+
raise ValueError(f"{p.name}: not a PANGAEA textfile export")
|
|
522
|
+
head, _, body = text.partition('*/')
|
|
523
|
+
meta = {'format': 'PANGAEA textfile export (self-describing header verbatim in file)'}
|
|
524
|
+
m = _re.search(r'Citation:\t([^\n]+)', head)
|
|
525
|
+
if m:
|
|
526
|
+
meta['citation'] = m.group(1).strip().rstrip(',')
|
|
527
|
+
m = _re.search(r'License:\t([^\n]+)', head)
|
|
528
|
+
if m:
|
|
529
|
+
meta['license'] = m.group(1).strip()
|
|
530
|
+
m = _re.search(r'LATITUDE:\s*(-?[\d.]+)\s*\*\s*LONGITUDE:\s*(-?[\d.]+)', head)
|
|
531
|
+
if m:
|
|
532
|
+
meta['latitude'], meta['longitude'] = m.group(1), m.group(2)
|
|
533
|
+
m = _re.search(r'ELEVATION:\s*(-?[\d.]+)', head)
|
|
534
|
+
if m:
|
|
535
|
+
meta['elevation_m'] = m.group(1)
|
|
536
|
+
lines = [l for l in body.split('\n') if l.strip()]
|
|
537
|
+
headers = lines[0].split('\t')
|
|
538
|
+
rows = [l.split('\t') for l in lines[1:]]
|
|
539
|
+
import numpy as _np
|
|
540
|
+
depth_i = next((i for i, h in enumerate(headers) if h.lower().startswith('depth')), None)
|
|
541
|
+
index_kind = 'depth'
|
|
542
|
+
def val(r, i):
|
|
543
|
+
v = r[i].strip() if i < len(r) else ''
|
|
544
|
+
try:
|
|
545
|
+
return float(v) if v else _np.nan
|
|
546
|
+
except ValueError:
|
|
547
|
+
return None
|
|
548
|
+
if depth_i is None:
|
|
549
|
+
index_kind = 'ordinal'
|
|
550
|
+
depth = _np.arange(1, len(rows) + 1, dtype=float)
|
|
551
|
+
else:
|
|
552
|
+
depth = _np.array([val(r, depth_i) for r in rows], dtype=float)
|
|
553
|
+
from .ports import StreamChannel
|
|
554
|
+
channels = {}
|
|
555
|
+
label_col = None
|
|
556
|
+
for i, h in enumerate(headers):
|
|
557
|
+
if i == depth_i:
|
|
558
|
+
continue
|
|
559
|
+
vals = [val(r, i) for r in rows]
|
|
560
|
+
if any(v is None for v in vals):
|
|
561
|
+
if label_col is None:
|
|
562
|
+
label_col = [r[i].strip() if i < len(r) else '' for r in rows]
|
|
563
|
+
continue
|
|
564
|
+
mu = _re.search(r'\[([^\]]+)\]', h)
|
|
565
|
+
name = _re.sub(r'\s*\[[^\]]+\]', '', h).strip()
|
|
566
|
+
base, _n2 = name, 2
|
|
567
|
+
while name in channels:
|
|
568
|
+
name = f"{base} ({_n2})" # v1.50.0: PANGAEA tables may repeat
|
|
569
|
+
_n2 += 1 # bare names (k, a, b1...); silent
|
|
570
|
+
channels[name] = StreamChannel( # overwrite would lose channels
|
|
571
|
+
name=name, unit=(mu.group(1) if mu else ''),
|
|
572
|
+
values=_np.array(vals, dtype=float))
|
|
573
|
+
if label_col and channels:
|
|
574
|
+
first = next(iter(channels))
|
|
575
|
+
channels[first].quality = label_col
|
|
576
|
+
return LiveStream(name=p.stem, source_format='pangaea_txt',
|
|
577
|
+
index_kind=index_kind, index=depth, channels=channels, meta=meta)
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def _csv_kind(path: Path) -> str:
|
|
581
|
+
"""Distinguish catalogue CSV kinds by header: 'd,t' = temperature profile;
|
|
582
|
+
survey headers = deviation survey; DEPTH,OrigDepth,CORE_NO = core analysis;
|
|
583
|
+
DATEPRD,WELL_BORE_CODE = daily production history (time-indexed)."""
|
|
584
|
+
with path.open(encoding='utf-8') as f:
|
|
585
|
+
header = f.readline().strip().lower()
|
|
586
|
+
if header.startswith('d,t'):
|
|
587
|
+
return 'temperature'
|
|
588
|
+
if header.startswith('depth,origdepth,core_no'):
|
|
589
|
+
return 'core'
|
|
590
|
+
if header.startswith('dateprd,well_bore_code'):
|
|
591
|
+
return 'production'
|
|
592
|
+
if 'depth' in header and ('truevertical' in header.replace(' ', '') or 'tvd' in header):
|
|
593
|
+
return 'survey'
|
|
594
|
+
raise ValueError(f"catalogue CSV {path.name}: unrecognized header kind")
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
@dataclass(frozen=True)
|
|
598
|
+
class CatalogEntry:
|
|
599
|
+
name: str
|
|
600
|
+
las_path: Path
|
|
601
|
+
provenance: dict
|
|
602
|
+
|
|
603
|
+
def stream(self) -> LiveStream:
|
|
604
|
+
if self.las_path.suffix.lower() == '.txt':
|
|
605
|
+
with self.las_path.open(encoding='utf-8') as _f:
|
|
606
|
+
_first = _f.readline()
|
|
607
|
+
if _first.startswith('/* IODP TABLE TRANSCRIPTION'):
|
|
608
|
+
return read_iodp_table(self.las_path)
|
|
609
|
+
if _first.startswith('/* OPERATOR TABLE TRANSCRIPTION'):
|
|
610
|
+
return read_operator_table(self.las_path)
|
|
611
|
+
return read_pangaea_txt(self.las_path)
|
|
612
|
+
if self.las_path.suffix.lower() == '.xls':
|
|
613
|
+
return read_drift_xls(self.las_path)
|
|
614
|
+
if self.las_path.suffix.lower() == '.dat':
|
|
615
|
+
import re as _re
|
|
616
|
+
with self.las_path.open(encoding='utf-8') as _f:
|
|
617
|
+
_head = _f.read(4000)
|
|
618
|
+
if _re.search(r'^!\s+\d+\s+"[^"]*"\s+C\d*', _head, _re.M):
|
|
619
|
+
return read_ktb_table(self.las_path)
|
|
620
|
+
return read_ktb_dat(self.las_path)
|
|
621
|
+
if self.las_path.suffix.lower() == '.csv':
|
|
622
|
+
kind = _csv_kind(self.las_path)
|
|
623
|
+
if kind == 'survey':
|
|
624
|
+
raise ValueError(f"{self.name} is a DEVIATION SURVEY entry - "
|
|
625
|
+
"use .survey() (it is a trajectory, not a log stream)")
|
|
626
|
+
if kind == 'core':
|
|
627
|
+
return read_core_csv(self.las_path)
|
|
628
|
+
if kind == 'production':
|
|
629
|
+
return read_production_csv(self.las_path)
|
|
630
|
+
return read_temperature_csv(self.las_path)
|
|
631
|
+
return read_las(self.las_path)
|
|
632
|
+
|
|
633
|
+
def survey(self):
|
|
634
|
+
if self.las_path.suffix.lower() == '.dat':
|
|
635
|
+
st = read_ktb_dat(self.las_path)
|
|
636
|
+
if 'TVD' not in st.channels:
|
|
637
|
+
raise ValueError(f"{self.name} is not a trajectory entry (no TVD channel)")
|
|
638
|
+
from .deviation import DeviationSurvey
|
|
639
|
+
return DeviationSurvey(md_ft=[float(x) for x in st.index],
|
|
640
|
+
tvd_ft=[float(x) for x in st.channels['TVD'].values],
|
|
641
|
+
name=self.name)
|
|
642
|
+
if self.las_path.suffix.lower() != '.csv' or _csv_kind(self.las_path) != 'survey':
|
|
643
|
+
raise ValueError(f"{self.name} is not a survey entry")
|
|
644
|
+
return read_survey_csv(self.las_path)
|
|
645
|
+
|
|
646
|
+
|
|
647
|
+
_OPERATOR_DIR = _CATALOG_DIR.parent / 'catalog_operator'
|
|
648
|
+
# v1.71.0 OPERATOR TIER: field data supplied by the operator/user, loaded with
|
|
649
|
+
# the SAME sidecar discipline as the public catalogue but PRIVATE by
|
|
650
|
+
# construction - the directory is .gitignore'd, never listed in pyproject
|
|
651
|
+
# data-files (gate-enforced), and therefore never ships in the wheel or
|
|
652
|
+
# reaches PyPI/GitHub. Entries carry provenance['tier']='operator'. Machines
|
|
653
|
+
# without the directory (CI, other installs) simply load zero operator
|
|
654
|
+
# entries; nothing in the gate or acceptance suite REQUIRES their presence.
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def _load_catalog() -> Dict[str, CatalogEntry]:
|
|
658
|
+
out: Dict[str, CatalogEntry] = {}
|
|
659
|
+
scan = [(_CATALOG_DIR, 'public'), (_OPERATOR_DIR, 'operator')]
|
|
660
|
+
for cat_dir, tier in scan:
|
|
661
|
+
if not cat_dir.is_dir():
|
|
662
|
+
continue
|
|
663
|
+
_load_catalog_dir(out, cat_dir, tier)
|
|
664
|
+
return out
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
def _load_catalog_dir(out, cat_dir, tier) -> None:
|
|
668
|
+
files = sorted(list(cat_dir.glob('*.las')) + list(cat_dir.glob('*.dat'))
|
|
669
|
+
+ list(cat_dir.glob('*.txt')) + list(cat_dir.glob('*.xls'))
|
|
670
|
+
+ [p for p in cat_dir.glob('*.csv') if not p.name.endswith('.provenance.json')])
|
|
671
|
+
for las in files:
|
|
672
|
+
prov_path = las.with_suffix('.provenance.json')
|
|
673
|
+
if not prov_path.exists():
|
|
674
|
+
raise ValueError(f"catalogue entry {las.name} has NO provenance sidecar - "
|
|
675
|
+
"an uncited catalogue entry is not a catalogue entry (Rule 7)")
|
|
676
|
+
with prov_path.open(encoding='utf-8') as f:
|
|
677
|
+
prov = json.load(f)
|
|
678
|
+
for req in ('source_database', 'source_url', 'license', 'fetch_date', 'coverage'):
|
|
679
|
+
if not prov.get(req):
|
|
680
|
+
raise ValueError(f"catalogue entry {las.name}: provenance missing '{req}'")
|
|
681
|
+
prov['tier'] = tier
|
|
682
|
+
out[las.stem] = CatalogEntry(name=las.stem, las_path=las, provenance=prov)
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
CATALOG: Dict[str, CatalogEntry] = _load_catalog()
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
# ---------------------------------------------------------------------------
|
|
689
|
+
# 3) The converter
|
|
690
|
+
# ---------------------------------------------------------------------------
|
|
691
|
+
def las_to_profile(stream_or_path, out_csv=None,
|
|
692
|
+
depth_unit: str = 'm',
|
|
693
|
+
surface_temp_F: float = 75.0, # anchor: template surface ambient
|
|
694
|
+
temp_gradient_F_per_ft: float = 0.018, # anchor: template geothermal gradient
|
|
695
|
+
surface_pressure_psi: float = 14.7, # anchor: 1 atm
|
|
696
|
+
pressure_gradient_psi_per_ft: float = 0.465 # anchor: industry hydrostatic
|
|
697
|
+
) -> dict:
|
|
698
|
+
"""Convert a LAS stream (or path) to the engine's profile CSV format
|
|
699
|
+
(depth_ft,pressure_psi,temp_F).
|
|
700
|
+
|
|
701
|
+
MEASURED path: when the LAS carries temperature/pressure curves (matched
|
|
702
|
+
on standard mnemonics) they are used, unit-converted, and the result is
|
|
703
|
+
stamped `derivation: MEASURED_CURVES`.
|
|
704
|
+
|
|
705
|
+
DERIVED path (most composite logs): no T/P curves exist - the REAL depth
|
|
706
|
+
stations are kept and conditions are filled from the DECLARED gradients,
|
|
707
|
+
stamped `derivation: DERIVED_GRADIENTS`. Useful for geometry-true
|
|
708
|
+
simulation; honest only because it says so (Rule 7).
|
|
709
|
+
"""
|
|
710
|
+
stream = stream_or_path if isinstance(stream_or_path, LiveStream) else read_las(stream_or_path)
|
|
711
|
+
if stream.index_kind != 'depth':
|
|
712
|
+
raise ValueError("las_to_profile needs a depth-indexed stream")
|
|
713
|
+
depths = np.asarray(stream.index, dtype=float)
|
|
714
|
+
ok = ~np.isnan(depths)
|
|
715
|
+
depths_ft = depths[ok] * (M_TO_FT if depth_unit.lower().startswith('m') else 1.0)
|
|
716
|
+
|
|
717
|
+
temp_ch = next((c for m in _TEMP_MNEMONICS for c in stream.channels if c.upper().startswith(m)), None)
|
|
718
|
+
pres_ch = next((c for m in _PRES_MNEMONICS for c in stream.channels if c.upper().startswith(m)), None)
|
|
719
|
+
|
|
720
|
+
# Real header anchors (v1.13.0, from the Kennetcook #2 catalogue well):
|
|
721
|
+
# a measured BHT + TD in the LAS ~P section gives a REAL two-point thermal
|
|
722
|
+
# profile - stronger than pure gradients, weaker than a full curve, and
|
|
723
|
+
# labeled as exactly that.
|
|
724
|
+
bht_F = td_ft = None
|
|
725
|
+
if stream.meta.get('BHT'):
|
|
726
|
+
try:
|
|
727
|
+
bht = float(stream.meta['BHT'])
|
|
728
|
+
bht_F = bht * 9.0 / 5.0 + 32.0 if stream.meta.get('BHT_UNIT', '').upper().startswith('DEGC') else bht
|
|
729
|
+
td_m = float(stream.meta.get('TDL') or stream.meta.get('TDD') or 0.0)
|
|
730
|
+
td_ft = td_m * (M_TO_FT if depth_unit.lower().startswith('m') else 1.0) or None
|
|
731
|
+
except (ValueError, TypeError):
|
|
732
|
+
bht_F = td_ft = None
|
|
733
|
+
|
|
734
|
+
if temp_ch or pres_ch:
|
|
735
|
+
derivation = 'MEASURED_CURVES'
|
|
736
|
+
t_vals = stream.channels[temp_ch].values[ok] if temp_ch else None
|
|
737
|
+
p_vals = stream.channels[pres_ch].values[ok] if pres_ch else None
|
|
738
|
+
if t_vals is not None and stream.channels[temp_ch].unit.upper().startswith('DEGC'):
|
|
739
|
+
t_vals = t_vals * 9.0 / 5.0 + 32.0
|
|
740
|
+
temp_F = (t_vals if t_vals is not None
|
|
741
|
+
else surface_temp_F + depths_ft * temp_gradient_F_per_ft)
|
|
742
|
+
pres_psi = (p_vals if p_vals is not None
|
|
743
|
+
else surface_pressure_psi + depths_ft * pressure_gradient_psi_per_ft)
|
|
744
|
+
elif bht_F is not None and td_ft:
|
|
745
|
+
derivation = 'DERIVED_FROM_MEASURED_BHT'
|
|
746
|
+
temp_F = surface_temp_F + (bht_F - surface_temp_F) * (depths_ft / td_ft)
|
|
747
|
+
pres_psi = surface_pressure_psi + depths_ft * pressure_gradient_psi_per_ft
|
|
748
|
+
else:
|
|
749
|
+
derivation = 'DERIVED_GRADIENTS'
|
|
750
|
+
temp_F = surface_temp_F + depths_ft * temp_gradient_F_per_ft
|
|
751
|
+
pres_psi = surface_pressure_psi + depths_ft * pressure_gradient_psi_per_ft
|
|
752
|
+
|
|
753
|
+
rows = [(float(d), float(p), float(t)) for d, p, t in zip(depths_ft, pres_psi, temp_F)
|
|
754
|
+
if not (np.isnan(p) or np.isnan(t))]
|
|
755
|
+
result = {
|
|
756
|
+
'stations': len(rows),
|
|
757
|
+
'depth_range_ft': [round(rows[0][0], 1), round(rows[-1][0], 1)] if rows else None,
|
|
758
|
+
'derivation': derivation,
|
|
759
|
+
'temperature_curve_used': temp_ch,
|
|
760
|
+
'pressure_curve_used': pres_ch,
|
|
761
|
+
'source_stream': stream.name,
|
|
762
|
+
'note': ('REAL depth stations; T/P filled from DECLARED gradients - labeled, not measured'
|
|
763
|
+
if derivation == 'DERIVED_GRADIENTS' else
|
|
764
|
+
'measured curves converted; gaps dropped'),
|
|
765
|
+
}
|
|
766
|
+
if out_csv is not None:
|
|
767
|
+
p = Path(out_csv)
|
|
768
|
+
with p.open('w', encoding='utf-8', newline='') as f:
|
|
769
|
+
f.write('depth_ft,pressure_psi,temp_F\n')
|
|
770
|
+
for d, pr, t in rows:
|
|
771
|
+
f.write(f'{d:.1f},{pr:.1f},{t:.1f}\n')
|
|
772
|
+
result['csv'] = str(p)
|
|
773
|
+
return result
|