integrate_module 0.99.3__tar.gz → 0.99.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {integrate_module-0.99.3/integrate_module.egg-info → integrate_module-0.99.6}/PKG-INFO +10 -1
  2. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/__init__.py +2 -0
  3. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate.py +84 -1
  4. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_borehole.py +96 -64
  5. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_io.py +95 -12
  6. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_plot.py +195 -127
  7. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_rejection.py +131 -37
  8. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_rejection_jax.py +29 -6
  9. integrate_module-0.99.6/integrate/mlmapping.py +1110 -0
  10. {integrate_module-0.99.3 → integrate_module-0.99.6/integrate_module.egg-info}/PKG-INFO +10 -1
  11. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate_module.egg-info/SOURCES.txt +1 -0
  12. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate_module.egg-info/requires.txt +12 -0
  13. {integrate_module-0.99.3 → integrate_module-0.99.6}/pyproject.toml +13 -1
  14. {integrate_module-0.99.3 → integrate_module-0.99.6}/LICENSE +0 -0
  15. {integrate_module-0.99.3 → integrate_module-0.99.6}/README.md +0 -0
  16. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/gex.py +0 -0
  17. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_hdf5_info_cli.py +0 -0
  18. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_query.py +0 -0
  19. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_rejection_cli.py +0 -0
  20. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_timing_cli.py +0 -0
  21. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate/integrate_www_cli.py +0 -0
  22. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate_module.egg-info/dependency_links.txt +0 -0
  23. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate_module.egg-info/entry_points.txt +0 -0
  24. {integrate_module-0.99.3 → integrate_module-0.99.6}/integrate_module.egg-info/top_level.txt +0 -0
  25. {integrate_module-0.99.3 → integrate_module-0.99.6}/setup.cfg +0 -0
  26. {integrate_module-0.99.3 → integrate_module-0.99.6}/tests/test_likelihood_multinomial.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: integrate_module
3
- Version: 0.99.3
3
+ Version: 0.99.6
4
4
  Summary: Localized probabilistic data integration
5
5
  Author-email: Thomas Mejer Hansen <tmeha@geo.au.dk>
6
6
  License: MIT
@@ -34,6 +34,13 @@ Requires-Dist: jax
34
34
  Provides-Extra: dev
35
35
  Requires-Dist: pytest; extra == "dev"
36
36
  Requires-Dist: black; extra == "dev"
37
+ Provides-Extra: ml
38
+ Requires-Dist: tensorflow; extra == "ml"
39
+ Requires-Dist: scikit-learn; extra == "ml"
40
+ Requires-Dist: keras_tuner; extra == "ml"
41
+ Provides-Extra: examples
42
+ Requires-Dist: geopandas; extra == "examples"
43
+ Requires-Dist: shapely; extra == "examples"
37
44
  Provides-Extra: docs
38
45
  Requires-Dist: sphinx; extra == "docs"
39
46
  Requires-Dist: nbsphinx; extra == "docs"
@@ -43,6 +50,8 @@ Requires-Dist: myst-parser; extra == "docs"
43
50
  Requires-Dist: sphinx-rtd-theme; extra == "docs"
44
51
  Requires-Dist: furo; extra == "docs"
45
52
  Requires-Dist: tomli; python_version < "3.11" and extra == "docs"
53
+ Provides-Extra: jax-cuda
54
+ Requires-Dist: jax[cuda12]; extra == "jax-cuda"
46
55
  Dynamic: license-file
47
56
 
48
57
  # INTEGRATE Python Module
@@ -36,6 +36,7 @@ from integrate.integrate import posterior_cumulative_thickness
36
36
  from integrate.integrate import use_parallel
37
37
  from integrate.integrate import kl_divergence
38
38
  from integrate.integrate import entropy
39
+ from integrate.integrate import discrete_data_entropy
39
40
  from integrate.integrate import class_id_to_idx
40
41
  from integrate.integrate import is_notebook
41
42
  from integrate.integrate import get_hypothesis_probability
@@ -94,6 +95,7 @@ from integrate.integrate_plot import plot_profile_discrete
94
95
  from integrate.integrate_plot import plot_cumulative_probability_profile
95
96
  from integrate.integrate_plot import plot_T_EV
96
97
  from integrate.integrate_plot import plot_data_xy
98
+ from integrate.integrate_plot import plot_discrete_data_entropy
97
99
  from integrate.integrate_plot import plot_data
98
100
  from integrate.integrate_plot import plot_data_prior_post
99
101
  from integrate.integrate_plot import plot_data_prior
@@ -2963,12 +2963,95 @@ def entropy(P, base = None):
2963
2963
  >>> entropy(P)
2964
2964
  array([1.0, 0.469])
2965
2965
  """
2966
- P = np.atleast_2d(P)
2966
+ P = np.atleast_2d(P)
2967
2967
  if base is None:
2968
2968
  base = P.shape[1]
2969
2969
  H = -np.sum(P*np.log(P)/np.log(base), axis=1)
2970
2970
  return H
2971
2971
 
2972
+
2973
+ def discrete_data_entropy(f_data_h5, id_list, depth_reduce='min', showInfo=1):
2974
+ """
2975
+ Compute the pointwise (per survey location) entropy of one or more
2976
+ multinomial discrete /D{id} data entries in a DATA HDF5 file.
2977
+
2978
+ Each location's discrete observation is a probability-over-classes
2979
+ profile spanning ``nm`` depth layers (shape ``(ns, nclass, nm)`` per id,
2980
+ as written by :func:`save_data_multinomial`, e.g. via
2981
+ :func:`save_borehole_data`). Entropy is computed per (location,
2982
+ depth-layer) using ``scipy.stats.entropy`` (which correctly handles
2983
+ exact-zero probabilities, unlike :func:`entropy`), the depth axis is
2984
+ then collapsed per ``depth_reduce``, and — if more than one id is given —
2985
+ the pointwise minimum across ids is returned (i.e. the best-informed id
2986
+ at each location; adding more ids can only lower or keep equal the
2987
+ entropy at a given location, never raise it).
2988
+
2989
+ Parameters
2990
+ ----------
2991
+ f_data_h5 : str
2992
+ Path to the DATA HDF5 file.
2993
+ id_list : int or list of int
2994
+ One or more dataset ids referencing multinomial /D{id} groups
2995
+ (e.g. from ``save_borehole_data()``'s ``id_out`` / ``id_borehole_list``).
2996
+ depth_reduce : {'min', 'mean'}, optional
2997
+ How to collapse the per-location depth-layer axis, per id, before
2998
+ combining across ids. ``'min'`` (default) takes the lowest (most
2999
+ informative) entropy value across depth layers, i.e. each id's best
3000
+ depth-layer represents it at that location; ``'mean'`` averages
3001
+ entropy across depth layers instead.
3002
+ showInfo : int, optional
3003
+ Verbosity level passed through to :func:`load_data`. Default 1.
3004
+
3005
+ Returns
3006
+ -------
3007
+ H : ndarray, shape (ns,)
3008
+ Pointwise entropy in [0, 1] (base = nclass). NaN at locations not
3009
+ covered by any of the given ids (``i_use == 0`` for all of them).
3010
+
3011
+ Examples
3012
+ --------
3013
+ >>> H = ig.discrete_data_entropy(f_data_h5, id_borehole_list)
3014
+ >>> H = ig.discrete_data_entropy(f_data_h5, id_borehole_list, depth_reduce='mean')
3015
+ """
3016
+ import scipy as sp
3017
+ import warnings
3018
+ import integrate as ig
3019
+
3020
+ if not isinstance(id_list, list):
3021
+ id_list = [id_list]
3022
+ if depth_reduce not in ('mean', 'min'):
3023
+ raise ValueError("depth_reduce must be 'mean' or 'min'")
3024
+
3025
+ DATA = ig.load_data(f_data_h5, id_arr=id_list, showInfo=showInfo)
3026
+
3027
+ H_per_id = []
3028
+ for i, id in enumerate(id_list):
3029
+ if DATA['noise_model'][i] != 'multinomial':
3030
+ raise ValueError(
3031
+ "D%d is not a multinomial dataset (noise_model=%r)" %
3032
+ (id, DATA['noise_model'][i]))
3033
+ P = DATA['d_obs'][i] # (ns, nclass, nm)
3034
+ i_use = np.asarray(DATA['i_use'][i]).ravel().astype(bool)
3035
+ ns, nclass, nm = P.shape
3036
+
3037
+ H = np.full((ns, nm), np.nan)
3038
+ if np.any(i_use):
3039
+ H[i_use, :] = sp.stats.entropy(
3040
+ P[i_use].transpose(1, 0, 2).reshape(nclass, -1),
3041
+ base=nclass
3042
+ ).reshape(-1, nm)
3043
+
3044
+ with warnings.catch_warnings():
3045
+ warnings.simplefilter('ignore', category=RuntimeWarning)
3046
+ H_id = np.nanmean(H, axis=1) if depth_reduce == 'mean' else np.nanmin(H, axis=1)
3047
+ H_per_id.append(H_id)
3048
+
3049
+ with warnings.catch_warnings():
3050
+ warnings.simplefilter('ignore', category=RuntimeWarning)
3051
+ H_map = np.nanmin(np.vstack(H_per_id), axis=0)
3052
+ return H_map
3053
+
3054
+
2972
3055
  def class_id_to_idx(D, class_id=None):
2973
3056
  """
2974
3057
  Convert class identifiers to indices.
@@ -572,8 +572,8 @@ def rescale_P_obs_temperature(P_obs, T=1.0):
572
572
 
573
573
  return P_obs_scaled
574
574
 
575
- def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
576
- nan_freq=0.8, r_data_i_use=None):
575
+ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, range_data=10, range_xyz=100, doPlot=False,
576
+ range_data_nan_freq=0.8, range_data_i_use=None):
577
577
  """
578
578
  Convert point-based discrete probability observations to gridded data with distance-based weighting.
579
579
 
@@ -594,23 +594,23 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
594
594
  Y coordinate (e.g., UTM Northing) of the observation point.
595
595
  f_data_h5 : str
596
596
  Path to HDF5 data file containing survey geometry (X, Y coordinates).
597
- r_data : float, optional
597
+ range_data : float, optional
598
598
  Inner radius in meters within which observations have full strength.
599
599
  Default is 10 meters.
600
- r_dis : float, optional
600
+ range_xyz : float, optional
601
601
  Outer radius in meters for distance-based weighting. Beyond this distance,
602
602
  observations are fully attenuated (temperature → ∞). Default is 100 meters.
603
603
  doPlot : bool, optional
604
604
  If True, creates diagnostic plots showing weight distributions.
605
605
  Default is False.
606
- nan_freq : float, optional
606
+ range_data_nan_freq : float, optional
607
607
  NaN-frequency threshold for automatic data-gate selection inside
608
608
  :func:`get_weight_from_position`. Gates where the fraction of
609
609
  non-NaN values is below this threshold are excluded. Default 0.8.
610
- Ignored when ``r_data_i_use`` is provided.
611
- r_data_i_use : array-like of int or None, optional
610
+ Ignored when ``range_data_i_use`` is provided.
611
+ range_data_i_use : array-like of int or None, optional
612
612
  Explicit gate/channel indices to use for data-distance computation
613
- inside :func:`get_weight_from_position`. Overrides ``nan_freq`` when
613
+ inside :func:`get_weight_from_position`. Overrides ``range_data_nan_freq`` when
614
614
  provided. Default None.
615
615
 
616
616
  Returns
@@ -649,7 +649,7 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
649
649
  >>> P_obs = compute_P_obs_discrete(depth_top, depth_bottom, lithology, z, class_id)
650
650
  >>> X_well, Y_well = 543000.0, 6175800.0
651
651
  >>> d_obs, i_use, T_all = Pobs_to_datagrid(P_obs, X_well, Y_well, 'survey_data.h5',
652
- ... r_data=10, r_dis=100)
652
+ ... range_data=10, range_xyz=100)
653
653
  >>> # Write to data file
654
654
  >>> ig.save_data_multinomial(d_obs, i_use=i_use, id=2, f_data_h5='survey_data.h5')
655
655
 
@@ -673,8 +673,8 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
673
673
 
674
674
  # Compute distance-based weights for all grid points
675
675
  w_combined, w_dis, w_data, i_use_from_func = ig.get_weight_from_position(
676
- f_data_h5, X, Y, r_data=r_data, r_dis=r_dis, doPlot=doPlot,
677
- nan_freq=nan_freq, r_data_i_use=r_data_i_use
676
+ f_data_h5, X, Y, range_data=range_data, range_xyz=range_xyz, doPlot=doPlot,
677
+ range_data_nan_freq=range_data_nan_freq, range_data_i_use=range_data_i_use
678
678
  )
679
679
 
680
680
  # Convert distance weight to temperature
@@ -703,9 +703,9 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
703
703
 
704
704
 
705
705
 
706
- def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400, r_data=2,
706
+ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, range_xyz=400, range_data=2,
707
707
  useLog=True, doPlot=False, plFile=None, showInfo=0,
708
- nan_freq=0.8, r_data_i_use=None):
708
+ range_data_nan_freq=0.8, range_data_i_use=None):
709
709
  """Calculate weights based on distance and data similarity to a reference point.
710
710
 
711
711
  This function computes three sets of weights:
@@ -723,9 +723,9 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
723
723
  Y coordinate of reference point (well). Default 0.
724
724
  i_ref : int, optional
725
725
  Index of reference point. Default -1 (auto-calculated as closest to x_well, y_well).
726
- r_dis : float, optional
726
+ range_xyz : float, optional
727
727
  Geographic XY distance range [m] for spatial weighting. Default 400.
728
- r_data : float, optional
728
+ range_data : float, optional
729
729
  Data-space similarity range parameter for data weighting. Default 2.
730
730
  useLog : bool, optional
731
731
  Apply log10 transform to data before computing similarity. Default True.
@@ -735,16 +735,16 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
735
735
  Output filename for the diagnostic plot. Auto-generated if None.
736
736
  showInfo : int, optional
737
737
  Verbosity level. Default 0.
738
- nan_freq : float, optional
738
+ range_data_nan_freq : float, optional
739
739
  NaN-frequency threshold for automatic gate selection. Gates where the
740
740
  fraction of non-NaN values across all soundings is below this threshold
741
741
  are excluded from the data-distance computation. Default 0.8.
742
- Ignored when ``r_data_i_use`` is provided.
743
- r_data_i_use : array-like of int or None, optional
742
+ Ignored when ``range_data_i_use`` is provided.
743
+ range_data_i_use : array-like of int or None, optional
744
744
  Explicit gate/channel indices to use for the data-distance computation.
745
- When provided, overrides the ``nan_freq`` automatic selection.
745
+ When provided, overrides the ``range_data_nan_freq`` automatic selection.
746
746
  A NaN check at the reference sounding is still applied.
747
- Default None (use ``nan_freq`` threshold).
747
+ Default None (use ``range_data_nan_freq`` threshold).
748
748
 
749
749
  Returns
750
750
  -------
@@ -760,8 +760,8 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
760
760
  Notes
761
761
  -----
762
762
  Weights are calculated using Gaussian functions:
763
- - Distance weights: exp(-dis² / r_dis²)
764
- - Data weights: exp(-sum_dd² / r_data²)
763
+ - Distance weights: exp(-dis² / range_xyz²)
764
+ - Data weights: exp(-sum_dd² / range_data²)
765
765
  where dis is geographic distance and sum_dd is cumulative data difference.
766
766
  """
767
767
  import integrate as ig
@@ -777,12 +777,12 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
777
777
  i_ref = np.argmin((X-x_well)**2 + (Y-y_well)**2)
778
778
 
779
779
  # Select gates to use for data-distance computation
780
- if r_data_i_use is not None:
781
- gates = np.asarray(r_data_i_use, dtype=int)
780
+ if range_data_i_use is not None:
781
+ gates = np.asarray(range_data_i_use, dtype=int)
782
782
  else:
783
783
  n_not_nan = np.sum(~np.isnan(d_obs), axis=0)
784
784
  n_not_nan_freq = n_not_nan / d_obs.shape[0]
785
- gates = np.where(n_not_nan_freq > nan_freq)[0]
785
+ gates = np.where(n_not_nan_freq > range_data_nan_freq)[0]
786
786
  # Remove gates that are NaN at the reference sounding
787
787
  gates = gates[~np.isnan(d_obs[i_ref, gates])]
788
788
  i_use = gates
@@ -795,12 +795,12 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
795
795
  d_test =d_obs[:,i_use]
796
796
  dd = np.abs(d_test - d_ref)
797
797
  sum_dd = np.sum(dd, axis=1)
798
- w_data = np.exp(-1*sum_dd**2/r_data**2)
799
-
798
+ w_data = np.exp(-1*sum_dd**2/range_data**2)
799
+
800
800
 
801
801
  # Compute the distance from each data point to the actual borehole location
802
802
  dis = np.sqrt((X-x_well)**2 + (Y-y_well)**2)
803
- w_dis = np.exp(-1*dis**2/r_dis**2)
803
+ w_dis = np.exp(-1*dis**2/range_xyz**2)
804
804
 
805
805
  w_combined = w_data * w_dis
806
806
 
@@ -838,7 +838,7 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
838
838
  plt.xlabel('X')
839
839
  plt.ylabel('Y')
840
840
  if plFile is None:
841
- plFile = 'weights_%d_%d_%d_rdis%d_rdata%d.png' % (x_well,y_well,i_ref,r_dis,r_data)
841
+ plFile = 'weights_%d_%d_%d_rdis%d_rdata%d.png' % (x_well,y_well,i_ref,range_xyz,range_data)
842
842
  plt.savefig(plFile, dpi=300)
843
843
 
844
844
  return w_combined, w_dis, w_data, i_ref
@@ -1023,46 +1023,54 @@ def save_borehole_data(f_prior_h5, f_data_h5, BH, **kwargs):
1023
1023
  Path to the prior HDF5 file.
1024
1024
  f_data_h5 : str
1025
1025
  Path to the observed-data HDF5 file.
1026
- BH : dict
1026
+ BH : dict or list of dict
1027
1027
  Borehole dictionary with keys depth_top, depth_bottom, class_obs,
1028
1028
  class_prob, X, Y, name, method. Two optional keys control the
1029
- distance-weighting radii when r_data / r_dis are not passed as
1030
- explicit kwargs:
1029
+ distance-weighting radii when range_data / range_xyz are not passed
1030
+ as explicit kwargs:
1031
1031
 
1032
1032
  * ``range_data`` (float, optional) — data-space similarity radius.
1033
1033
  Survey points whose EM data response is similar to the borehole
1034
1034
  location receive higher weight; points that are more dissimilar are
1035
- down-weighted. Default: 1,000,000 (effectively no cutoff).
1036
- * ``range_dis`` (float, optional) — geographic XY distance [m] beyond
1035
+ down-weighted. Only used if present and non-negative; otherwise the
1036
+ default applies. Default: 1,000,000 (effectively no cutoff).
1037
+ * ``range_xyz`` (float, optional) — geographic XY distance [m] beyond
1037
1038
  which the borehole exerts no influence on nearby survey points.
1038
- Default: 300 m.
1039
- * ``nan_freq`` (float, optional) — NaN-frequency threshold for automatic
1039
+ Only used if present and non-negative; otherwise the default
1040
+ applies. Default: 300 m.
1041
+ * ``range_data_nan_freq`` (float, optional) — NaN-frequency threshold for automatic
1040
1042
  data-gate selection. Default: 0.8.
1041
- * ``r_data_i_use`` (list of int, optional) — explicit gate indices for
1042
- data-distance computation; overrides ``nan_freq`` when provided.
1043
+ * ``range_data_i_use`` (list of int, optional) — explicit gate indices for
1044
+ data-distance computation; overrides ``range_data_nan_freq`` when provided.
1045
+
1046
+ If ``BH`` is a list of dicts, each borehole is processed in turn
1047
+ (one call per element, in order) using the same ``**kwargs`` for
1048
+ every borehole; a per-borehole ``range_data``/``range_xyz``/etc. key
1049
+ set inside an individual dict still takes effect for that borehole
1050
+ exactly as in the single-dict case. See ``Returns`` below.
1043
1051
  **kwargs
1044
1052
  im_prior : int, optional
1045
1053
  Index of the discrete model parameter in f_prior_h5 (e.g. 2 for /M2).
1046
1054
  Default is 2.
1047
1055
  parallel : bool, optional
1048
1056
  Enable parallel mode computation. Default is False.
1049
- r_data : float, optional
1057
+ range_data : float, optional
1050
1058
  Data-space similarity radius. Overrides ``BH['range_data']`` when
1051
- provided. Resolution order: explicit kwarg > BH['range_data'] >
1052
- 1,000,000 (no cutoff).
1053
- r_dis : float, optional
1054
- Geographic XY fade-out distance [m]. Overrides ``BH['range_dis']``
1055
- when provided. Resolution order: explicit kwarg > BH['range_dis'] >
1056
- 300 m.
1057
- nan_freq : float, optional
1059
+ provided. Resolution order: explicit kwarg > BH['range_data']
1060
+ (if present and non-negative) > 1,000,000 (no cutoff).
1061
+ range_xyz : float, optional
1062
+ Geographic XY fade-out distance [m]. Overrides ``BH['range_xyz']``
1063
+ when provided. Resolution order: explicit kwarg > BH['range_xyz']
1064
+ (if present and non-negative) > 300 m.
1065
+ range_data_nan_freq : float, optional
1058
1066
  NaN-frequency threshold for automatic data-gate selection. Gates
1059
1067
  where the fraction of non-NaN soundings is below this threshold are
1060
1068
  excluded from data-distance computation. Resolution order: explicit
1061
- kwarg > BH['nan_freq'] > 0.8. Ignored when ``r_data_i_use`` is set.
1062
- r_data_i_use : list of int or None, optional
1069
+ kwarg > BH['range_data_nan_freq'] > 0.8. Ignored when ``range_data_i_use`` is set.
1070
+ range_data_i_use : list of int or None, optional
1063
1071
  Explicit gate/channel indices to use for data-distance computation.
1064
- Overrides ``nan_freq`` when provided. Resolution order: explicit
1065
- kwarg > BH['r_data_i_use'] > None.
1072
+ Overrides ``range_data_nan_freq`` when provided. Resolution order: explicit
1073
+ kwarg > BH['range_data_i_use'] > None.
1066
1074
  doPlot : bool, optional
1067
1075
  Plot distance-weight maps. Default is False.
1068
1076
  showInfo : int, optional
@@ -1072,32 +1080,56 @@ def save_borehole_data(f_prior_h5, f_data_h5, BH, **kwargs):
1072
1080
  Returns
1073
1081
  -------
1074
1082
  id_prior : int
1075
- Dataset index of the new /D entry added to f_prior_h5.
1083
+ If ``BH`` is a single dict: dataset index of the new /D entry added
1084
+ to f_prior_h5.
1076
1085
  id_out : int
1077
- Dataset index of the new /D entry added to f_data_h5.
1086
+ If ``BH`` is a single dict: dataset index of the new /D entry added
1087
+ to f_data_h5.
1088
+ id_prior_list : list of int
1089
+ If ``BH`` is a list of dicts: dataset indices of the new /D entries
1090
+ added to f_prior_h5, one per borehole, same order as ``BH``.
1091
+ id_borehole_list : list of int
1092
+ If ``BH`` is a list of dicts: dataset indices of the new /D entries
1093
+ added to f_data_h5, one per borehole, same order as ``BH``.
1078
1094
 
1079
1095
  Examples
1080
1096
  --------
1081
1097
  >>> # Single borehole
1082
1098
  >>> id_prior, id_data = ig.save_borehole_data(f_prior_h5, f_data_h5, BH)
1083
1099
 
1084
- >>> # All boreholes — collect data IDs for joint inversion
1085
- >>> id_borehole_list = []
1086
- >>> for BH in BHOLES:
1087
- ... _, id_out = ig.save_borehole_data(f_prior_h5, f_data_h5, BH,
1088
- ... r_data=2, r_dis=300, parallel=True)
1089
- ... id_borehole_list.append(id_out)
1100
+ >>> # All boreholes in one call — collect data IDs for joint inversion
1101
+ >>> id_prior_list, id_borehole_list = ig.save_borehole_data(
1102
+ ... f_prior_h5, f_data_h5, BHOLES,
1103
+ ... range_data=2, range_xyz=300, parallel=True)
1090
1104
  >>> f_post_h5 = ig.integrate_rejection(f_prior_h5, f_data_h5,
1091
1105
  ... id_use=[1] + id_borehole_list)
1092
1106
  """
1107
+ if isinstance(BH, list):
1108
+ id_prior_list = []
1109
+ id_borehole_list = []
1110
+ for bh in BH:
1111
+ id_prior, id_out = save_borehole_data(f_prior_h5, f_data_h5, bh, **kwargs)
1112
+ id_prior_list.append(id_prior)
1113
+ id_borehole_list.append(id_out)
1114
+ return id_prior_list, id_borehole_list
1115
+
1093
1116
  import integrate as ig
1094
1117
 
1118
+ def _resolve_radius(name, default):
1119
+ val = kwargs.get(name, None)
1120
+ if val is not None:
1121
+ return val
1122
+ val = BH.get(name, None)
1123
+ if val is not None and val >= 0:
1124
+ return val
1125
+ return default
1126
+
1095
1127
  im_prior = kwargs.get('im_prior', 2)
1096
1128
  parallel = kwargs.get('parallel', False)
1097
- r_data = kwargs.get('r_data', BH.get('range_data', 1_000_000))
1098
- r_dis = kwargs.get('r_dis', BH.get('range_dis', 300))
1099
- nan_freq = kwargs.get('nan_freq', BH.get('nan_freq', 0.8))
1100
- r_data_i_use = kwargs.get('r_data_i_use', BH.get('r_data_i_use', None))
1129
+ range_data = _resolve_radius('range_data', 1_000_000)
1130
+ range_xyz = _resolve_radius('range_xyz', 300)
1131
+ range_data_nan_freq = kwargs.get('range_data_nan_freq', BH.get('range_data_nan_freq', 0.8))
1132
+ range_data_i_use = kwargs.get('range_data_i_use', BH.get('range_data_i_use', None))
1101
1133
  doPlot = kwargs.get('doPlot', False)
1102
1134
  showInfo = kwargs.get('showInfo', 1)
1103
1135
 
@@ -1109,8 +1141,8 @@ def save_borehole_data(f_prior_h5, f_data_h5, BH, **kwargs):
1109
1141
  # Step 2: extrapolate point observations to the survey grid
1110
1142
  d_obs, i_use, T_use = Pobs_to_datagrid(
1111
1143
  P_obs, BH['X'], BH['Y'], f_data_h5,
1112
- r_data=r_data, r_dis=r_dis, doPlot=doPlot,
1113
- nan_freq=nan_freq, r_data_i_use=r_data_i_use)
1144
+ range_data=range_data, range_xyz=range_xyz, doPlot=doPlot,
1145
+ range_data_nan_freq=range_data_nan_freq, range_data_i_use=range_data_i_use)
1114
1146
 
1115
1147
  # Step 3: save gridded observations to f_data_h5
1116
1148
  id_out, _ = ig.save_data_multinomial(
@@ -3240,7 +3240,7 @@ def save_data_gaussian(D_obs, D_std = [], d_std=[], Cd=[], id=1, id_prior=None,
3240
3240
  return f_data_h5
3241
3241
 
3242
3242
 
3243
- def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None, nan_value=None, showInfo=0, disregardFullNan=True, data_obs=None, data_std=None):
3243
+ def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None, nan_value=None, showInfo=0, disregardFullNan=True, data_obs=None, data_std=None, altitude=None, altitude_std=None, tx_altitude=None, tx_altitude_std=None, rx_altitude=None, rx_altitude_std=None):
3244
3244
  """
3245
3245
  Convert Aarhus Workbench XYZ export file(s) to an INTEGRATE HDF5 data file.
3246
3246
 
@@ -3292,6 +3292,38 @@ def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None
3292
3292
  same length as ``data_obs``. Use ``None`` for an individual entry to
3293
3293
  fall back to ``0.05 * |d_obs|`` for that column. If the whole
3294
3294
  parameter is omitted, all columns default to ``0.05 * |d_obs|``.
3295
+ altitude : str, optional
3296
+ Flightlines column name (case-insensitive) holding the platform's
3297
+ flight altitude/height, e.g. ``'Alt'``. When given, written as its
3298
+ own Gaussian data block with ``id=2`` (the second dataset, after the
3299
+ ``/D1`` dbdt data). Any ``data_obs`` columns are then written
3300
+ starting at ``id=3`` instead of ``id=2``.
3301
+ altitude_std : str, float, or None, optional
3302
+ Uncertainty for ``altitude``.
3303
+ - A **string** is treated as another flightlines column name
3304
+ (case-insensitive) holding the absolute std directly.
3305
+ - A **number** with ``abs(altitude_std) < 1`` is treated as a
3306
+ *relative* std: ``std = altitude_std * altitude``.
3307
+ - A **number** with ``abs(altitude_std) >= 1`` is treated as an
3308
+ *absolute* std in meters, constant for all soundings.
3309
+ - If ``None`` (default), falls back to ``0.05 * |altitude|``.
3310
+ tx_altitude : str, optional
3311
+ Flightlines column name (case-insensitive) holding the transmitter
3312
+ altitude/height. When given, written as its own Gaussian data block,
3313
+ immediately after ``altitude`` (if also given). Optional — omitted
3314
+ entirely if not given.
3315
+ tx_altitude_std : str, float, or None, optional
3316
+ Uncertainty for ``tx_altitude``. Same rules as ``altitude_std``
3317
+ (string column name / relative number / absolute number / default
3318
+ 5% relative if ``None``). Only used if ``tx_altitude`` is given.
3319
+ rx_altitude : str, optional
3320
+ Flightlines column name (case-insensitive) holding the receiver
3321
+ altitude/height. When given, written as its own Gaussian data block,
3322
+ after ``altitude`` and ``tx_altitude`` (if also given). Optional —
3323
+ omitted entirely if not given.
3324
+ rx_altitude_std : str, float, or None, optional
3325
+ Uncertainty for ``rx_altitude``. Same rules as ``altitude_std``.
3326
+ Only used if ``rx_altitude`` is given.
3295
3327
 
3296
3328
  Returns
3297
3329
  -------
@@ -3380,9 +3412,22 @@ def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None
3380
3412
  ld = {k: pd.concat([xyz.layer_data[k] for xyz in xyz_list], ignore_index=True)
3381
3413
  for k in xyz_list[0].layer_data}
3382
3414
 
3383
- # Handle XYZ files that use 'x'/'y' instead of 'utmx'/'utmy'
3415
+ # Handle XYZ files that use alternate column names for geometry
3416
+ # (e.g. tTEM: utmx/utmy/line_no/elevation, SkyTEM: e/n/line/dem)
3384
3417
  if 'utmx' not in fl.columns and 'x' in fl.columns:
3385
3418
  fl = fl.rename(columns={'x': 'utmx', 'y': 'utmy'})
3419
+ if 'utmx' not in fl.columns and 'e' in fl.columns:
3420
+ fl = fl.rename(columns={'e': 'utmx', 'n': 'utmy'})
3421
+ if 'line_no' not in fl.columns and 'line' in fl.columns:
3422
+ fl = fl.rename(columns={'line': 'line_no'})
3423
+ if 'elevation' not in fl.columns and 'dem' in fl.columns:
3424
+ fl = fl.rename(columns={'dem': 'elevation'})
3425
+
3426
+ # Handle single-channel XYZ files (e.g. SkyTEM) that store the sounding
3427
+ # data under a plain component name instead of the tTEM 'ch1gt' naming
3428
+ if 'dbdt_ch1gt' not in ld and 'z_dbdt' in ld:
3429
+ ld['dbdt_ch1gt'] = ld['z_dbdt']
3430
+ ld['dbdt_std_ch1gt'] = ld['relunc_z_dbdt']
3386
3431
 
3387
3432
  # Determine dummy/missing value: explicit arg > XYZ header > fallback 9999
3388
3433
  if nan_value is None:
@@ -3393,7 +3438,12 @@ def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None
3393
3438
  # --- pair ch1 / ch2 rows (mirrors MATLAB logic) ---
3394
3439
  # Every ch1 row becomes a sounding. HM data is filled where the
3395
3440
  # immediately following row is ch2; otherwise those columns stay NaN.
3396
- channel_arr = fl['channel_no'].values
3441
+ # Single-channel systems (e.g. SkyTEM) have no 'channel_no' column at
3442
+ # all: every row is its own (channel-1) sounding.
3443
+ if 'channel_no' in fl.columns:
3444
+ channel_arr = fl['channel_no'].values
3445
+ else:
3446
+ channel_arr = np.ones(len(fl))
3397
3447
  ch1_pos = np.where(channel_arr == 1)[0]
3398
3448
 
3399
3449
  # geometry from channel-1 rows (all of them)
@@ -3475,7 +3525,38 @@ def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None
3475
3525
  if n_channels >= 2:
3476
3526
  hf.create_dataset('/D1/i_hm', data=np.arange(i_hm_start, i_hm_end))
3477
3527
 
3478
- # --- write additional data columns as D2, D3, ... ---
3528
+ # --- write altitude / rx_altitude / tx_altitude (if given) as their own Gaussian data blocks ---
3529
+ def _resolve_std(obs, std_arg):
3530
+ if isinstance(std_arg, str):
3531
+ return fl[std_arg.lower()].values[ch1_pos][keep].reshape(-1, 1).astype(float)
3532
+ elif isinstance(std_arg, (int, float)):
3533
+ if abs(std_arg) < 1:
3534
+ return std_arg * np.abs(obs) # relative
3535
+ else:
3536
+ return np.full_like(obs, float(std_arg)) # absolute, meters
3537
+ else:
3538
+ return 0.05 * np.abs(obs) # default: 5% relative
3539
+
3540
+ next_id = 2
3541
+ for col, col_std, name in (
3542
+ (altitude, altitude_std, 'Altitude'),
3543
+ (tx_altitude, tx_altitude_std, 'Tx_altitude'),
3544
+ (rx_altitude, rx_altitude_std, 'Rx_altitude'),
3545
+ ):
3546
+ if col is not None:
3547
+ obs = fl[col.lower()].values[ch1_pos][keep].reshape(-1, 1).astype(float)
3548
+ std = _resolve_std(obs, col_std)
3549
+ save_data_gaussian(
3550
+ obs, D_std=std,
3551
+ f_data_h5=f_data_h5,
3552
+ id=next_id,
3553
+ name=name,
3554
+ delete_if_exist=False,
3555
+ showInfo=showInfo,
3556
+ )
3557
+ next_id += 1
3558
+
3559
+ # --- write additional data columns as D2, D3, ... (or shifted if altitude present) ---
3479
3560
  if data_obs is not None:
3480
3561
  _data_std = data_std if data_std is not None else [None] * len(data_obs)
3481
3562
  for i, col_obs in enumerate(data_obs):
@@ -3488,7 +3569,7 @@ def xyz_to_h5(file_xyz, file_gex, f_data_h5=None, i_lm_skip=None, i_hm_skip=None
3488
3569
  save_data_gaussian(
3489
3570
  obs, D_std=std,
3490
3571
  f_data_h5=f_data_h5,
3491
- id=i + 2,
3572
+ id=next_id + i,
3492
3573
  name=col_obs,
3493
3574
  delete_if_exist=False,
3494
3575
  showInfo=showInfo,
@@ -5198,15 +5279,17 @@ def write_borehole(W, filename, **kwargs):
5198
5279
  Used only by :func:`plot_boreholes` to place the well on a shared
5199
5280
  elevation axis. Has no effect on inversion.
5200
5281
  * ``range_data`` (float, optional) – data-space similarity radius used by
5201
- :func:`save_borehole_data` when ``r_data`` is not passed explicitly.
5282
+ :func:`save_borehole_data` when ``range_data`` is not passed explicitly.
5283
+ Only used if present and non-negative, otherwise the default applies.
5202
5284
  Default: 1,000,000 (no cutoff).
5203
- * ``range_dis`` (float, optional) – geographic XY fade-out distance [m]
5204
- used by :func:`save_borehole_data` when ``r_dis`` is not passed explicitly.
5205
- Default: 300 m.
5206
- * ``nan_freq`` (float, optional) – NaN-frequency threshold for automatic
5285
+ * ``range_xyz`` (float, optional) – geographic XY fade-out distance [m]
5286
+ used by :func:`save_borehole_data` when ``range_xyz`` is not passed
5287
+ explicitly. Only used if present and non-negative, otherwise the
5288
+ default applies. Default: 300 m.
5289
+ * ``range_data_nan_freq`` (float, optional) – NaN-frequency threshold for automatic
5207
5290
  data-gate selection in :func:`save_borehole_data`. Default: 0.8.
5208
- * ``r_data_i_use`` (list of int, optional) – explicit gate/channel indices
5209
- for data-distance computation; overrides ``nan_freq`` when provided.
5291
+ * ``range_data_i_use`` (list of int, optional) – explicit gate/channel indices
5292
+ for data-distance computation; overrides ``range_data_nan_freq`` when provided.
5210
5293
 
5211
5294
  numpy arrays and scalars are automatically converted to plain
5212
5295
  Python lists/numbers so the file is human-readable JSON.