integrate_module 0.99.2__tar.gz → 0.99.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {integrate_module-0.99.2/integrate_module.egg-info → integrate_module-0.99.5}/PKG-INFO +1 -1
  2. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/__init__.py +2 -0
  3. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate.py +84 -1
  4. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_borehole.py +96 -64
  5. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_io.py +29 -8
  6. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_plot.py +58 -0
  7. {integrate_module-0.99.2 → integrate_module-0.99.5/integrate_module.egg-info}/PKG-INFO +1 -1
  8. {integrate_module-0.99.2 → integrate_module-0.99.5}/pyproject.toml +1 -1
  9. {integrate_module-0.99.2 → integrate_module-0.99.5}/LICENSE +0 -0
  10. {integrate_module-0.99.2 → integrate_module-0.99.5}/README.md +0 -0
  11. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/gex.py +0 -0
  12. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_hdf5_info_cli.py +0 -0
  13. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_query.py +0 -0
  14. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_rejection.py +0 -0
  15. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_rejection_cli.py +0 -0
  16. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_rejection_jax.py +0 -0
  17. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_timing_cli.py +0 -0
  18. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate/integrate_www_cli.py +0 -0
  19. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate_module.egg-info/SOURCES.txt +0 -0
  20. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate_module.egg-info/dependency_links.txt +0 -0
  21. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate_module.egg-info/entry_points.txt +0 -0
  22. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate_module.egg-info/requires.txt +0 -0
  23. {integrate_module-0.99.2 → integrate_module-0.99.5}/integrate_module.egg-info/top_level.txt +0 -0
  24. {integrate_module-0.99.2 → integrate_module-0.99.5}/setup.cfg +0 -0
  25. {integrate_module-0.99.2 → integrate_module-0.99.5}/tests/test_likelihood_multinomial.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: integrate_module
3
- Version: 0.99.2
3
+ Version: 0.99.5
4
4
  Summary: Localized probabilistic data integration
5
5
  Author-email: Thomas Mejer Hansen <tmeha@geo.au.dk>
6
6
  License: MIT
@@ -36,6 +36,7 @@ from integrate.integrate import posterior_cumulative_thickness
36
36
  from integrate.integrate import use_parallel
37
37
  from integrate.integrate import kl_divergence
38
38
  from integrate.integrate import entropy
39
+ from integrate.integrate import discrete_data_entropy
39
40
  from integrate.integrate import class_id_to_idx
40
41
  from integrate.integrate import is_notebook
41
42
  from integrate.integrate import get_hypothesis_probability
@@ -94,6 +95,7 @@ from integrate.integrate_plot import plot_profile_discrete
94
95
  from integrate.integrate_plot import plot_cumulative_probability_profile
95
96
  from integrate.integrate_plot import plot_T_EV
96
97
  from integrate.integrate_plot import plot_data_xy
98
+ from integrate.integrate_plot import plot_discrete_data_entropy
97
99
  from integrate.integrate_plot import plot_data
98
100
  from integrate.integrate_plot import plot_data_prior_post
99
101
  from integrate.integrate_plot import plot_data_prior
@@ -2963,12 +2963,95 @@ def entropy(P, base = None):
2963
2963
  >>> entropy(P)
2964
2964
  array([1.0, 0.469])
2965
2965
  """
2966
- P = np.atleast_2d(P)
2966
+ P = np.atleast_2d(P)
2967
2967
  if base is None:
2968
2968
  base = P.shape[1]
2969
2969
  H = -np.sum(P*np.log(P)/np.log(base), axis=1)
2970
2970
  return H
2971
2971
 
2972
+
2973
+ def discrete_data_entropy(f_data_h5, id_list, depth_reduce='min', showInfo=1):
2974
+ """
2975
+ Compute the pointwise (per survey location) entropy of one or more
2976
+ multinomial discrete /D{id} data entries in a DATA HDF5 file.
2977
+
2978
+ Each location's discrete observation is a probability-over-classes
2979
+ profile spanning ``nm`` depth layers (shape ``(ns, nclass, nm)`` per id,
2980
+ as written by :func:`save_data_multinomial`, e.g. via
2981
+ :func:`save_borehole_data`). Entropy is computed per (location,
2982
+ depth-layer) using ``scipy.stats.entropy`` (which correctly handles
2983
+ exact-zero probabilities, unlike :func:`entropy`), the depth axis is
2984
+ then collapsed per ``depth_reduce``, and — if more than one id is given —
2985
+ the pointwise minimum across ids is returned (i.e. the best-informed id
2986
+ at each location; adding more ids can only lower or keep equal the
2987
+ entropy at a given location, never raise it).
2988
+
2989
+ Parameters
2990
+ ----------
2991
+ f_data_h5 : str
2992
+ Path to the DATA HDF5 file.
2993
+ id_list : int or list of int
2994
+ One or more dataset ids referencing multinomial /D{id} groups
2995
+ (e.g. from ``save_borehole_data()``'s ``id_out`` / ``id_borehole_list``).
2996
+ depth_reduce : {'min', 'mean'}, optional
2997
+ How to collapse the per-location depth-layer axis, per id, before
2998
+ combining across ids. ``'min'`` (default) takes the lowest (most
2999
+ informative) entropy value across depth layers, i.e. each id's best
3000
+ depth-layer represents it at that location; ``'mean'`` averages
3001
+ entropy across depth layers instead.
3002
+ showInfo : int, optional
3003
+ Verbosity level passed through to :func:`load_data`. Default 1.
3004
+
3005
+ Returns
3006
+ -------
3007
+ H : ndarray, shape (ns,)
3008
+ Pointwise entropy in [0, 1] (base = nclass). NaN at locations not
3009
+ covered by any of the given ids (``i_use == 0`` for all of them).
3010
+
3011
+ Examples
3012
+ --------
3013
+ >>> H = ig.discrete_data_entropy(f_data_h5, id_borehole_list)
3014
+ >>> H = ig.discrete_data_entropy(f_data_h5, id_borehole_list, depth_reduce='mean')
3015
+ """
3016
+ import scipy as sp
3017
+ import warnings
3018
+ import integrate as ig
3019
+
3020
+ if not isinstance(id_list, list):
3021
+ id_list = [id_list]
3022
+ if depth_reduce not in ('mean', 'min'):
3023
+ raise ValueError("depth_reduce must be 'mean' or 'min'")
3024
+
3025
+ DATA = ig.load_data(f_data_h5, id_arr=id_list, showInfo=showInfo)
3026
+
3027
+ H_per_id = []
3028
+ for i, id in enumerate(id_list):
3029
+ if DATA['noise_model'][i] != 'multinomial':
3030
+ raise ValueError(
3031
+ "D%d is not a multinomial dataset (noise_model=%r)" %
3032
+ (id, DATA['noise_model'][i]))
3033
+ P = DATA['d_obs'][i] # (ns, nclass, nm)
3034
+ i_use = np.asarray(DATA['i_use'][i]).ravel().astype(bool)
3035
+ ns, nclass, nm = P.shape
3036
+
3037
+ H = np.full((ns, nm), np.nan)
3038
+ if np.any(i_use):
3039
+ H[i_use, :] = sp.stats.entropy(
3040
+ P[i_use].transpose(1, 0, 2).reshape(nclass, -1),
3041
+ base=nclass
3042
+ ).reshape(-1, nm)
3043
+
3044
+ with warnings.catch_warnings():
3045
+ warnings.simplefilter('ignore', category=RuntimeWarning)
3046
+ H_id = np.nanmean(H, axis=1) if depth_reduce == 'mean' else np.nanmin(H, axis=1)
3047
+ H_per_id.append(H_id)
3048
+
3049
+ with warnings.catch_warnings():
3050
+ warnings.simplefilter('ignore', category=RuntimeWarning)
3051
+ H_map = np.nanmin(np.vstack(H_per_id), axis=0)
3052
+ return H_map
3053
+
3054
+
2972
3055
  def class_id_to_idx(D, class_id=None):
2973
3056
  """
2974
3057
  Convert class identifiers to indices.
@@ -572,8 +572,8 @@ def rescale_P_obs_temperature(P_obs, T=1.0):
572
572
 
573
573
  return P_obs_scaled
574
574
 
575
- def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
576
- nan_freq=0.8, r_data_i_use=None):
575
+ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, range_data=10, range_xyz=100, doPlot=False,
576
+ range_data_nan_freq=0.8, range_data_i_use=None):
577
577
  """
578
578
  Convert point-based discrete probability observations to gridded data with distance-based weighting.
579
579
 
@@ -594,23 +594,23 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
594
594
  Y coordinate (e.g., UTM Northing) of the observation point.
595
595
  f_data_h5 : str
596
596
  Path to HDF5 data file containing survey geometry (X, Y coordinates).
597
- r_data : float, optional
597
+ range_data : float, optional
598
598
  Inner radius in meters within which observations have full strength.
599
599
  Default is 10 meters.
600
- r_dis : float, optional
600
+ range_xyz : float, optional
601
601
  Outer radius in meters for distance-based weighting. Beyond this distance,
602
602
  observations are fully attenuated (temperature → ∞). Default is 100 meters.
603
603
  doPlot : bool, optional
604
604
  If True, creates diagnostic plots showing weight distributions.
605
605
  Default is False.
606
- nan_freq : float, optional
606
+ range_data_nan_freq : float, optional
607
607
  NaN-frequency threshold for automatic data-gate selection inside
608
608
  :func:`get_weight_from_position`. Gates where the fraction of
609
609
  non-NaN values is below this threshold are excluded. Default 0.8.
610
- Ignored when ``r_data_i_use`` is provided.
611
- r_data_i_use : array-like of int or None, optional
610
+ Ignored when ``range_data_i_use`` is provided.
611
+ range_data_i_use : array-like of int or None, optional
612
612
  Explicit gate/channel indices to use for data-distance computation
613
- inside :func:`get_weight_from_position`. Overrides ``nan_freq`` when
613
+ inside :func:`get_weight_from_position`. Overrides ``range_data_nan_freq`` when
614
614
  provided. Default None.
615
615
 
616
616
  Returns
@@ -649,7 +649,7 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
649
649
  >>> P_obs = compute_P_obs_discrete(depth_top, depth_bottom, lithology, z, class_id)
650
650
  >>> X_well, Y_well = 543000.0, 6175800.0
651
651
  >>> d_obs, i_use, T_all = Pobs_to_datagrid(P_obs, X_well, Y_well, 'survey_data.h5',
652
- ... r_data=10, r_dis=100)
652
+ ... range_data=10, range_xyz=100)
653
653
  >>> # Write to data file
654
654
  >>> ig.save_data_multinomial(d_obs, i_use=i_use, id=2, f_data_h5='survey_data.h5')
655
655
 
@@ -673,8 +673,8 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
673
673
 
674
674
  # Compute distance-based weights for all grid points
675
675
  w_combined, w_dis, w_data, i_use_from_func = ig.get_weight_from_position(
676
- f_data_h5, X, Y, r_data=r_data, r_dis=r_dis, doPlot=doPlot,
677
- nan_freq=nan_freq, r_data_i_use=r_data_i_use
676
+ f_data_h5, X, Y, range_data=range_data, range_xyz=range_xyz, doPlot=doPlot,
677
+ range_data_nan_freq=range_data_nan_freq, range_data_i_use=range_data_i_use
678
678
  )
679
679
 
680
680
  # Convert distance weight to temperature
@@ -703,9 +703,9 @@ def Pobs_to_datagrid(P_obs, X, Y, f_data_h5, r_data=10, r_dis=100, doPlot=False,
703
703
 
704
704
 
705
705
 
706
- def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400, r_data=2,
706
+ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, range_xyz=400, range_data=2,
707
707
  useLog=True, doPlot=False, plFile=None, showInfo=0,
708
- nan_freq=0.8, r_data_i_use=None):
708
+ range_data_nan_freq=0.8, range_data_i_use=None):
709
709
  """Calculate weights based on distance and data similarity to a reference point.
710
710
 
711
711
  This function computes three sets of weights:
@@ -723,9 +723,9 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
723
723
  Y coordinate of reference point (well). Default 0.
724
724
  i_ref : int, optional
725
725
  Index of reference point. Default -1 (auto-calculated as closest to x_well, y_well).
726
- r_dis : float, optional
726
+ range_xyz : float, optional
727
727
  Geographic XY distance range [m] for spatial weighting. Default 400.
728
- r_data : float, optional
728
+ range_data : float, optional
729
729
  Data-space similarity range parameter for data weighting. Default 2.
730
730
  useLog : bool, optional
731
731
  Apply log10 transform to data before computing similarity. Default True.
@@ -735,16 +735,16 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
735
735
  Output filename for the diagnostic plot. Auto-generated if None.
736
736
  showInfo : int, optional
737
737
  Verbosity level. Default 0.
738
- nan_freq : float, optional
738
+ range_data_nan_freq : float, optional
739
739
  NaN-frequency threshold for automatic gate selection. Gates where the
740
740
  fraction of non-NaN values across all soundings is below this threshold
741
741
  are excluded from the data-distance computation. Default 0.8.
742
- Ignored when ``r_data_i_use`` is provided.
743
- r_data_i_use : array-like of int or None, optional
742
+ Ignored when ``range_data_i_use`` is provided.
743
+ range_data_i_use : array-like of int or None, optional
744
744
  Explicit gate/channel indices to use for the data-distance computation.
745
- When provided, overrides the ``nan_freq`` automatic selection.
745
+ When provided, overrides the ``range_data_nan_freq`` automatic selection.
746
746
  A NaN check at the reference sounding is still applied.
747
- Default None (use ``nan_freq`` threshold).
747
+ Default None (use ``range_data_nan_freq`` threshold).
748
748
 
749
749
  Returns
750
750
  -------
@@ -760,8 +760,8 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
760
760
  Notes
761
761
  -----
762
762
  Weights are calculated using Gaussian functions:
763
- - Distance weights: exp(-dis² / r_dis²)
764
- - Data weights: exp(-sum_dd² / r_data²)
763
+ - Distance weights: exp(-dis² / range_xyz²)
764
+ - Data weights: exp(-sum_dd² / range_data²)
765
765
  where dis is geographic distance and sum_dd is cumulative data difference.
766
766
  """
767
767
  import integrate as ig
@@ -777,12 +777,12 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
777
777
  i_ref = np.argmin((X-x_well)**2 + (Y-y_well)**2)
778
778
 
779
779
  # Select gates to use for data-distance computation
780
- if r_data_i_use is not None:
781
- gates = np.asarray(r_data_i_use, dtype=int)
780
+ if range_data_i_use is not None:
781
+ gates = np.asarray(range_data_i_use, dtype=int)
782
782
  else:
783
783
  n_not_nan = np.sum(~np.isnan(d_obs), axis=0)
784
784
  n_not_nan_freq = n_not_nan / d_obs.shape[0]
785
- gates = np.where(n_not_nan_freq > nan_freq)[0]
785
+ gates = np.where(n_not_nan_freq > range_data_nan_freq)[0]
786
786
  # Remove gates that are NaN at the reference sounding
787
787
  gates = gates[~np.isnan(d_obs[i_ref, gates])]
788
788
  i_use = gates
@@ -795,12 +795,12 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
795
795
  d_test =d_obs[:,i_use]
796
796
  dd = np.abs(d_test - d_ref)
797
797
  sum_dd = np.sum(dd, axis=1)
798
- w_data = np.exp(-1*sum_dd**2/r_data**2)
799
-
798
+ w_data = np.exp(-1*sum_dd**2/range_data**2)
799
+
800
800
 
801
801
  # Compute the distance from each data point to the actual borehole location
802
802
  dis = np.sqrt((X-x_well)**2 + (Y-y_well)**2)
803
- w_dis = np.exp(-1*dis**2/r_dis**2)
803
+ w_dis = np.exp(-1*dis**2/range_xyz**2)
804
804
 
805
805
  w_combined = w_data * w_dis
806
806
 
@@ -838,7 +838,7 @@ def get_weight_from_position(f_data_h5, x_well=0, y_well=0, i_ref=-1, r_dis=400,
838
838
  plt.xlabel('X')
839
839
  plt.ylabel('Y')
840
840
  if plFile is None:
841
- plFile = 'weights_%d_%d_%d_rdis%d_rdata%d.png' % (x_well,y_well,i_ref,r_dis,r_data)
841
+ plFile = 'weights_%d_%d_%d_rdis%d_rdata%d.png' % (x_well,y_well,i_ref,range_xyz,range_data)
842
842
  plt.savefig(plFile, dpi=300)
843
843
 
844
844
  return w_combined, w_dis, w_data, i_ref
@@ -1023,46 +1023,54 @@ def save_borehole_data(f_prior_h5, f_data_h5, BH, **kwargs):
1023
1023
  Path to the prior HDF5 file.
1024
1024
  f_data_h5 : str
1025
1025
  Path to the observed-data HDF5 file.
1026
- BH : dict
1026
+ BH : dict or list of dict
1027
1027
  Borehole dictionary with keys depth_top, depth_bottom, class_obs,
1028
1028
  class_prob, X, Y, name, method. Two optional keys control the
1029
- distance-weighting radii when r_data / r_dis are not passed as
1030
- explicit kwargs:
1029
+ distance-weighting radii when range_data / range_xyz are not passed
1030
+ as explicit kwargs:
1031
1031
 
1032
1032
  * ``range_data`` (float, optional) — data-space similarity radius.
1033
1033
  Survey points whose EM data response is similar to the borehole
1034
1034
  location receive higher weight; points that are more dissimilar are
1035
- down-weighted. Default: 1,000,000 (effectively no cutoff).
1036
- * ``range_dis`` (float, optional) — geographic XY distance [m] beyond
1035
+ down-weighted. Only used if present and non-negative; otherwise the
1036
+ default applies. Default: 1,000,000 (effectively no cutoff).
1037
+ * ``range_xyz`` (float, optional) — geographic XY distance [m] beyond
1037
1038
  which the borehole exerts no influence on nearby survey points.
1038
- Default: 300 m.
1039
- * ``nan_freq`` (float, optional) — NaN-frequency threshold for automatic
1039
+ Only used if present and non-negative; otherwise the default
1040
+ applies. Default: 300 m.
1041
+ * ``range_data_nan_freq`` (float, optional) — NaN-frequency threshold for automatic
1040
1042
  data-gate selection. Default: 0.8.
1041
- * ``r_data_i_use`` (list of int, optional) — explicit gate indices for
1042
- data-distance computation; overrides ``nan_freq`` when provided.
1043
+ * ``range_data_i_use`` (list of int, optional) — explicit gate indices for
1044
+ data-distance computation; overrides ``range_data_nan_freq`` when provided.
1045
+
1046
+ If ``BH`` is a list of dicts, each borehole is processed in turn
1047
+ (one call per element, in order) using the same ``**kwargs`` for
1048
+ every borehole; a per-borehole ``range_data``/``range_xyz``/etc. key
1049
+ set inside an individual dict still takes effect for that borehole
1050
+ exactly as in the single-dict case. See ``Returns`` below.
1043
1051
  **kwargs
1044
1052
  im_prior : int, optional
1045
1053
  Index of the discrete model parameter in f_prior_h5 (e.g. 2 for /M2).
1046
1054
  Default is 2.
1047
1055
  parallel : bool, optional
1048
1056
  Enable parallel mode computation. Default is False.
1049
- r_data : float, optional
1057
+ range_data : float, optional
1050
1058
  Data-space similarity radius. Overrides ``BH['range_data']`` when
1051
- provided. Resolution order: explicit kwarg > BH['range_data'] >
1052
- 1,000,000 (no cutoff).
1053
- r_dis : float, optional
1054
- Geographic XY fade-out distance [m]. Overrides ``BH['range_dis']``
1055
- when provided. Resolution order: explicit kwarg > BH['range_dis'] >
1056
- 300 m.
1057
- nan_freq : float, optional
1059
+ provided. Resolution order: explicit kwarg > BH['range_data']
1060
+ (if present and non-negative) > 1,000,000 (no cutoff).
1061
+ range_xyz : float, optional
1062
+ Geographic XY fade-out distance [m]. Overrides ``BH['range_xyz']``
1063
+ when provided. Resolution order: explicit kwarg > BH['range_xyz']
1064
+ (if present and non-negative) > 300 m.
1065
+ range_data_nan_freq : float, optional
1058
1066
  NaN-frequency threshold for automatic data-gate selection. Gates
1059
1067
  where the fraction of non-NaN soundings is below this threshold are
1060
1068
  excluded from data-distance computation. Resolution order: explicit
1061
- kwarg > BH['nan_freq'] > 0.8. Ignored when ``r_data_i_use`` is set.
1062
- r_data_i_use : list of int or None, optional
1069
+ kwarg > BH['range_data_nan_freq'] > 0.8. Ignored when ``range_data_i_use`` is set.
1070
+ range_data_i_use : list of int or None, optional
1063
1071
  Explicit gate/channel indices to use for data-distance computation.
1064
- Overrides ``nan_freq`` when provided. Resolution order: explicit
1065
- kwarg > BH['r_data_i_use'] > None.
1072
+ Overrides ``range_data_nan_freq`` when provided. Resolution order: explicit
1073
+ kwarg > BH['range_data_i_use'] > None.
1066
1074
  doPlot : bool, optional
1067
1075
  Plot distance-weight maps. Default is False.
1068
1076
  showInfo : int, optional
@@ -1072,32 +1080,56 @@ def save_borehole_data(f_prior_h5, f_data_h5, BH, **kwargs):
1072
1080
  Returns
1073
1081
  -------
1074
1082
  id_prior : int
1075
- Dataset index of the new /D entry added to f_prior_h5.
1083
+ If ``BH`` is a single dict: dataset index of the new /D entry added
1084
+ to f_prior_h5.
1076
1085
  id_out : int
1077
- Dataset index of the new /D entry added to f_data_h5.
1086
+ If ``BH`` is a single dict: dataset index of the new /D entry added
1087
+ to f_data_h5.
1088
+ id_prior_list : list of int
1089
+ If ``BH`` is a list of dicts: dataset indices of the new /D entries
1090
+ added to f_prior_h5, one per borehole, same order as ``BH``.
1091
+ id_borehole_list : list of int
1092
+ If ``BH`` is a list of dicts: dataset indices of the new /D entries
1093
+ added to f_data_h5, one per borehole, same order as ``BH``.
1078
1094
 
1079
1095
  Examples
1080
1096
  --------
1081
1097
  >>> # Single borehole
1082
1098
  >>> id_prior, id_data = ig.save_borehole_data(f_prior_h5, f_data_h5, BH)
1083
1099
 
1084
- >>> # All boreholes — collect data IDs for joint inversion
1085
- >>> id_borehole_list = []
1086
- >>> for BH in BHOLES:
1087
- ... _, id_out = ig.save_borehole_data(f_prior_h5, f_data_h5, BH,
1088
- ... r_data=2, r_dis=300, parallel=True)
1089
- ... id_borehole_list.append(id_out)
1100
+ >>> # All boreholes in one call — collect data IDs for joint inversion
1101
+ >>> id_prior_list, id_borehole_list = ig.save_borehole_data(
1102
+ ... f_prior_h5, f_data_h5, BHOLES,
1103
+ ... range_data=2, range_xyz=300, parallel=True)
1090
1104
  >>> f_post_h5 = ig.integrate_rejection(f_prior_h5, f_data_h5,
1091
1105
  ... id_use=[1] + id_borehole_list)
1092
1106
  """
1107
+ if isinstance(BH, list):
1108
+ id_prior_list = []
1109
+ id_borehole_list = []
1110
+ for bh in BH:
1111
+ id_prior, id_out = save_borehole_data(f_prior_h5, f_data_h5, bh, **kwargs)
1112
+ id_prior_list.append(id_prior)
1113
+ id_borehole_list.append(id_out)
1114
+ return id_prior_list, id_borehole_list
1115
+
1093
1116
  import integrate as ig
1094
1117
 
1118
+ def _resolve_radius(name, default):
1119
+ val = kwargs.get(name, None)
1120
+ if val is not None:
1121
+ return val
1122
+ val = BH.get(name, None)
1123
+ if val is not None and val >= 0:
1124
+ return val
1125
+ return default
1126
+
1095
1127
  im_prior = kwargs.get('im_prior', 2)
1096
1128
  parallel = kwargs.get('parallel', False)
1097
- r_data = kwargs.get('r_data', BH.get('range_data', 1_000_000))
1098
- r_dis = kwargs.get('r_dis', BH.get('range_dis', 300))
1099
- nan_freq = kwargs.get('nan_freq', BH.get('nan_freq', 0.8))
1100
- r_data_i_use = kwargs.get('r_data_i_use', BH.get('r_data_i_use', None))
1129
+ range_data = _resolve_radius('range_data', 1_000_000)
1130
+ range_xyz = _resolve_radius('range_xyz', 300)
1131
+ range_data_nan_freq = kwargs.get('range_data_nan_freq', BH.get('range_data_nan_freq', 0.8))
1132
+ range_data_i_use = kwargs.get('range_data_i_use', BH.get('range_data_i_use', None))
1101
1133
  doPlot = kwargs.get('doPlot', False)
1102
1134
  showInfo = kwargs.get('showInfo', 1)
1103
1135
 
@@ -1109,8 +1141,8 @@ def save_borehole_data(f_prior_h5, f_data_h5, BH, **kwargs):
1109
1141
  # Step 2: extrapolate point observations to the survey grid
1110
1142
  d_obs, i_use, T_use = Pobs_to_datagrid(
1111
1143
  P_obs, BH['X'], BH['Y'], f_data_h5,
1112
- r_data=r_data, r_dis=r_dis, doPlot=doPlot,
1113
- nan_freq=nan_freq, r_data_i_use=r_data_i_use)
1144
+ range_data=range_data, range_xyz=range_xyz, doPlot=doPlot,
1145
+ range_data_nan_freq=range_data_nan_freq, range_data_i_use=range_data_i_use)
1114
1146
 
1115
1147
  # Step 3: save gridded observations to f_data_h5
1116
1148
  id_out, _ = ig.save_data_multinomial(
@@ -2967,6 +2967,10 @@ def get_case_data(case='DAUGAARD', loadAll=False, loadType='', filelist=None, **
2967
2967
  print("filelist to download:")
2968
2968
  print(filelist)
2969
2969
 
2970
+ # Direct-download endpoint for the shared data. Use ERDA's 'share_redirect'
2971
+ # form (serves raw file bytes, returns proper 404s), NOT the 'sharelink'
2972
+ # browse link, which returns an HTML directory page with status 200 for any
2973
+ # path and would be saved as a corrupt file.
2970
2974
  urlErda = 'https://anon.erda.au.dk/share_redirect/dxOLKDtoul'
2971
2975
  urlErdaCase = '%s/%s' % (urlErda,case)
2972
2976
  from tqdm import tqdm
@@ -2976,7 +2980,22 @@ def get_case_data(case='DAUGAARD', loadAll=False, loadType='', filelist=None, **
2976
2980
  if showInfo>-1:
2977
2981
  print('--> Got data for case: %s' % case)
2978
2982
 
2979
- return [f.replace('\\', '/').split('/')[-1] for f in filelist]
2983
+ # Return the local filename for each requested file, but only if it was
2984
+ # actually obtained (downloaded or already present locally). download_file
2985
+ # saves to download_dir='.' using the basename, so we check for that here.
2986
+ # Files that could not be obtained (e.g. missing on the remote server)
2987
+ # return an empty string, so callers can test `if len(name) == 0`.
2988
+ result = []
2989
+ for f in filelist:
2990
+ basename = f.replace('\\', '/').split('/')[-1]
2991
+ if os.path.exists(basename):
2992
+ result.append(basename)
2993
+ else:
2994
+ if showInfo>-1:
2995
+ print('File %s was not obtained (missing locally and on remote); returning empty string.' % basename)
2996
+ result.append('')
2997
+
2998
+ return result
2980
2999
 
2981
3000
 
2982
3001
 
@@ -5179,15 +5198,17 @@ def write_borehole(W, filename, **kwargs):
5179
5198
  Used only by :func:`plot_boreholes` to place the well on a shared
5180
5199
  elevation axis. Has no effect on inversion.
5181
5200
  * ``range_data`` (float, optional) – data-space similarity radius used by
5182
- :func:`save_borehole_data` when ``r_data`` is not passed explicitly.
5201
+ :func:`save_borehole_data` when ``range_data`` is not passed explicitly.
5202
+ Only used if present and non-negative, otherwise the default applies.
5183
5203
  Default: 1,000,000 (no cutoff).
5184
- * ``range_dis`` (float, optional) – geographic XY fade-out distance [m]
5185
- used by :func:`save_borehole_data` when ``r_dis`` is not passed explicitly.
5186
- Default: 300 m.
5187
- * ``nan_freq`` (float, optional) – NaN-frequency threshold for automatic
5204
+ * ``range_xyz`` (float, optional) – geographic XY fade-out distance [m]
5205
+ used by :func:`save_borehole_data` when ``range_xyz`` is not passed
5206
+ explicitly. Only used if present and non-negative, otherwise the
5207
+ default applies. Default: 300 m.
5208
+ * ``range_data_nan_freq`` (float, optional) – NaN-frequency threshold for automatic
5188
5209
  data-gate selection in :func:`save_borehole_data`. Default: 0.8.
5189
- * ``r_data_i_use`` (list of int, optional) – explicit gate/channel indices
5190
- for data-distance computation; overrides ``nan_freq`` when provided.
5210
+ * ``range_data_i_use`` (list of int, optional) – explicit gate/channel indices
5211
+ for data-distance computation; overrides ``range_data_nan_freq`` when provided.
5191
5212
 
5192
5213
  numpy arrays and scalars are automatically converted to plain
5193
5214
  Python lists/numbers so the file is human-readable JSON.
@@ -134,6 +134,8 @@ def get_colormap_and_limits(cmap_type='default', custom_clim=None):
134
134
  - 'default': Red-white-blue-black colormap for general use
135
135
  - 'resistivity': Log-scale colormap optimized for resistivity data
136
136
  - 'entropy': Grayscale colormap for uncertainty visualization
137
+ - 'discrete_data_entropy': Black-red-white colormap where 0 (informative)
138
+ is black and 1 (uninformative) is white
137
139
  - 'discrete': Categorical colormap for discrete classifications
138
140
  - 'temperature': Hot colormap for temperature/evidence fields
139
141
  - 'elevation': Terrain-like colormap for topographic data
@@ -201,6 +203,13 @@ def get_colormap_and_limits(cmap_type='default', custom_clim=None):
201
203
  cmap = LinearSegmentedColormap.from_list('entropy', colors, N=256)
202
204
  clim = [0, 1]
203
205
 
206
+ elif cmap_type == 'discrete_data_entropy':
207
+ # Black-red-white colormap: 0 (informative, low entropy) -> black,
208
+ # 1 (uninformative, high entropy) -> white
209
+ colors = ['black', 'red', 'white']
210
+ cmap = LinearSegmentedColormap.from_list('discrete_data_entropy', colors, N=256)
211
+ clim = [0, 1]
212
+
204
213
  elif cmap_type == 'discrete':
205
214
  # Categorical colormap for discrete classifications
206
215
  cmap = plt.cm.Set1
@@ -2844,6 +2853,55 @@ def plot_data_xy(f_data_h5, Dkey='D1', data_key='d_obs', data_channel=0, uselog=
2844
2853
 
2845
2854
  return fig
2846
2855
 
2856
+
2857
+ def plot_discrete_data_entropy(f_data_h5, id_list, depth_reduce='min', **kwargs):
2858
+ """
2859
+ Plot the pointwise entropy of one or more multinomial discrete data
2860
+ entries in a DATA HDF5 file, as a 2D map via :func:`plot_xy`.
2861
+
2862
+ See :func:`discrete_data_entropy` for how the entropy value at each
2863
+ location is computed. Useful for visualizing, on a map, where one or
2864
+ more discrete-type observations (e.g. borehole-derived data written by
2865
+ :func:`save_borehole_data`) constrain the inversion versus where they
2866
+ don't — low entropy marks confident/informative locations, high entropy
2867
+ marks uninformative ones.
2868
+
2869
+ Parameters
2870
+ ----------
2871
+ f_data_h5 : str
2872
+ Path to the DATA HDF5 file.
2873
+ id_list : int or list of int
2874
+ One or more dataset ids referencing multinomial /D{id} groups
2875
+ (e.g. from ``save_borehole_data()``'s ``id_out`` / ``id_borehole_list``).
2876
+ depth_reduce : {'min', 'mean'}, optional
2877
+ See :func:`discrete_data_entropy`. Default ``'min'``.
2878
+ **kwargs
2879
+ Forwarded to :func:`plot_xy` (e.g. cmap, clim, hardcopy, title, ax, ...).
2880
+
2881
+ Returns
2882
+ -------
2883
+ fig, ax, sc
2884
+ As returned by :func:`plot_xy`.
2885
+
2886
+ Examples
2887
+ --------
2888
+ >>> fig, ax, sc = ig.plot_discrete_data_entropy(f_data_h5, id_borehole_list)
2889
+ """
2890
+ import integrate as ig
2891
+
2892
+ H = ig.discrete_data_entropy(f_data_h5, id_list, depth_reduce=depth_reduce,
2893
+ showInfo=kwargs.pop('showInfo', 1))
2894
+
2895
+ ids_str = ','.join(str(i) for i in (id_list if isinstance(id_list, list) else [id_list]))
2896
+ kwargs.setdefault('title', 'Minimum discrete data entropy (D%s)' % ids_str)
2897
+ kwargs.setdefault('colorbar_label', 'Entropy')
2898
+ _cmap, _clim = get_colormap_and_limits('discrete_data_entropy')
2899
+ kwargs.setdefault('cmap', _cmap)
2900
+ kwargs.setdefault('clim', _clim)
2901
+
2902
+ return plot_xy(H, f_data_h5=f_data_h5, **kwargs)
2903
+
2904
+
2847
2905
  def plot_data(f_data_h5, i_plot=[], Dkey=[], plType='imshow', uselog=True, **kwargs):
2848
2906
  """
2849
2907
  Plot observational data from an HDF5 file.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: integrate_module
3
- Version: 0.99.2
3
+ Version: 0.99.5
4
4
  Summary: Localized probabilistic data integration
5
5
  Author-email: Thomas Mejer Hansen <tmeha@geo.au.dk>
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "integrate_module"
7
- version = "0.99.02"
7
+ version = "0.99.5"
8
8
  description = "Localized probabilistic data integration"
9
9
  readme = { file = "README.md", content-type = "text/markdown" }
10
10
  requires-python = ">=3.10"