graphica-plot 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphica/Graphica.ico +0 -0
- graphica/__init__.py +1 -0
- graphica/__main__.py +111 -0
- graphica/assets/__init__.py +0 -0
- graphica/assets/icons/__init__.py +0 -0
- graphica/assets/icons/arrow-right.svg +21 -0
- graphica/assets/icons/bold.svg +20 -0
- graphica/assets/icons/calculator.svg +26 -0
- graphica/assets/icons/chart-histogram.svg +24 -0
- graphica/assets/icons/chart-line.svg +20 -0
- graphica/assets/icons/chevron-down.svg +19 -0
- graphica/assets/icons/chevron-right.svg +19 -0
- graphica/assets/icons/color-swatch.svg +22 -0
- graphica/assets/icons/column-insert-right.svg +21 -0
- graphica/assets/icons/column-remove.svg +21 -0
- graphica/assets/icons/copy.svg +20 -0
- graphica/assets/icons/download.svg +21 -0
- graphica/assets/icons/edit.svg +21 -0
- graphica/assets/icons/eye-off.svg +21 -0
- graphica/assets/icons/eye.svg +20 -0
- graphica/assets/icons/file-plus.svg +22 -0
- graphica/assets/icons/folder-plus.svg +21 -0
- graphica/assets/icons/folder.svg +19 -0
- graphica/assets/icons/highlight.svg +22 -0
- graphica/assets/icons/history.svg +20 -0
- graphica/assets/icons/italic.svg +21 -0
- graphica/assets/icons/layout-grid.svg +22 -0
- graphica/assets/icons/math-function.svg +22 -0
- graphica/assets/icons/message-2.svg +21 -0
- graphica/assets/icons/mountain.svg +20 -0
- graphica/assets/icons/palette.svg +22 -0
- graphica/assets/icons/pointer.svg +19 -0
- graphica/assets/icons/refresh.svg +20 -0
- graphica/assets/icons/row-insert-bottom.svg +21 -0
- graphica/assets/icons/row-remove.svg +21 -0
- graphica/assets/icons/search.svg +20 -0
- graphica/assets/icons/select-all.svg +35 -0
- graphica/assets/icons/subscript.svg +20 -0
- graphica/assets/icons/superscript.svg +20 -0
- graphica/assets/icons/table.svg +21 -0
- graphica/assets/icons/trash.svg +23 -0
- graphica/assets/icons/typography.svg +23 -0
- graphica/assets/icons/x.svg +20 -0
- graphica/core/__init__.py +0 -0
- graphica/core/analysis.py +1278 -0
- graphica/core/app_paths.py +35 -0
- graphica/core/axis_settings.py +128 -0
- graphica/core/caption_export.py +47 -0
- graphica/core/color_palettes.py +56 -0
- graphica/core/commands.py +226 -0
- graphica/core/cvd_simulation.py +49 -0
- graphica/core/dataset.py +443 -0
- graphica/core/diagnostics.py +105 -0
- graphica/core/excel_utils.py +58 -0
- graphica/core/fit_models.py +379 -0
- graphica/core/grid_data.py +128 -0
- graphica/core/i18n.py +44 -0
- graphica/core/json_utils.py +27 -0
- graphica/core/label_utils.py +21 -0
- graphica/core/methods_text.py +115 -0
- graphica/core/named_colors.py +126 -0
- graphica/core/plugin_api.py +515 -0
- graphica/core/plugin_context.py +167 -0
- graphica/core/plugin_install.py +82 -0
- graphica/core/plugin_manifest.py +67 -0
- graphica/core/plugin_testing.py +213 -0
- graphica/core/plugin_types.py +118 -0
- graphica/core/provenance.py +19 -0
- graphica/core/report_export.py +51 -0
- graphica/core/safe_eval.py +194 -0
- graphica/core/script_export.py +272 -0
- graphica/core/translations_en.py +530 -0
- graphica/core/unit_conversion.py +51 -0
- graphica/core/update_check.py +43 -0
- graphica/core/version.py +14 -0
- graphica/gui/__init__.py +0 -0
- graphica/gui/app_settings.py +119 -0
- graphica/gui/axis_bindings.py +173 -0
- graphica/gui/binding.py +135 -0
- graphica/gui/builders/__init__.py +1 -0
- graphica/gui/builders/axis_panel.py +431 -0
- graphica/gui/builders/canvas_area.py +176 -0
- graphica/gui/builders/common.py +120 -0
- graphica/gui/builders/dataset_panel.py +433 -0
- graphica/gui/builders/property_sections.py +208 -0
- graphica/gui/canvas.py +519 -0
- graphica/gui/color_history.py +47 -0
- graphica/gui/color_picker_widget.py +198 -0
- graphica/gui/crash_handler.py +77 -0
- graphica/gui/cvd_preview.py +23 -0
- graphica/gui/data_editor.py +761 -0
- graphica/gui/data_import_flow.py +355 -0
- graphica/gui/dataset_bindings.py +107 -0
- graphica/gui/dataset_style_icon.py +80 -0
- graphica/gui/datasets/__init__.py +1 -0
- graphica/gui/datasets/actions_menu.py +114 -0
- graphica/gui/datasets/colors.py +218 -0
- graphica/gui/datasets/fitting.py +70 -0
- graphica/gui/datasets/host.py +162 -0
- graphica/gui/datasets/operations/__init__.py +1 -0
- graphica/gui/datasets/operations/fitting.py +468 -0
- graphica/gui/datasets/operations/peaks.py +109 -0
- graphica/gui/datasets/operations/processing.py +659 -0
- graphica/gui/datasets/operations/runner.py +148 -0
- graphica/gui/datasets/operations/transfer.py +227 -0
- graphica/gui/datasets/order.py +310 -0
- graphica/gui/datasets/overlays.py +74 -0
- graphica/gui/datasets/peaks.py +21 -0
- graphica/gui/datasets/plugin_runs.py +99 -0
- graphica/gui/datasets/processing.py +68 -0
- graphica/gui/datasets/property_panel.py +538 -0
- graphica/gui/datasets/transfer.py +36 -0
- graphica/gui/detached_canvas_window.py +22 -0
- graphica/gui/dialogs/__init__.py +137 -0
- graphica/gui/dialogs/analysis.py +1322 -0
- graphica/gui/dialogs/app.py +1015 -0
- graphica/gui/dialogs/appearance.py +851 -0
- graphica/gui/dialogs/data_edit.py +577 -0
- graphica/gui/dialogs/data_import.py +679 -0
- graphica/gui/dialogs/export.py +433 -0
- graphica/gui/dock_layout.py +176 -0
- graphica/gui/export_preview_panel.py +321 -0
- graphica/gui/export_settings.py +16 -0
- graphica/gui/file_association.py +97 -0
- graphica/gui/icon_utils.py +44 -0
- graphica/gui/main_app_window.py +210 -0
- graphica/gui/main_window.py +1151 -0
- graphica/gui/mathtext_preview.py +168 -0
- graphica/gui/menu_bar.py +320 -0
- graphica/gui/minimap_widget.py +151 -0
- graphica/gui/mixins/__init__.py +0 -0
- graphica/gui/mixins/export_mixin.py +520 -0
- graphica/gui/mixins/help_mixin.py +127 -0
- graphica/gui/mixins/project_io_mixin.py +289 -0
- graphica/gui/mixins/quick_access_mixin.py +194 -0
- graphica/gui/mixins/ui_setup_mixin.py +206 -0
- graphica/gui/notify.py +61 -0
- graphica/gui/panels/__init__.py +43 -0
- graphica/gui/panels/axis_settings.py +605 -0
- graphica/gui/panels/dataset_tree.py +424 -0
- graphica/gui/plot_type_drawers.py +131 -0
- graphica/gui/plugin_context.py +119 -0
- graphica/gui/project_files.py +361 -0
- graphica/gui/provenance_panel.py +65 -0
- graphica/gui/rendering/__init__.py +1 -0
- graphica/gui/rendering/annotations.py +177 -0
- graphica/gui/rendering/appearance.py +368 -0
- graphica/gui/rendering/common.py +316 -0
- graphica/gui/rendering/data_1d.py +376 -0
- graphica/gui/rendering/data_2d.py +66 -0
- graphica/gui/residual_panel.py +62 -0
- graphica/gui/resources.py +16 -0
- graphica/gui/single_instance.py +141 -0
- graphica/gui/splash.py +83 -0
- graphica/gui/task_runner.py +41 -0
- graphica/gui/theme.py +864 -0
- graphica/gui/tools/__init__.py +56 -0
- graphica/gui/tools/annotation.py +275 -0
- graphica/gui/tools/cursor.py +259 -0
- graphica/gui/tools/layout_edit.py +285 -0
- graphica/gui/tools/manager.py +78 -0
- graphica/gui/tools/peak_placement.py +132 -0
- graphica/gui/tools/range_select.py +191 -0
- graphica/gui/tools/region_highlight.py +222 -0
- graphica/gui/tools/slice_extraction.py +200 -0
- graphica/gui/widget_translation.py +46 -0
- graphica/gui/workers.py +280 -0
- graphica/models/__init__.py +0 -0
- graphica/models/project.py +197 -0
- graphica/plugin/__init__.py +20 -0
- graphica/plugin/testing.py +57 -0
- graphica/sample_data/__init__.py +0 -0
- graphica/sample_data/cooling_curve_sample.csv +42 -0
- graphica/ui_main_window.py +581 -0
- graphica_plot-2.0.0.dist-info/METADATA +317 -0
- graphica_plot-2.0.0.dist-info/RECORD +181 -0
- graphica_plot-2.0.0.dist-info/WHEEL +5 -0
- graphica_plot-2.0.0.dist-info/entry_points.txt +2 -0
- graphica_plot-2.0.0.dist-info/licenses/LICENSE +30 -0
- graphica_plot-2.0.0.dist-info/licenses/THIRD_PARTY_LICENSES.md +91 -0
- graphica_plot-2.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1278 @@
|
|
|
1
|
+
from typing import Any, Callable
|
|
2
|
+
import numpy as np
|
|
3
|
+
from scipy import sparse
|
|
4
|
+
from scipy.sparse.linalg import spsolve
|
|
5
|
+
from scipy.integrate import simpson, cumulative_trapezoid, cumulative_simpson
|
|
6
|
+
from scipy.interpolate import CubicSpline
|
|
7
|
+
from scipy.ndimage import uniform_filter1d, median_filter, gaussian_filter1d
|
|
8
|
+
from scipy.optimize import curve_fit
|
|
9
|
+
from scipy.signal import find_peaks, peak_widths, savgol_filter, correlate, correlation_lags
|
|
10
|
+
from scipy.special import wofz
|
|
11
|
+
from scipy.stats import gaussian_kde
|
|
12
|
+
|
|
13
|
+
from graphica.core.fit_models import RESERVED_FIT_TYPE_NAMES, fit_model_id, resolve_fit_model, resolve_fit_param_names
|
|
14
|
+
|
|
15
|
+
CURVE_FIT_MAX_ITERATIONS = 5000
|
|
16
|
+
|
|
17
|
+
# プラグインのフィット関数。{name: {"func": f(x, *params), "params": [...], "p0": list | callable | None}}
|
|
18
|
+
_PLUGIN_FIT_FUNCTIONS: dict[str, dict[str, Any]] = {}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def register_fit_function(name: str, func: Callable[..., Any], param_names: list[str],
|
|
22
|
+
p0: list[float] | Callable[..., Any] | None = None) -> None:
|
|
23
|
+
"""p0 は初期値のリストか (x_data, y_data) -> list を返す関数。省略時は全て 1.0。"""
|
|
24
|
+
if not name or not name.strip():
|
|
25
|
+
raise ValueError("フィット関数名が空です。")
|
|
26
|
+
if name in RESERVED_FIT_TYPE_NAMES:
|
|
27
|
+
raise ValueError(f"'{name}' は組み込みのフィットタイプ名と衝突します。")
|
|
28
|
+
if name in _PLUGIN_FIT_FUNCTIONS:
|
|
29
|
+
raise ValueError(f"フィット関数 '{name}' は既に登録されています。")
|
|
30
|
+
if not param_names:
|
|
31
|
+
raise ValueError("param_names が空です。")
|
|
32
|
+
_PLUGIN_FIT_FUNCTIONS[name] = {"func": func, "params": list(param_names), "p0": p0}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def get_plugin_fit_type_names() -> list[str]:
|
|
36
|
+
return list(_PLUGIN_FIT_FUNCTIONS.keys())
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def get_fit_param_names(fit_type: str, custom_formula: str | None = None, model_id: str | None = None) -> list[str]:
|
|
40
|
+
"""フィットせずにパラメータ名を返す(フィットの前に入力欄を組み立てるため)。model_id が分かればそれを優先する。"""
|
|
41
|
+
return resolve_fit_param_names(fit_type, custom_formula, _PLUGIN_FIT_FUNCTIONS, model_id)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def get_fit_model_id(fit_type: str) -> str | None:
|
|
45
|
+
"""保存用の安定した ID(組み込みは 'gaussian' など、カスタム数式、'plugin:<名前>')。分からなければ None。"""
|
|
46
|
+
return fit_model_id(fit_type, _PLUGIN_FIT_FUNCTIONS)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
_ROBUST_LOSS_FUNCTIONS = ('linear', 'soft_l1', 'huber')
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _run_curve_fit_with_overrides(fit_func: Callable[..., Any], params_info: list[str], p0: Any, x_data: Any, y_data: Any,
|
|
53
|
+
sigma: Any, p0_overrides: dict[str, float] | None, fixed_params: dict[str, float] | None,
|
|
54
|
+
bounds: dict[str, tuple[float, float]] | None, fit_type_label: str,
|
|
55
|
+
loss: str = 'linear') -> tuple[np.ndarray, np.ndarray]:
|
|
56
|
+
"""p0 の上書き・固定・範囲拘束を適用して curve_fit を実行する。
|
|
57
|
+
|
|
58
|
+
popt と pcov は固定パラメータも含めたフルサイズで返す(固定分の pcov は 0)。
|
|
59
|
+
"""
|
|
60
|
+
if loss not in _ROBUST_LOSS_FUNCTIONS:
|
|
61
|
+
raise ValueError(f"未知の損失関数です(loss): '{loss}' (使用可能: {_ROBUST_LOSS_FUNCTIONS})")
|
|
62
|
+
if len(x_data) < len(params_info):
|
|
63
|
+
raise ValueError(
|
|
64
|
+
f"データ点数 ({len(x_data)}) がフィットに必要なパラメータ数 "
|
|
65
|
+
f"({len(params_info)}) より少ないため、フィッティングできません。"
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
p0_overrides = p0_overrides or {}
|
|
69
|
+
fixed_params = fixed_params or {}
|
|
70
|
+
bounds = bounds or {}
|
|
71
|
+
|
|
72
|
+
for name_dict, label in (
|
|
73
|
+
(p0_overrides, "p0_overrides"), (fixed_params, "fixed_params"), (bounds, "bounds"),
|
|
74
|
+
):
|
|
75
|
+
for pname in name_dict:
|
|
76
|
+
if pname not in params_info:
|
|
77
|
+
raise ValueError(
|
|
78
|
+
f"未知のパラメータ名です({label}): '{pname}' "
|
|
79
|
+
f"(このフィットタイプのパラメータ: {params_info})"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
if fixed_params and len(fixed_params) >= len(params_info):
|
|
83
|
+
raise ValueError(
|
|
84
|
+
"すべてのパラメータを固定することはできません"
|
|
85
|
+
"(最適化する自由パラメータが1つも残りません)。"
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
p0 = list(p0)
|
|
89
|
+
for i, name in enumerate(params_info):
|
|
90
|
+
if name in p0_overrides:
|
|
91
|
+
p0[i] = float(p0_overrides[name])
|
|
92
|
+
|
|
93
|
+
free_indices = [i for i, name in enumerate(params_info) if name not in fixed_params]
|
|
94
|
+
fixed_indices = [i for i, name in enumerate(params_info) if name in fixed_params]
|
|
95
|
+
fixed_values = {i: float(fixed_params[params_info[i]]) for i in fixed_indices}
|
|
96
|
+
|
|
97
|
+
if fixed_indices:
|
|
98
|
+
# curve_fit には自由パラメータだけを渡し、固定値は関数の中で元の位置に挿し込む
|
|
99
|
+
original_fit_func = fit_func
|
|
100
|
+
|
|
101
|
+
def fit_func_for_curve_fit(x: Any, *free_args: float) -> Any:
|
|
102
|
+
full_params: list[float | None] = [None] * len(params_info)
|
|
103
|
+
for i in fixed_indices:
|
|
104
|
+
full_params[i] = fixed_values[i]
|
|
105
|
+
for idx, i in enumerate(free_indices):
|
|
106
|
+
full_params[i] = free_args[idx]
|
|
107
|
+
return original_fit_func(x, *full_params)
|
|
108
|
+
|
|
109
|
+
p0_for_curve_fit = [p0[i] for i in free_indices]
|
|
110
|
+
else:
|
|
111
|
+
fit_func_for_curve_fit = fit_func
|
|
112
|
+
p0_for_curve_fit = p0
|
|
113
|
+
|
|
114
|
+
curve_fit_kwargs = {
|
|
115
|
+
"p0": p0_for_curve_fit,
|
|
116
|
+
"sigma": sigma,
|
|
117
|
+
"absolute_sigma": sigma is not None,
|
|
118
|
+
}
|
|
119
|
+
if bounds:
|
|
120
|
+
lower, upper = [], []
|
|
121
|
+
for i in free_indices:
|
|
122
|
+
lo, hi = bounds.get(params_info[i], (-np.inf, np.inf))
|
|
123
|
+
lower.append(lo)
|
|
124
|
+
upper.append(hi)
|
|
125
|
+
# bounds は p0 が境界の厳密に内側にあることを要求する(等しいだけで例外)。
|
|
126
|
+
# 「初期値=下限」のような自然な入力で落ちないよう、わずかに内側へずらす。
|
|
127
|
+
for idx in range(len(p0_for_curve_fit)):
|
|
128
|
+
lo, hi = lower[idx], upper[idx]
|
|
129
|
+
val = p0_for_curve_fit[idx]
|
|
130
|
+
if val <= lo or val >= hi:
|
|
131
|
+
span = hi - lo
|
|
132
|
+
nudge = span * 1e-6 if np.isfinite(span) and span > 0 else max(abs(val), 1.0) * 1e-6 or 1e-9
|
|
133
|
+
p0_for_curve_fit[idx] = min(max(val, lo + nudge), hi - nudge)
|
|
134
|
+
curve_fit_kwargs["bounds"] = (lower, upper)
|
|
135
|
+
# bounds 付きは trf 法になり、maxfev ではなく max_nfev しか受け付けない
|
|
136
|
+
curve_fit_kwargs["max_nfev"] = CURVE_FIT_MAX_ITERATIONS
|
|
137
|
+
else:
|
|
138
|
+
curve_fit_kwargs["maxfev"] = CURVE_FIT_MAX_ITERATIONS
|
|
139
|
+
|
|
140
|
+
if loss != 'linear':
|
|
141
|
+
# 既定の 'lm' は loss を受け付けないので trf にし、maxfev も max_nfev に付け替える
|
|
142
|
+
curve_fit_kwargs["loss"] = loss
|
|
143
|
+
curve_fit_kwargs.setdefault("method", "trf")
|
|
144
|
+
if "maxfev" in curve_fit_kwargs:
|
|
145
|
+
curve_fit_kwargs["max_nfev"] = curve_fit_kwargs.pop("maxfev")
|
|
146
|
+
|
|
147
|
+
try:
|
|
148
|
+
popt_free, pcov_free = curve_fit(fit_func_for_curve_fit, x_data, y_data, **curve_fit_kwargs)
|
|
149
|
+
except RuntimeError as e:
|
|
150
|
+
raise RuntimeError(
|
|
151
|
+
f"フィッティングが収束しませんでした({fit_type_label})。データの分布が"
|
|
152
|
+
f"このモデルに適していない可能性があります。詳細: {e}"
|
|
153
|
+
) from e
|
|
154
|
+
|
|
155
|
+
if fixed_indices:
|
|
156
|
+
popt = np.empty(len(params_info))
|
|
157
|
+
for i in fixed_indices:
|
|
158
|
+
popt[i] = fixed_values[i]
|
|
159
|
+
for idx, i in enumerate(free_indices):
|
|
160
|
+
popt[i] = popt_free[idx]
|
|
161
|
+
|
|
162
|
+
pcov = np.zeros((len(params_info), len(params_info)))
|
|
163
|
+
for row_idx, i in enumerate(free_indices):
|
|
164
|
+
for col_idx, j in enumerate(free_indices):
|
|
165
|
+
pcov[i, j] = pcov_free[row_idx, col_idx]
|
|
166
|
+
else:
|
|
167
|
+
popt, pcov = popt_free, pcov_free
|
|
168
|
+
|
|
169
|
+
return popt, pcov
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def calculate_curve_fit(x_data: Any, y_data: Any, fit_type: str, custom_formula: str | None = None, sigma: Any = None,
|
|
173
|
+
x_range: tuple[float, float] | None = None, p0_overrides: dict[str, float] | None = None,
|
|
174
|
+
fixed_params: dict[str, float] | None = None,
|
|
175
|
+
bounds: dict[str, tuple[float, float]] | None = None, loss: str = 'linear',
|
|
176
|
+
model_id: str | None = None) -> dict[str, Any]:
|
|
177
|
+
"""曲線フィット。popt / pcov / perr / x_fit・y_fit / r_squared / residuals などの dict を返す。
|
|
178
|
+
|
|
179
|
+
x_range は両端を含み、範囲外の点は p0 の推定にも使わない。sigma は absolute_sigma=True で渡す。
|
|
180
|
+
"""
|
|
181
|
+
x_data = np.asarray(x_data)
|
|
182
|
+
y_data = np.asarray(y_data)
|
|
183
|
+
if sigma is not None:
|
|
184
|
+
sigma = np.asarray(sigma)
|
|
185
|
+
if x_range is not None:
|
|
186
|
+
x_min, x_max = x_range
|
|
187
|
+
range_mask = (x_data >= x_min) & (x_data <= x_max)
|
|
188
|
+
x_data, y_data = x_data[range_mask], y_data[range_mask]
|
|
189
|
+
if sigma is not None:
|
|
190
|
+
sigma = sigma[range_mask]
|
|
191
|
+
|
|
192
|
+
# NaN が残ると curve_fit は収束失敗ではない ValueError を出し、p0 の推定も落ちる
|
|
193
|
+
nan_mask = np.isnan(x_data) | np.isnan(y_data)
|
|
194
|
+
if sigma is not None:
|
|
195
|
+
nan_mask |= np.isnan(sigma)
|
|
196
|
+
if nan_mask.any():
|
|
197
|
+
x_data, y_data = x_data[~nan_mask], y_data[~nan_mask]
|
|
198
|
+
if sigma is not None:
|
|
199
|
+
sigma = sigma[~nan_mask]
|
|
200
|
+
|
|
201
|
+
if len(x_data) == 0:
|
|
202
|
+
raise ValueError("有効なデータ点がありません(すべて欠損値です)。フィッティングできません。")
|
|
203
|
+
|
|
204
|
+
model = resolve_fit_model(fit_type, custom_formula, _PLUGIN_FIT_FUNCTIONS, model_id)
|
|
205
|
+
if model.check_data is not None:
|
|
206
|
+
model.check_data(x_data)
|
|
207
|
+
fit_func, params_info = model.func, model.param_names
|
|
208
|
+
p0 = model.initial_guess(x_data, y_data)
|
|
209
|
+
|
|
210
|
+
popt, pcov = _run_curve_fit_with_overrides(
|
|
211
|
+
fit_func, params_info, p0, x_data, y_data, sigma,
|
|
212
|
+
p0_overrides, fixed_params, bounds, fit_type, loss=loss,
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
# 退化したフィットでは pcov の対角が負や inf になる。例外にせず nan / inf のまま返す
|
|
216
|
+
perr = np.sqrt(np.diag(pcov))
|
|
217
|
+
|
|
218
|
+
x_fit = np.linspace(x_data.min(), x_data.max(), 200)
|
|
219
|
+
y_fit = fit_func(x_fit, *popt)
|
|
220
|
+
|
|
221
|
+
residuals = y_data - fit_func(x_data, *popt)
|
|
222
|
+
ss_res = np.sum(residuals ** 2)
|
|
223
|
+
ss_tot = np.sum((y_data - np.mean(y_data)) ** 2)
|
|
224
|
+
r_squared = 1.0 if ss_tot == 0 else 1.0 - (ss_res / ss_tot)
|
|
225
|
+
|
|
226
|
+
return {
|
|
227
|
+
'popt': popt,
|
|
228
|
+
'pcov': pcov,
|
|
229
|
+
'perr': perr,
|
|
230
|
+
'param_names': params_info,
|
|
231
|
+
# 信頼帯の計算用。保存はしない(Dataset.fit_result には入れない)
|
|
232
|
+
'fit_func': fit_func,
|
|
233
|
+
'x_fit': x_fit,
|
|
234
|
+
'y_fit': y_fit,
|
|
235
|
+
'r_squared': r_squared,
|
|
236
|
+
'residuals': residuals,
|
|
237
|
+
'x_data_used': x_data,
|
|
238
|
+
'y_data_used': y_data,
|
|
239
|
+
'loss': loss,
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
# 多峰分離の成分関数。ベースラインを全成分で共有するので、単峰版と違いオフセットを持たない。
|
|
244
|
+
|
|
245
|
+
def _gaussian_component(x: Any, a: float, b: float, c: float) -> Any:
|
|
246
|
+
return a * np.exp(-((x - b) ** 2) / (2 * c ** 2))
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _lorentzian_component(x: Any, a: float, b: float, c: float) -> Any:
|
|
250
|
+
return a / (1 + ((x - b) / c) ** 2)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _pseudo_voigt_component(x: Any, a: float, b: float, c: float, eta: float) -> Any:
|
|
254
|
+
lorentzian_shape = 1 / (1 + ((x - b) / c) ** 2)
|
|
255
|
+
gaussian_shape = np.exp(-4 * np.log(2) * ((x - b) / c) ** 2)
|
|
256
|
+
return a * (eta * lorentzian_shape + (1 - eta) * gaussian_shape)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _voigt_component(x: Any, a: float, b: float, sigma: float, gamma: float) -> Any:
|
|
260
|
+
z = ((x - b) + 1j * gamma) / (sigma * np.sqrt(2))
|
|
261
|
+
return a * np.real(wofz(z)) / (sigma * np.sqrt(2 * np.pi))
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _constant_baseline(x: Any, c: float) -> Any:
|
|
265
|
+
return np.full_like(np.asarray(x, dtype=float), c)
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _linear_baseline(x: Any, m: float, b: float) -> Any:
|
|
269
|
+
return m * np.asarray(x, dtype=float) + b
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
_MULTI_PEAK_COMPONENT_TYPES: dict[str, dict[str, Any]] = {
|
|
273
|
+
'gaussian': {'label': 'ガウシアン', 'func': _gaussian_component, 'param_names': ['a', 'b', 'c']},
|
|
274
|
+
'lorentzian': {'label': 'ローレンツ', 'func': _lorentzian_component, 'param_names': ['a', 'b', 'c']},
|
|
275
|
+
'pseudo_voigt': {'label': '擬似フォークト', 'func': _pseudo_voigt_component, 'param_names': ['a', 'b', 'c', 'eta']},
|
|
276
|
+
'voigt': {'label': 'フォークト', 'func': _voigt_component, 'param_names': ['a', 'b', 'sigma', 'gamma']},
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def calculate_multi_peak_fit(x_data: Any, y_data: Any, component_type: str, initial_guesses: list[dict[str, float]],
|
|
281
|
+
baseline_type: str = 'constant', x_range: tuple[float, float] | None = None,
|
|
282
|
+
sigma: Any = None, p0_overrides: dict[str, float] | None = None,
|
|
283
|
+
fixed_params: dict[str, float] | None = None,
|
|
284
|
+
bounds: dict[str, tuple[float, float]] | None = None) -> dict[str, Any]:
|
|
285
|
+
"""同じ種類の N 成分とベースラインを同時にフィットする。
|
|
286
|
+
|
|
287
|
+
initial_guesses は [{'center', 'height', 'width'(FWHM)}, ...]。戻り値は calculate_curve_fit() の
|
|
288
|
+
dict に component_type / n_components / baseline_type / components を足したもの。
|
|
289
|
+
"""
|
|
290
|
+
if component_type not in _MULTI_PEAK_COMPONENT_TYPES:
|
|
291
|
+
raise ValueError(f"不明な成分タイプ: {component_type}")
|
|
292
|
+
if not initial_guesses:
|
|
293
|
+
raise ValueError("少なくとも1つのピークの初期値が必要です。")
|
|
294
|
+
if baseline_type not in ('none', 'constant', 'linear'):
|
|
295
|
+
raise ValueError(f"不明なベースラインタイプ: {baseline_type}")
|
|
296
|
+
|
|
297
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
298
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
299
|
+
if sigma is not None:
|
|
300
|
+
sigma = np.asarray(sigma, dtype=float)
|
|
301
|
+
if x_range is not None:
|
|
302
|
+
x_min, x_max = x_range
|
|
303
|
+
range_mask = (x_data >= x_min) & (x_data <= x_max)
|
|
304
|
+
x_data, y_data = x_data[range_mask], y_data[range_mask]
|
|
305
|
+
if sigma is not None:
|
|
306
|
+
sigma = sigma[range_mask]
|
|
307
|
+
|
|
308
|
+
# calculate_curve_fit() と同じ理由で NaN を除く
|
|
309
|
+
nan_mask = np.isnan(x_data) | np.isnan(y_data)
|
|
310
|
+
if sigma is not None:
|
|
311
|
+
nan_mask |= np.isnan(sigma)
|
|
312
|
+
if nan_mask.any():
|
|
313
|
+
x_data, y_data = x_data[~nan_mask], y_data[~nan_mask]
|
|
314
|
+
if sigma is not None:
|
|
315
|
+
sigma = sigma[~nan_mask]
|
|
316
|
+
|
|
317
|
+
if len(x_data) == 0:
|
|
318
|
+
raise ValueError("有効なデータ点がありません(すべて欠損値です)。フィッティングできません。")
|
|
319
|
+
|
|
320
|
+
component_info = _MULTI_PEAK_COMPONENT_TYPES[component_type]
|
|
321
|
+
component_func = component_info['func']
|
|
322
|
+
component_param_names = component_info['param_names']
|
|
323
|
+
n_component_params = len(component_param_names)
|
|
324
|
+
n_components = len(initial_guesses)
|
|
325
|
+
|
|
326
|
+
params_info: list[str] = []
|
|
327
|
+
p0: list[float] = []
|
|
328
|
+
for i, guess in enumerate(initial_guesses, start=1):
|
|
329
|
+
height = float(guess['height'])
|
|
330
|
+
center = float(guess['center'])
|
|
331
|
+
width = float(guess.get('width') or 1.0) or 1.0
|
|
332
|
+
params_info.extend(f"{p}{i}" for p in component_param_names)
|
|
333
|
+
if component_type == 'gaussian':
|
|
334
|
+
p0.extend([height, center, width])
|
|
335
|
+
elif component_type == 'lorentzian':
|
|
336
|
+
# width は FWHM、c は HWHM
|
|
337
|
+
p0.extend([height, center, (width / 2) or 1.0])
|
|
338
|
+
elif component_type == 'pseudo_voigt':
|
|
339
|
+
# この定義の c は FWHM そのもの
|
|
340
|
+
p0.extend([height, center, width, 0.5])
|
|
341
|
+
else: # voigt
|
|
342
|
+
# 単峰のフォークトと同じ初期値の割り振り
|
|
343
|
+
sigma0 = (width / 2.355) or 1.0
|
|
344
|
+
gamma0 = (width / 4) or 1.0
|
|
345
|
+
amplitude0 = height * sigma0 * np.sqrt(2 * np.pi)
|
|
346
|
+
p0.extend([amplitude0, center, sigma0, gamma0])
|
|
347
|
+
|
|
348
|
+
baseline_func: Callable[..., Any] | None
|
|
349
|
+
if baseline_type == 'constant':
|
|
350
|
+
params_info.append('baseline_c')
|
|
351
|
+
p0.append(float(np.nanmin(y_data)))
|
|
352
|
+
baseline_func = _constant_baseline
|
|
353
|
+
elif baseline_type == 'linear':
|
|
354
|
+
params_info.extend(['baseline_m', 'baseline_b'])
|
|
355
|
+
p0.extend([0.0, float(np.nanmin(y_data))])
|
|
356
|
+
baseline_func = _linear_baseline
|
|
357
|
+
else:
|
|
358
|
+
baseline_func = None
|
|
359
|
+
|
|
360
|
+
def fit_func(x: Any, *params: float) -> Any:
|
|
361
|
+
total = np.zeros_like(np.asarray(x, dtype=float))
|
|
362
|
+
for i in range(n_components):
|
|
363
|
+
comp_params = params[i * n_component_params:(i + 1) * n_component_params]
|
|
364
|
+
total = total + component_func(x, *comp_params)
|
|
365
|
+
if baseline_func is not None:
|
|
366
|
+
baseline_params = params[n_components * n_component_params:]
|
|
367
|
+
total = total + baseline_func(x, *baseline_params)
|
|
368
|
+
return total
|
|
369
|
+
|
|
370
|
+
fit_type_label = f"多峰分離({component_info['label']} x{n_components})"
|
|
371
|
+
popt, pcov = _run_curve_fit_with_overrides(
|
|
372
|
+
fit_func, params_info, p0, x_data, y_data, sigma,
|
|
373
|
+
p0_overrides, fixed_params, bounds, fit_type_label,
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
perr = np.sqrt(np.diag(pcov))
|
|
377
|
+
x_fit = np.linspace(x_data.min(), x_data.max(), 200)
|
|
378
|
+
y_fit = fit_func(x_fit, *popt)
|
|
379
|
+
residuals = y_data - fit_func(x_data, *popt)
|
|
380
|
+
ss_res = np.sum(residuals ** 2)
|
|
381
|
+
ss_tot = np.sum((y_data - np.mean(y_data)) ** 2)
|
|
382
|
+
r_squared = 1.0 if ss_tot == 0 else 1.0 - (ss_res / ss_tot)
|
|
383
|
+
|
|
384
|
+
components = []
|
|
385
|
+
for i in range(n_components):
|
|
386
|
+
start = i * n_component_params
|
|
387
|
+
end = start + n_component_params
|
|
388
|
+
components.append({
|
|
389
|
+
'type': component_type,
|
|
390
|
+
'param_names': list(params_info[start:end]),
|
|
391
|
+
'params': [float(v) for v in popt[start:end]],
|
|
392
|
+
})
|
|
393
|
+
|
|
394
|
+
return {
|
|
395
|
+
'popt': popt,
|
|
396
|
+
'pcov': pcov,
|
|
397
|
+
'perr': perr,
|
|
398
|
+
'param_names': params_info,
|
|
399
|
+
'fit_func': fit_func,
|
|
400
|
+
'x_fit': x_fit,
|
|
401
|
+
'y_fit': y_fit,
|
|
402
|
+
'r_squared': r_squared,
|
|
403
|
+
'residuals': residuals,
|
|
404
|
+
'x_data_used': x_data,
|
|
405
|
+
'y_data_used': y_data,
|
|
406
|
+
'component_type': component_type,
|
|
407
|
+
'n_components': n_components,
|
|
408
|
+
'baseline_type': baseline_type,
|
|
409
|
+
'components': components,
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def fit_curve_task(x_data: Any, y_data: Any, fit_type: str, custom_formula: str | None = None, sigma: Any = None,
|
|
414
|
+
x_range: tuple[float, float] | None = None, p0_overrides: dict[str, float] | None = None,
|
|
415
|
+
fixed_params: dict[str, float] | None = None,
|
|
416
|
+
bounds: dict[str, tuple[float, float]] | None = None, loss: str = 'linear',
|
|
417
|
+
report_progress: Callable[..., Any] | None = None,
|
|
418
|
+
is_cancelled: Callable[[], bool] | None = None) -> dict[str, Any]:
|
|
419
|
+
"""TaskRunner 用。curve_fit は進捗も中断もできないので、report_progress と is_cancelled は受け取るだけ。"""
|
|
420
|
+
return calculate_curve_fit(
|
|
421
|
+
x_data, y_data, fit_type, custom_formula=custom_formula, sigma=sigma, x_range=x_range,
|
|
422
|
+
p0_overrides=p0_overrides, fixed_params=fixed_params, bounds=bounds, loss=loss,
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def multi_peak_fit_task(x_data: Any, y_data: Any, component_type: str, initial_guesses: list[dict[str, float]],
|
|
427
|
+
baseline_type: str = 'constant', x_range: tuple[float, float] | None = None,
|
|
428
|
+
sigma: Any = None, p0_overrides: dict[str, float] | None = None,
|
|
429
|
+
fixed_params: dict[str, float] | None = None,
|
|
430
|
+
bounds: dict[str, tuple[float, float]] | None = None,
|
|
431
|
+
report_progress: Callable[..., Any] | None = None,
|
|
432
|
+
is_cancelled: Callable[[], bool] | None = None) -> dict[str, Any]:
|
|
433
|
+
"""TaskRunner 用。fit_curve_task() と同じく進捗と中断は受け取るだけ。"""
|
|
434
|
+
return calculate_multi_peak_fit(
|
|
435
|
+
x_data, y_data, component_type, initial_guesses, baseline_type=baseline_type,
|
|
436
|
+
x_range=x_range, sigma=sigma, p0_overrides=p0_overrides, fixed_params=fixed_params, bounds=bounds,
|
|
437
|
+
)
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
_INTEGRAL_METHODS = ("trapezoid", "simpson")
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _trapezoid_integrate(y: Any, x: Any) -> float:
|
|
444
|
+
"""NumPy 2.0 で trapz が trapezoid になった。古い版でも動くようにする。"""
|
|
445
|
+
trapezoid_func = getattr(np, "trapezoid", None)
|
|
446
|
+
if trapezoid_func is None:
|
|
447
|
+
trapezoid_func = np.trapz
|
|
448
|
+
return float(trapezoid_func(y, x))
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def calculate_interval_integral(x_data: Any, y_data: Any, x_range: tuple[float, float], method: str = "trapezoid",
|
|
452
|
+
subtract_baseline: bool = False) -> dict[str, Any]:
|
|
453
|
+
"""x_range(両端を含む)で y を x について積分する。
|
|
454
|
+
|
|
455
|
+
subtract_baseline は範囲の両端を結ぶ直線を引いてから積分する。両端の Y は、その X 位置で補間した値。
|
|
456
|
+
"""
|
|
457
|
+
if method not in _INTEGRAL_METHODS:
|
|
458
|
+
raise ValueError(f"未知の積分方法です: {method}")
|
|
459
|
+
if x_range is None or len(x_range) != 2:
|
|
460
|
+
raise ValueError("積分範囲(x_range)を指定してください。")
|
|
461
|
+
|
|
462
|
+
x_min, x_max = float(x_range[0]), float(x_range[1])
|
|
463
|
+
if x_min >= x_max:
|
|
464
|
+
raise ValueError("積分範囲の最小値は最大値より小さい値である必要があります。")
|
|
465
|
+
|
|
466
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
467
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
468
|
+
|
|
469
|
+
# NaN があると積分値が nan になる
|
|
470
|
+
nan_mask = np.isnan(x_data) | np.isnan(y_data)
|
|
471
|
+
if nan_mask.any():
|
|
472
|
+
x_data, y_data = x_data[~nan_mask], y_data[~nan_mask]
|
|
473
|
+
|
|
474
|
+
if len(x_data) == 0:
|
|
475
|
+
raise ValueError("有効なデータ点がありません(すべて欠損値です)。積分できません。")
|
|
476
|
+
|
|
477
|
+
# trapezoid / simpson は X が単調増加であることを前提にする
|
|
478
|
+
order = np.argsort(x_data)
|
|
479
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
480
|
+
|
|
481
|
+
x_min_data, x_max_data = float(x_sorted[0]), float(x_sorted[-1])
|
|
482
|
+
if x_min < x_min_data or x_max > x_max_data:
|
|
483
|
+
raise ValueError(
|
|
484
|
+
f"積分範囲はデータのX範囲({x_min_data:.6g} 〜 {x_max_data:.6g})内で指定してください。"
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
range_mask = (x_sorted >= x_min) & (x_sorted <= x_max)
|
|
488
|
+
x_in, y_in = x_sorted[range_mask], y_sorted[range_mask]
|
|
489
|
+
|
|
490
|
+
if len(x_in) < 2:
|
|
491
|
+
raise ValueError(f"積分範囲内に十分なデータ点がありません(最低2点必要、現在{len(x_in)}点)。")
|
|
492
|
+
|
|
493
|
+
baseline_used = None
|
|
494
|
+
y_for_integration = y_in
|
|
495
|
+
if subtract_baseline:
|
|
496
|
+
y_at_min = float(np.interp(x_min, x_sorted, y_sorted))
|
|
497
|
+
y_at_max = float(np.interp(x_max, x_sorted, y_sorted))
|
|
498
|
+
baseline_used = y_at_min + (y_at_max - y_at_min) * (x_in - x_min) / (x_max - x_min)
|
|
499
|
+
y_for_integration = y_in - baseline_used
|
|
500
|
+
|
|
501
|
+
if method == "trapezoid":
|
|
502
|
+
integral = _trapezoid_integrate(y_for_integration, x_in)
|
|
503
|
+
else:
|
|
504
|
+
integral = float(simpson(y_for_integration, x=x_in))
|
|
505
|
+
|
|
506
|
+
return {
|
|
507
|
+
'integral': integral,
|
|
508
|
+
'method': method,
|
|
509
|
+
'x_range': (x_min, x_max),
|
|
510
|
+
'subtract_baseline': subtract_baseline,
|
|
511
|
+
'x_used': x_in,
|
|
512
|
+
'y_used': y_for_integration,
|
|
513
|
+
'y_raw_used': y_in,
|
|
514
|
+
'baseline_used': baseline_used,
|
|
515
|
+
'n_points': len(x_in),
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def calculate_cumulative_integral(x_data: Any, y_data: Any, method: str = "trapezoid") -> dict[str, Any]:
|
|
520
|
+
"""各点までの累積積分 ∫[x_min, x_i] y dx を返す(先頭は 0)。"""
|
|
521
|
+
if method not in _INTEGRAL_METHODS:
|
|
522
|
+
raise ValueError(f"未知の積分方法です: {method}")
|
|
523
|
+
|
|
524
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
525
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
526
|
+
|
|
527
|
+
nan_mask = np.isnan(x_data) | np.isnan(y_data)
|
|
528
|
+
if nan_mask.any():
|
|
529
|
+
x_data, y_data = x_data[~nan_mask], y_data[~nan_mask]
|
|
530
|
+
|
|
531
|
+
if len(x_data) < 2:
|
|
532
|
+
raise ValueError("有効なデータ点が不足しています(最低2点必要です)。")
|
|
533
|
+
|
|
534
|
+
order = np.argsort(x_data)
|
|
535
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
536
|
+
|
|
537
|
+
if method == "trapezoid":
|
|
538
|
+
y_cumulative = cumulative_trapezoid(y_sorted, x_sorted, initial=0.0)
|
|
539
|
+
else:
|
|
540
|
+
y_cumulative = cumulative_simpson(y_sorted, x=x_sorted, initial=0.0)
|
|
541
|
+
|
|
542
|
+
return {
|
|
543
|
+
'x_used': x_sorted,
|
|
544
|
+
'y_cumulative': y_cumulative,
|
|
545
|
+
'method': method,
|
|
546
|
+
'n_points': len(x_sorted),
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def _peak_detection_signal_and_kwargs(x_data: Any, y_data: Any, peak_type: str,
|
|
551
|
+
settings: dict[str, Any]) -> tuple[np.ndarray, dict[str, Any]]:
|
|
552
|
+
"""calculate_peaks と calculate_peak_quantification の共通の前処理。
|
|
553
|
+
|
|
554
|
+
谷は -y_data で探すので height も符号を反転する(利用者は谷でも「Y < -10」を -10 と入力する)。
|
|
555
|
+
height=None は高さで絞らない。
|
|
556
|
+
"""
|
|
557
|
+
y_data_to_find = y_data
|
|
558
|
+
height = settings["height"]
|
|
559
|
+
|
|
560
|
+
if "下に凸" in peak_type:
|
|
561
|
+
y_data_to_find = -y_data
|
|
562
|
+
if height is not None:
|
|
563
|
+
height = -height
|
|
564
|
+
|
|
565
|
+
kwargs = {"height": height}
|
|
566
|
+
|
|
567
|
+
if settings["prominence"] is not None:
|
|
568
|
+
kwargs["prominence"] = settings["prominence"]
|
|
569
|
+
|
|
570
|
+
if len(x_data) > 1 and settings["distance_x"] > 0:
|
|
571
|
+
sorted_x = np.sort(x_data)
|
|
572
|
+
avg_x_diff = np.mean(np.diff(sorted_x))
|
|
573
|
+
if avg_x_diff > 0:
|
|
574
|
+
kwargs["distance"] = max(1, int(np.ceil(settings["distance_x"] / avg_x_diff)))
|
|
575
|
+
else:
|
|
576
|
+
kwargs["distance"] = 1
|
|
577
|
+
else:
|
|
578
|
+
kwargs["distance"] = 1
|
|
579
|
+
|
|
580
|
+
return y_data_to_find, kwargs
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def calculate_peaks(x_data: Any, y_data: Any, peak_type: str, settings: dict[str, Any]) -> tuple[np.ndarray, np.ndarray]:
|
|
584
|
+
y_data_to_find, kwargs = _peak_detection_signal_and_kwargs(x_data, y_data, peak_type, settings)
|
|
585
|
+
|
|
586
|
+
peak_indices, _ = find_peaks(y_data_to_find, **kwargs)
|
|
587
|
+
|
|
588
|
+
if len(peak_indices) == 0:
|
|
589
|
+
return np.array([]), np.array([])
|
|
590
|
+
|
|
591
|
+
return x_data[peak_indices], y_data[peak_indices]
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def calculate_peak_quantification(x_data: Any, y_data: Any, peak_type: str, settings: dict[str, Any]) -> dict[str, Any]:
|
|
595
|
+
"""ピーク/谷の位置に加え、FWHM・面積・重心を返す。
|
|
596
|
+
|
|
597
|
+
谷は反転した信号の上で計算するので、面積は谷でも正の値(検出方向への突出量)になる。
|
|
598
|
+
ローカル基線は peak_widths(rel_height=1.0) の左右の裾を結ぶ直線(裾の高さが違う
|
|
599
|
+
非対称なピークでも面積と重心が定まる)。重心の重みは生の Y ではなく基線からの高さ
|
|
600
|
+
(大きなオフセットの上のピークで重心が引っ張られない)。x_data の並びがそのまま
|
|
601
|
+
ピークの前後関係になるので、ソートは呼び出し側で行う。
|
|
602
|
+
"""
|
|
603
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
604
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
605
|
+
|
|
606
|
+
y_signal, kwargs = _peak_detection_signal_and_kwargs(x_data, y_data, peak_type, settings)
|
|
607
|
+
peak_indices, _ = find_peaks(y_signal, **kwargs)
|
|
608
|
+
|
|
609
|
+
empty = np.array([])
|
|
610
|
+
if len(peak_indices) == 0:
|
|
611
|
+
return {'peak_x': empty, 'peak_y': empty, 'fwhm': empty, 'area': empty, 'centroid': empty}
|
|
612
|
+
|
|
613
|
+
idx_axis = np.arange(len(x_data))
|
|
614
|
+
|
|
615
|
+
# 幅はインデックス単位で出るので、x_data に補間して X の単位にする(等間隔でなくてよい)
|
|
616
|
+
_, _, left_ips_half, right_ips_half = peak_widths(y_signal, peak_indices, rel_height=0.5)
|
|
617
|
+
x_left_half = np.interp(left_ips_half, idx_axis, x_data)
|
|
618
|
+
x_right_half = np.interp(right_ips_half, idx_axis, x_data)
|
|
619
|
+
fwhm = x_right_half - x_left_half
|
|
620
|
+
|
|
621
|
+
# rel_height=1.0 は prominence を測った高さでの幅、つまりピークの裾
|
|
622
|
+
_, _, left_ips_full, right_ips_full = peak_widths(y_signal, peak_indices, rel_height=1.0)
|
|
623
|
+
|
|
624
|
+
area = np.empty(len(peak_indices))
|
|
625
|
+
centroid = np.empty(len(peak_indices))
|
|
626
|
+
|
|
627
|
+
for i, pk in enumerate(peak_indices):
|
|
628
|
+
li, ri = left_ips_full[i], right_ips_full[i]
|
|
629
|
+
|
|
630
|
+
x_left = np.interp(li, idx_axis, x_data)
|
|
631
|
+
x_right = np.interp(ri, idx_axis, x_data)
|
|
632
|
+
y_left = np.interp(li, idx_axis, y_signal)
|
|
633
|
+
y_right = np.interp(ri, idx_axis, y_signal)
|
|
634
|
+
|
|
635
|
+
li_idx, ri_idx = int(np.floor(li)), int(np.ceil(ri))
|
|
636
|
+
inner_idx = np.arange(max(li_idx, 0), min(ri_idx, len(x_data) - 1) + 1)
|
|
637
|
+
xs_inner, ys_inner = x_data[inner_idx], y_signal[inner_idx]
|
|
638
|
+
inner_mask = (xs_inner >= x_left) & (xs_inner <= x_right)
|
|
639
|
+
xs_inner, ys_inner = xs_inner[inner_mask], ys_inner[inner_mask]
|
|
640
|
+
|
|
641
|
+
xs = np.concatenate(([x_left], xs_inner, [x_right]))
|
|
642
|
+
ys = np.concatenate(([y_left], ys_inner, [y_right]))
|
|
643
|
+
xs, unique_order = np.unique(xs, return_index=True)
|
|
644
|
+
ys = ys[unique_order]
|
|
645
|
+
|
|
646
|
+
baseline = np.interp(xs, [x_left, x_right], [y_left, y_right])
|
|
647
|
+
y_above = ys - baseline
|
|
648
|
+
|
|
649
|
+
area[i] = np.trapezoid(y_above, xs)
|
|
650
|
+
|
|
651
|
+
weights = np.clip(y_above, 0, None)
|
|
652
|
+
total_weight = np.sum(weights)
|
|
653
|
+
if total_weight > 0:
|
|
654
|
+
centroid[i] = np.sum(xs * weights) / total_weight
|
|
655
|
+
else:
|
|
656
|
+
centroid[i] = x_data[pk]
|
|
657
|
+
|
|
658
|
+
return {
|
|
659
|
+
'peak_x': x_data[peak_indices],
|
|
660
|
+
'peak_y': y_data[peak_indices],
|
|
661
|
+
'fwhm': fwhm,
|
|
662
|
+
'area': area,
|
|
663
|
+
'centroid': centroid,
|
|
664
|
+
}
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
def calculate_savgol(x_data: Any, y_data: Any, window_length: int, polyorder: int,
|
|
668
|
+
deriv: int = 0) -> tuple[np.ndarray, np.ndarray]:
|
|
669
|
+
"""Savitzky-Golay による平滑化(deriv=0)または微分(deriv=1, 2)。
|
|
670
|
+
|
|
671
|
+
delta に X の間隔の中央値を渡すので、微分は dy/dx の大きさになる(X がほぼ等間隔である前提)。
|
|
672
|
+
"""
|
|
673
|
+
if window_length % 2 == 0:
|
|
674
|
+
raise ValueError("窓幅(window_length)は奇数である必要があります。")
|
|
675
|
+
if polyorder >= window_length:
|
|
676
|
+
raise ValueError("多項式の次数は窓幅より小さくする必要があります。")
|
|
677
|
+
if window_length > len(y_data):
|
|
678
|
+
raise ValueError(f"窓幅({window_length})がデータ点数({len(y_data)})を超えています。")
|
|
679
|
+
|
|
680
|
+
order = np.argsort(x_data)
|
|
681
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
682
|
+
diffs = np.diff(x_sorted)
|
|
683
|
+
dx = float(np.median(diffs)) if len(diffs) > 0 and np.median(diffs) > 0 else 1.0
|
|
684
|
+
|
|
685
|
+
y_result = savgol_filter(y_sorted, window_length, polyorder, deriv=deriv, delta=dx)
|
|
686
|
+
return x_sorted, y_result
|
|
687
|
+
|
|
688
|
+
|
|
689
|
+
# 線の平滑化(Dataset.smoothing_method)。描画用で元データは変えない。
|
|
690
|
+
# どれも X でソートして実データ点のまま返す(CubicSpline と違い点を補間で増やさない)。
|
|
691
|
+
|
|
692
|
+
def calculate_moving_average_smooth(x_data: Any, y_data: Any, window: int = 5) -> tuple[np.ndarray, np.ndarray]:
|
|
693
|
+
"""window は点数。データ点数を超えたら切り詰める。端は端の値を延長する。"""
|
|
694
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
695
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
696
|
+
order = np.argsort(x_data)
|
|
697
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
698
|
+
|
|
699
|
+
n = len(y_sorted)
|
|
700
|
+
w = max(1, min(int(window), n)) if n > 0 else 1
|
|
701
|
+
if w <= 1:
|
|
702
|
+
return x_sorted, y_sorted.copy()
|
|
703
|
+
return x_sorted, uniform_filter1d(y_sorted, size=w, mode='nearest')
|
|
704
|
+
|
|
705
|
+
|
|
706
|
+
def calculate_median_smooth(x_data: Any, y_data: Any, window: int = 5) -> tuple[np.ndarray, np.ndarray]:
|
|
707
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
708
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
709
|
+
order = np.argsort(x_data)
|
|
710
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
711
|
+
|
|
712
|
+
n = len(y_sorted)
|
|
713
|
+
w = max(1, min(int(window), n)) if n > 0 else 1
|
|
714
|
+
if w <= 1:
|
|
715
|
+
return x_sorted, y_sorted.copy()
|
|
716
|
+
return x_sorted, median_filter(y_sorted, size=w, mode='nearest')
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def calculate_gaussian_smooth(x_data: Any, y_data: Any, sigma: float = 2.0) -> tuple[np.ndarray, np.ndarray]:
|
|
720
|
+
"""sigma はデータ点のインデックス単位(X の間隔ではない)。"""
|
|
721
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
722
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
723
|
+
order = np.argsort(x_data)
|
|
724
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
725
|
+
|
|
726
|
+
if len(y_sorted) < 2 or sigma <= 0:
|
|
727
|
+
return x_sorted, y_sorted.copy()
|
|
728
|
+
return x_sorted, gaussian_filter1d(y_sorted, sigma=float(sigma), mode='nearest')
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
# ベースライン補正。どの手法も (ソート済みの x, ベースライン, 差し引いた y) を返す。
|
|
732
|
+
|
|
733
|
+
def _sort_xy_for_baseline(x_data: Any, y_data: Any) -> tuple[np.ndarray, np.ndarray]:
|
|
734
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
735
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
736
|
+
order = np.argsort(x_data)
|
|
737
|
+
return x_data[order], y_data[order]
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
def calculate_baseline_als(x_data: Any, y_data: Any, lam: float = 1e5, p: float = 0.01,
|
|
741
|
+
niter: int = 10) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
742
|
+
"""Asymmetric Least Squares (Eilers & Boelens 2005)。
|
|
743
|
+
|
|
744
|
+
lam が大きいほど滑らか、p が小さいほどピークを避けて下側を通る(目安 0.001〜0.1)。
|
|
745
|
+
"""
|
|
746
|
+
if lam <= 0:
|
|
747
|
+
raise ValueError("lam(平滑化パラメータ)は正の値である必要があります。")
|
|
748
|
+
if not (0 < p < 1):
|
|
749
|
+
raise ValueError("p(非対称重み)は0より大きく1より小さい値である必要があります。")
|
|
750
|
+
if niter < 1:
|
|
751
|
+
raise ValueError("反復回数(niter)は1以上である必要があります。")
|
|
752
|
+
|
|
753
|
+
x_sorted, y_sorted = _sort_xy_for_baseline(x_data, y_data)
|
|
754
|
+
n_points = len(y_sorted)
|
|
755
|
+
if n_points < 3:
|
|
756
|
+
raise ValueError(f"ALSベースライン補正には少なくとも3点のデータが必要です(現在{n_points}点)。")
|
|
757
|
+
|
|
758
|
+
# 2階差分の二乗和へのペナルティ
|
|
759
|
+
diff_matrix = sparse.diags([1, -2, 1], [0, -1, -2], shape=(n_points, n_points - 2))
|
|
760
|
+
penalty = lam * (diff_matrix @ diff_matrix.transpose())
|
|
761
|
+
|
|
762
|
+
weights = np.ones(n_points)
|
|
763
|
+
baseline = y_sorted.copy()
|
|
764
|
+
for _ in range(niter):
|
|
765
|
+
weight_matrix = sparse.diags(weights, 0, shape=(n_points, n_points))
|
|
766
|
+
baseline = spsolve((weight_matrix + penalty).tocsc(), weights * y_sorted)
|
|
767
|
+
weights = p * (y_sorted > baseline) + (1 - p) * (y_sorted <= baseline)
|
|
768
|
+
|
|
769
|
+
return x_sorted, baseline, y_sorted - baseline
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def calculate_baseline_polynomial(x_data: Any, y_data: Any, degree: int = 3,
|
|
773
|
+
iterations: int = 10) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
774
|
+
"""反復多項式フィット(Lieber & Mahadevan-Jansen 2003 の ModPoly)。
|
|
775
|
+
|
|
776
|
+
フィット曲線を上回る点をフィット値で置き換えて再フィットし、ピークを少しずつ削る。
|
|
777
|
+
"""
|
|
778
|
+
if degree < 0:
|
|
779
|
+
raise ValueError("多項式の次数は0以上である必要があります。")
|
|
780
|
+
if iterations < 1:
|
|
781
|
+
raise ValueError("反復回数(iterations)は1以上である必要があります。")
|
|
782
|
+
|
|
783
|
+
x_sorted, y_sorted = _sort_xy_for_baseline(x_data, y_data)
|
|
784
|
+
if degree >= len(y_sorted):
|
|
785
|
+
raise ValueError(f"多項式の次数({degree})がデータ点数({len(y_sorted)})以上です。")
|
|
786
|
+
|
|
787
|
+
work_y = y_sorted.copy()
|
|
788
|
+
baseline = work_y
|
|
789
|
+
for _ in range(iterations):
|
|
790
|
+
coeffs = np.polyfit(x_sorted, work_y, degree)
|
|
791
|
+
baseline = np.polyval(coeffs, x_sorted)
|
|
792
|
+
work_y = np.minimum(work_y, baseline)
|
|
793
|
+
|
|
794
|
+
return x_sorted, baseline, y_sorted - baseline
|
|
795
|
+
|
|
796
|
+
|
|
797
|
+
def calculate_baseline_rubberband(x_data: Any, y_data: Any) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
798
|
+
"""ラバーバンド法。下側凸包の頂点を区分線形でつないだものをベースラインにする。"""
|
|
799
|
+
x_sorted, y_sorted = _sort_xy_for_baseline(x_data, y_data)
|
|
800
|
+
n_points = len(x_sorted)
|
|
801
|
+
if n_points < 3:
|
|
802
|
+
raise ValueError(f"ラバーバンド法には少なくとも3点のデータが必要です(現在{n_points}点)。")
|
|
803
|
+
|
|
804
|
+
# Andrew の monotone chain の下側だけ
|
|
805
|
+
hull_indices: list[int] = []
|
|
806
|
+
for i in range(n_points):
|
|
807
|
+
while len(hull_indices) >= 2:
|
|
808
|
+
o, a = hull_indices[-2], hull_indices[-1]
|
|
809
|
+
cross = ((x_sorted[a] - x_sorted[o]) * (y_sorted[i] - y_sorted[o])
|
|
810
|
+
- (y_sorted[a] - y_sorted[o]) * (x_sorted[i] - x_sorted[o]))
|
|
811
|
+
if cross <= 0:
|
|
812
|
+
hull_indices.pop()
|
|
813
|
+
else:
|
|
814
|
+
break
|
|
815
|
+
hull_indices.append(i)
|
|
816
|
+
|
|
817
|
+
x_hull = x_sorted[hull_indices]
|
|
818
|
+
y_hull = y_sorted[hull_indices]
|
|
819
|
+
baseline = np.interp(x_sorted, x_hull, y_hull)
|
|
820
|
+
|
|
821
|
+
return x_sorted, baseline, y_sorted - baseline
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def calculate_baseline_manual(x_data: Any, y_data: Any, anchor_x: Any,
|
|
825
|
+
method: str = "linear") -> tuple[np.ndarray, np.ndarray, np.ndarray]:
|
|
826
|
+
"""指定した X 位置でデータを補間した点を、線形("linear")か 3 次スプライン("spline")で結ぶ。"""
|
|
827
|
+
if method not in ("linear", "spline"):
|
|
828
|
+
raise ValueError(f"未知の補間方法です: {method}")
|
|
829
|
+
|
|
830
|
+
x_sorted, y_sorted = _sort_xy_for_baseline(x_data, y_data)
|
|
831
|
+
|
|
832
|
+
anchor_x = np.unique(np.asarray(anchor_x, dtype=float))
|
|
833
|
+
if len(anchor_x) < 2:
|
|
834
|
+
raise ValueError("アンカー点は重複しない値で2点以上指定してください。")
|
|
835
|
+
if method == "spline" and len(anchor_x) < 3:
|
|
836
|
+
raise ValueError("スプライン補間には3点以上のアンカー点が必要です。")
|
|
837
|
+
|
|
838
|
+
x_min, x_max = x_sorted[0], x_sorted[-1]
|
|
839
|
+
if anchor_x[0] < x_min or anchor_x[-1] > x_max:
|
|
840
|
+
raise ValueError(
|
|
841
|
+
f"アンカー点はデータのX範囲({x_min:.6g} 〜 {x_max:.6g})内で指定してください。"
|
|
842
|
+
)
|
|
843
|
+
|
|
844
|
+
anchor_y = np.interp(anchor_x, x_sorted, y_sorted)
|
|
845
|
+
|
|
846
|
+
if method == "linear":
|
|
847
|
+
baseline = np.interp(x_sorted, anchor_x, anchor_y)
|
|
848
|
+
else:
|
|
849
|
+
baseline = CubicSpline(anchor_x, anchor_y)(x_sorted)
|
|
850
|
+
|
|
851
|
+
return x_sorted, baseline, y_sorted - baseline
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def calculate_confidence_band(x_eval: Any, fit_func: Callable[..., Any], popt: Any, pcov: Any, residuals: Any,
|
|
855
|
+
confidence: float = 0.95, band_type: str = "confidence") -> dict[str, Any]:
|
|
856
|
+
"""非線形回帰の信頼帯・予測帯をデルタ法(線形化)で近似する。
|
|
857
|
+
|
|
858
|
+
各点で J(x) @ pcov @ J(x).T を分散とする。予測帯はさらに残差の平均二乗誤差を足す。
|
|
859
|
+
固定パラメータは pcov の行と列が 0 なので不確かさに寄与しない。
|
|
860
|
+
"""
|
|
861
|
+
if band_type not in ("confidence", "prediction"):
|
|
862
|
+
raise ValueError(f"未知のband_typeです: {band_type}")
|
|
863
|
+
if not (0.0 < confidence < 1.0):
|
|
864
|
+
raise ValueError("confidence(信頼水準)は0より大きく1より小さい値である必要があります。")
|
|
865
|
+
|
|
866
|
+
x_eval = np.asarray(x_eval, dtype=float)
|
|
867
|
+
popt = np.asarray(popt, dtype=float)
|
|
868
|
+
pcov = np.asarray(pcov, dtype=float)
|
|
869
|
+
residuals = np.asarray(residuals, dtype=float)
|
|
870
|
+
|
|
871
|
+
n_params = len(popt)
|
|
872
|
+
n_data = len(residuals)
|
|
873
|
+
dof = n_data - n_params
|
|
874
|
+
if dof < 1:
|
|
875
|
+
raise ValueError(
|
|
876
|
+
f"自由度(データ点数{n_data} - パラメータ数{n_params})が1未満のため、"
|
|
877
|
+
"信頼帯・予測帯を計算できません。"
|
|
878
|
+
)
|
|
879
|
+
|
|
880
|
+
from scipy.stats import t as _t_dist
|
|
881
|
+
t_value = _t_dist.ppf(1.0 - (1.0 - confidence) / 2.0, dof)
|
|
882
|
+
|
|
883
|
+
y_center = fit_func(x_eval, *popt)
|
|
884
|
+
jacobian = np.empty((len(x_eval), n_params))
|
|
885
|
+
for i in range(n_params):
|
|
886
|
+
# 刻みはパラメータの大きさに比例させる(極端な値でも数値誤差が出にくい)
|
|
887
|
+
step = max(abs(popt[i]), 1.0) * 1e-6
|
|
888
|
+
popt_plus, popt_minus = popt.copy(), popt.copy()
|
|
889
|
+
popt_plus[i] += step
|
|
890
|
+
popt_minus[i] -= step
|
|
891
|
+
jacobian[:, i] = (fit_func(x_eval, *popt_plus) - fit_func(x_eval, *popt_minus)) / (2 * step)
|
|
892
|
+
|
|
893
|
+
# diag(J @ pcov @ J.T) だけを求める
|
|
894
|
+
var_yhat = np.einsum('ij,jk,ik->i', jacobian, pcov, jacobian)
|
|
895
|
+
# 数値誤差でわずかに負になることがある
|
|
896
|
+
var_yhat = np.clip(var_yhat, 0.0, None)
|
|
897
|
+
|
|
898
|
+
if band_type == "prediction":
|
|
899
|
+
mse = np.sum(residuals ** 2) / dof
|
|
900
|
+
variance = var_yhat + mse
|
|
901
|
+
else:
|
|
902
|
+
variance = var_yhat
|
|
903
|
+
|
|
904
|
+
margin = t_value * np.sqrt(variance)
|
|
905
|
+
|
|
906
|
+
return {
|
|
907
|
+
'y_center': y_center,
|
|
908
|
+
'y_lower': y_center - margin,
|
|
909
|
+
'y_upper': y_center + margin,
|
|
910
|
+
'confidence': confidence,
|
|
911
|
+
'band_type': band_type,
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def calculate_resample_to_grid(x_data: Any, y_data: Any, target_x: Any, method: str = "linear",
|
|
916
|
+
extrapolate: bool = False) -> np.ndarray:
|
|
917
|
+
"""(x_data, y_data) を target_x の格子に補間する。
|
|
918
|
+
|
|
919
|
+
extrapolate=False(既定)は元の X の範囲外を NaN にする。True のとき "linear" は両端の傾きで
|
|
920
|
+
外挿する(np.interp は端の値で止めるだけで外挿しない)。
|
|
921
|
+
"""
|
|
922
|
+
if method not in ("linear", "cubic"):
|
|
923
|
+
raise ValueError(f"未知の補間方法です: {method}")
|
|
924
|
+
|
|
925
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
926
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
927
|
+
target_x = np.asarray(target_x, dtype=float)
|
|
928
|
+
|
|
929
|
+
nan_mask = np.isnan(x_data) | np.isnan(y_data)
|
|
930
|
+
if nan_mask.any():
|
|
931
|
+
x_data, y_data = x_data[~nan_mask], y_data[~nan_mask]
|
|
932
|
+
|
|
933
|
+
if len(x_data) == 0:
|
|
934
|
+
raise ValueError("有効なデータ点がありません(すべて欠損値です)。リサンプリングできません。")
|
|
935
|
+
|
|
936
|
+
order = np.argsort(x_data)
|
|
937
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
938
|
+
# 補間は X が厳密に単調増加であることを要求するので、同じ X は最後の値を残してまとめる。
|
|
939
|
+
# np.unique は最初の出現を返すので、反転してから除く。
|
|
940
|
+
x_rev, y_rev = x_sorted[::-1], y_sorted[::-1]
|
|
941
|
+
x_sorted, rev_unique_indices = np.unique(x_rev, return_index=True)
|
|
942
|
+
y_sorted = y_rev[rev_unique_indices]
|
|
943
|
+
|
|
944
|
+
n_points = len(x_sorted)
|
|
945
|
+
if method == "linear" and n_points < 2:
|
|
946
|
+
raise ValueError(f"線形補間には少なくとも2点のデータが必要です(現在{n_points}点)。")
|
|
947
|
+
if method == "cubic" and n_points < 4:
|
|
948
|
+
raise ValueError(f"3次スプライン補間には少なくとも4点のデータが必要です(現在{n_points}点)。")
|
|
949
|
+
|
|
950
|
+
x_min, x_max = x_sorted[0], x_sorted[-1]
|
|
951
|
+
|
|
952
|
+
if method == "linear":
|
|
953
|
+
if extrapolate:
|
|
954
|
+
# np.interp は範囲外を端の値で止めるだけなので、両端の傾きで外挿する
|
|
955
|
+
result = np.interp(target_x, x_sorted, y_sorted)
|
|
956
|
+
below = target_x < x_min
|
|
957
|
+
if below.any():
|
|
958
|
+
slope = (y_sorted[1] - y_sorted[0]) / (x_sorted[1] - x_sorted[0])
|
|
959
|
+
result[below] = y_sorted[0] + slope * (target_x[below] - x_min)
|
|
960
|
+
above = target_x > x_max
|
|
961
|
+
if above.any():
|
|
962
|
+
slope = (y_sorted[-1] - y_sorted[-2]) / (x_sorted[-1] - x_sorted[-2])
|
|
963
|
+
result[above] = y_sorted[-1] + slope * (target_x[above] - x_max)
|
|
964
|
+
else:
|
|
965
|
+
result = np.interp(target_x, x_sorted, y_sorted)
|
|
966
|
+
out_of_range = (target_x < x_min) | (target_x > x_max)
|
|
967
|
+
result = result.astype(float)
|
|
968
|
+
result[out_of_range] = np.nan
|
|
969
|
+
else:
|
|
970
|
+
spline = CubicSpline(x_sorted, y_sorted, extrapolate=extrapolate)
|
|
971
|
+
result = spline(target_x)
|
|
972
|
+
if not extrapolate:
|
|
973
|
+
out_of_range = (target_x < x_min) | (target_x > x_max)
|
|
974
|
+
result = np.asarray(result, dtype=float)
|
|
975
|
+
result[out_of_range] = np.nan
|
|
976
|
+
|
|
977
|
+
return result
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
def calculate_average_duplicate_x(x_data: Any, y_data: Any) -> dict[str, Any]:
|
|
981
|
+
"""同じ X の行の Y を平均する。重複の「除去」は index ラベルが要るので呼び出し側で行う。"""
|
|
982
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
983
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
984
|
+
|
|
985
|
+
nan_mask = np.isnan(x_data) | np.isnan(y_data)
|
|
986
|
+
if nan_mask.any():
|
|
987
|
+
x_data, y_data = x_data[~nan_mask], y_data[~nan_mask]
|
|
988
|
+
|
|
989
|
+
if len(x_data) == 0:
|
|
990
|
+
raise ValueError("有効なデータ点がありません(すべて欠損値です)。")
|
|
991
|
+
|
|
992
|
+
order = np.argsort(x_data, kind='stable')
|
|
993
|
+
x_sorted, y_sorted = x_data[order], y_data[order]
|
|
994
|
+
|
|
995
|
+
unique_x, inverse, counts = np.unique(x_sorted, return_inverse=True, return_counts=True)
|
|
996
|
+
# NumPy の版によって inverse の形が (N,) か (N, 1) になる
|
|
997
|
+
inverse = np.asarray(inverse).reshape(-1)
|
|
998
|
+
sums = np.zeros(len(unique_x))
|
|
999
|
+
np.add.at(sums, inverse, y_sorted)
|
|
1000
|
+
y_averaged = sums / counts
|
|
1001
|
+
|
|
1002
|
+
return {
|
|
1003
|
+
'x_used': unique_x,
|
|
1004
|
+
'y_averaged': y_averaged,
|
|
1005
|
+
'group_sizes': counts,
|
|
1006
|
+
'n_duplicate_groups': int(np.sum(counts > 1)),
|
|
1007
|
+
'n_points_in': len(x_sorted),
|
|
1008
|
+
'n_points_out': len(unique_x),
|
|
1009
|
+
}
|
|
1010
|
+
|
|
1011
|
+
|
|
1012
|
+
# 外れ値の検出。マスクへの適用は利用者が選ぶので、ここでは検出だけ。
|
|
1013
|
+
# is_outlier は入力と同じ長さと並び(NaN は False)。visible_df.index[is_outlier] でそのまま引けるように。
|
|
1014
|
+
|
|
1015
|
+
def sample_standard_deviation(values: Any) -> float:
|
|
1016
|
+
"""標本標準偏差(n−1 で割る)。有効な値が 2 未満なら NaN。
|
|
1017
|
+
|
|
1018
|
+
アプリ内の「標準偏差」はすべてこれで計算する。場所によって n と n−1 が混ざると同じデータで値が違って見える。
|
|
1019
|
+
"""
|
|
1020
|
+
arr = np.asarray(values, dtype=float)
|
|
1021
|
+
arr = arr[~np.isnan(arr)]
|
|
1022
|
+
if arr.size < 2:
|
|
1023
|
+
return float('nan')
|
|
1024
|
+
return float(np.std(arr, ddof=1))
|
|
1025
|
+
|
|
1026
|
+
|
|
1027
|
+
def calculate_zscore_outliers(y_data: Any, threshold: float = 3.0) -> dict[str, Any]:
|
|
1028
|
+
if threshold <= 0:
|
|
1029
|
+
raise ValueError("しきい値は正の値である必要があります。")
|
|
1030
|
+
|
|
1031
|
+
y = np.asarray(y_data, dtype=float)
|
|
1032
|
+
is_outlier = np.zeros(len(y), dtype=bool)
|
|
1033
|
+
z_scores = np.full(len(y), np.nan)
|
|
1034
|
+
valid = ~np.isnan(y)
|
|
1035
|
+
|
|
1036
|
+
if valid.sum() >= 2:
|
|
1037
|
+
mean = np.mean(y[valid])
|
|
1038
|
+
std = sample_standard_deviation(y[valid])
|
|
1039
|
+
if std > 0:
|
|
1040
|
+
z_scores[valid] = (y[valid] - mean) / std
|
|
1041
|
+
is_outlier[valid] = np.abs(z_scores[valid]) > threshold
|
|
1042
|
+
|
|
1043
|
+
return {
|
|
1044
|
+
'is_outlier': is_outlier,
|
|
1045
|
+
'z_scores': z_scores,
|
|
1046
|
+
'threshold': threshold,
|
|
1047
|
+
'n_outliers': int(np.sum(is_outlier)),
|
|
1048
|
+
}
|
|
1049
|
+
|
|
1050
|
+
|
|
1051
|
+
def calculate_iqr_outliers(y_data: Any, multiplier: float = 1.5) -> dict[str, Any]:
|
|
1052
|
+
"""[Q1 - multiplier*IQR, Q3 + multiplier*IQR] の外を外れ値とする。"""
|
|
1053
|
+
if multiplier <= 0:
|
|
1054
|
+
raise ValueError("係数は正の値である必要があります。")
|
|
1055
|
+
|
|
1056
|
+
y = np.asarray(y_data, dtype=float)
|
|
1057
|
+
is_outlier = np.zeros(len(y), dtype=bool)
|
|
1058
|
+
valid = ~np.isnan(y)
|
|
1059
|
+
lower_bound = upper_bound = None
|
|
1060
|
+
|
|
1061
|
+
if valid.sum() >= 4:
|
|
1062
|
+
q1, q3 = np.percentile(y[valid], [25, 75])
|
|
1063
|
+
iqr = q3 - q1
|
|
1064
|
+
lower_bound = float(q1 - multiplier * iqr)
|
|
1065
|
+
upper_bound = float(q3 + multiplier * iqr)
|
|
1066
|
+
is_outlier[valid] = (y[valid] < lower_bound) | (y[valid] > upper_bound)
|
|
1067
|
+
|
|
1068
|
+
return {
|
|
1069
|
+
'is_outlier': is_outlier,
|
|
1070
|
+
'lower_bound': lower_bound,
|
|
1071
|
+
'upper_bound': upper_bound,
|
|
1072
|
+
'multiplier': multiplier,
|
|
1073
|
+
'n_outliers': int(np.sum(is_outlier)),
|
|
1074
|
+
}
|
|
1075
|
+
|
|
1076
|
+
|
|
1077
|
+
def calculate_lttb_downsample(x_data: Any, y_data: Any, n_out: int) -> np.ndarray:
|
|
1078
|
+
"""Largest-Triangle-Three-Buckets(Steinarsson 2013)による表示用の間引き。
|
|
1079
|
+
|
|
1080
|
+
x_data はソート済みであること。選んだ点のインデックス(先頭と末尾を含む昇順)を返すので、
|
|
1081
|
+
同じインデックスを誤差などほかの配列にも使える。n_out 以下なら間引かない。
|
|
1082
|
+
"""
|
|
1083
|
+
n = len(x_data)
|
|
1084
|
+
if n_out < 3 or n <= n_out:
|
|
1085
|
+
return np.arange(n)
|
|
1086
|
+
|
|
1087
|
+
x_data = np.asarray(x_data, dtype=float)
|
|
1088
|
+
y_data = np.asarray(y_data, dtype=float)
|
|
1089
|
+
|
|
1090
|
+
selected = np.empty(n_out, dtype=np.int64)
|
|
1091
|
+
selected[0] = 0
|
|
1092
|
+
selected[-1] = n - 1
|
|
1093
|
+
|
|
1094
|
+
bucket_edges = np.linspace(1, n - 1, n_out - 1).astype(np.int64)
|
|
1095
|
+
a = 0
|
|
1096
|
+
for i in range(n_out - 2):
|
|
1097
|
+
bucket_start, bucket_end = bucket_edges[i], bucket_edges[i + 1]
|
|
1098
|
+
if bucket_end <= bucket_start:
|
|
1099
|
+
bucket_end = bucket_start + 1
|
|
1100
|
+
next_start = bucket_end
|
|
1101
|
+
next_end = bucket_edges[i + 2] if i + 2 < len(bucket_edges) else n
|
|
1102
|
+
if next_end <= next_start:
|
|
1103
|
+
next_end = min(next_start + 1, n)
|
|
1104
|
+
avg_x = np.mean(x_data[next_start:next_end])
|
|
1105
|
+
avg_y = np.mean(y_data[next_start:next_end])
|
|
1106
|
+
|
|
1107
|
+
cand_x = x_data[bucket_start:bucket_end]
|
|
1108
|
+
cand_y = y_data[bucket_start:bucket_end]
|
|
1109
|
+
ax_, ay_ = x_data[a], y_data[a]
|
|
1110
|
+
areas = np.abs(
|
|
1111
|
+
(ax_ - avg_x) * (cand_y - ay_) - (ax_ - cand_x) * (avg_y - ay_)
|
|
1112
|
+
)
|
|
1113
|
+
best_local = np.argmax(areas)
|
|
1114
|
+
chosen = bucket_start + best_local
|
|
1115
|
+
selected[i + 1] = chosen
|
|
1116
|
+
a = chosen
|
|
1117
|
+
|
|
1118
|
+
return selected
|
|
1119
|
+
|
|
1120
|
+
|
|
1121
|
+
def calculate_histogram(data: Any, bins: Any = 'auto', density: bool = False) -> dict[str, Any]:
|
|
1122
|
+
"""ビン中心と度数(density=True なら確率密度)を返す。"""
|
|
1123
|
+
data = np.asarray(data, dtype=float)
|
|
1124
|
+
data = data[~np.isnan(data)]
|
|
1125
|
+
if len(data) == 0:
|
|
1126
|
+
raise ValueError("有効なデータ点がありません(すべて欠損値です)。")
|
|
1127
|
+
counts, bin_edges = np.histogram(data, bins=bins, density=density)
|
|
1128
|
+
bin_centers = (bin_edges[:-1] + bin_edges[1:]) / 2
|
|
1129
|
+
return {
|
|
1130
|
+
'bin_centers': bin_centers,
|
|
1131
|
+
'bin_edges': bin_edges,
|
|
1132
|
+
'counts': counts,
|
|
1133
|
+
'n_points_used': len(data),
|
|
1134
|
+
}
|
|
1135
|
+
|
|
1136
|
+
|
|
1137
|
+
def calculate_kde(data: Any, n_points: int = 200, bw_method: Any = None) -> dict[str, Any]:
|
|
1138
|
+
data = np.asarray(data, dtype=float)
|
|
1139
|
+
data = data[~np.isnan(data)]
|
|
1140
|
+
if len(data) < 2:
|
|
1141
|
+
raise ValueError("カーネル密度推定には少なくとも2点の有効なデータが必要です。")
|
|
1142
|
+
try:
|
|
1143
|
+
kde = gaussian_kde(data, bw_method=bw_method)
|
|
1144
|
+
x_grid = np.linspace(data.min(), data.max(), n_points)
|
|
1145
|
+
density = kde(x_grid)
|
|
1146
|
+
except np.linalg.LinAlgError:
|
|
1147
|
+
raise ValueError("データにばらつきが無いため、カーネル密度推定を計算できません。") from None
|
|
1148
|
+
return {'x_grid': x_grid, 'density': density, 'n_points_used': len(data)}
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
def calculate_error_propagation(operation: str, value_a: Any, error_a: Any, value_b: Any, error_b: Any) -> np.ndarray:
|
|
1152
|
+
"""独立な誤差を仮定した誤差伝播。
|
|
1153
|
+
|
|
1154
|
+
除算は、分子がゼロのときに NaN にならないよう、比ではなく分母で割る形の式にしている。
|
|
1155
|
+
"""
|
|
1156
|
+
a = np.asarray(value_a, dtype=float)
|
|
1157
|
+
ea = np.asarray(error_a, dtype=float)
|
|
1158
|
+
b = np.asarray(value_b, dtype=float)
|
|
1159
|
+
eb = np.asarray(error_b, dtype=float)
|
|
1160
|
+
|
|
1161
|
+
if operation in ("A - B", "B - A", "A + B"):
|
|
1162
|
+
return np.sqrt(ea ** 2 + eb ** 2)
|
|
1163
|
+
if operation == "A × B":
|
|
1164
|
+
return np.sqrt((ea * b) ** 2 + (eb * a) ** 2)
|
|
1165
|
+
with np.errstate(divide='ignore', invalid='ignore'):
|
|
1166
|
+
if operation == "A ÷ B":
|
|
1167
|
+
return np.sqrt((ea / b) ** 2 + (a * eb / b ** 2) ** 2)
|
|
1168
|
+
else:
|
|
1169
|
+
return np.sqrt((eb / a) ** 2 + (b * ea / a ** 2) ** 2)
|
|
1170
|
+
|
|
1171
|
+
|
|
1172
|
+
# 共通グリッドの点数の上限(大きいデータでフリーズしないため)
|
|
1173
|
+
_XCORR_MAX_GRID_POINTS = 20000
|
|
1174
|
+
|
|
1175
|
+
|
|
1176
|
+
def calculate_cross_correlation_alignment(x_a: Any, y_a: Any, x_b: Any, y_b: Any) -> dict[str, float]:
|
|
1177
|
+
"""B を A に重ねるための X のシフト量(x_b + shift)を相互相関のピークから求める。
|
|
1178
|
+
|
|
1179
|
+
両者の範囲全体を覆う等間隔グリッドに補間し(範囲外は 0)、平均を引いてから相関を取る。
|
|
1180
|
+
"""
|
|
1181
|
+
x_a = np.asarray(x_a, dtype=float)
|
|
1182
|
+
y_a = np.asarray(y_a, dtype=float)
|
|
1183
|
+
x_b = np.asarray(x_b, dtype=float)
|
|
1184
|
+
y_b = np.asarray(y_b, dtype=float)
|
|
1185
|
+
|
|
1186
|
+
valid_a = ~(np.isnan(x_a) | np.isnan(y_a))
|
|
1187
|
+
valid_b = ~(np.isnan(x_b) | np.isnan(y_b))
|
|
1188
|
+
x_a, y_a = x_a[valid_a], y_a[valid_a]
|
|
1189
|
+
x_b, y_b = x_b[valid_b], y_b[valid_b]
|
|
1190
|
+
if len(x_a) < 2 or len(x_b) < 2:
|
|
1191
|
+
raise ValueError("有効なデータ点が不足しています(それぞれ最低2点必要です)。")
|
|
1192
|
+
|
|
1193
|
+
order_a = np.argsort(x_a)
|
|
1194
|
+
x_a, y_a = x_a[order_a], y_a[order_a]
|
|
1195
|
+
order_b = np.argsort(x_b)
|
|
1196
|
+
x_b, y_b = x_b[order_b], y_b[order_b]
|
|
1197
|
+
|
|
1198
|
+
step_a = np.median(np.diff(x_a))
|
|
1199
|
+
step_b = np.median(np.diff(x_b))
|
|
1200
|
+
candidates = [s for s in (step_a, step_b) if s and s > 0]
|
|
1201
|
+
if not candidates:
|
|
1202
|
+
raise ValueError("X間隔を推定できません(Xの値が一定、または重複しています)。")
|
|
1203
|
+
step = min(candidates)
|
|
1204
|
+
|
|
1205
|
+
lo = min(x_a.min(), x_b.min())
|
|
1206
|
+
hi = max(x_a.max(), x_b.max())
|
|
1207
|
+
n_points = int(np.ceil((hi - lo) / step)) + 1
|
|
1208
|
+
if n_points > _XCORR_MAX_GRID_POINTS:
|
|
1209
|
+
step = (hi - lo) / _XCORR_MAX_GRID_POINTS
|
|
1210
|
+
n_points = _XCORR_MAX_GRID_POINTS + 1
|
|
1211
|
+
grid = np.linspace(lo, hi, n_points)
|
|
1212
|
+
|
|
1213
|
+
# 範囲外を含む全体で平均を引くので中心化は厳密でないが、ピーク位置を探すには足りる
|
|
1214
|
+
sig_a = np.interp(grid, x_a, y_a, left=0.0, right=0.0)
|
|
1215
|
+
sig_b = np.interp(grid, x_b, y_b, left=0.0, right=0.0)
|
|
1216
|
+
sig_a = sig_a - sig_a.mean()
|
|
1217
|
+
sig_b = sig_b - sig_b.mean()
|
|
1218
|
+
|
|
1219
|
+
corr = correlate(sig_a, sig_b, mode='full')
|
|
1220
|
+
lags = correlation_lags(len(sig_a), len(sig_b), mode='full')
|
|
1221
|
+
best_idx = int(np.argmax(corr))
|
|
1222
|
+
shift = float(lags[best_idx] * step)
|
|
1223
|
+
|
|
1224
|
+
return {'shift': shift, 'grid_step': float(step), 'correlation_peak': float(corr[best_idx])}
|
|
1225
|
+
|
|
1226
|
+
|
|
1227
|
+
def assign_peak_label_levels(peak_x_values: Any, x_axis_span: float, proximity_ratio: float = 0.05,
|
|
1228
|
+
max_levels: int = 4) -> list[int]:
|
|
1229
|
+
"""X が近いピークのラベルに互い違いの段番号を割り当てる(ずらす量は呼び出し側で決める)。"""
|
|
1230
|
+
peak_x_values = list(peak_x_values)
|
|
1231
|
+
if x_axis_span <= 0 or not peak_x_values:
|
|
1232
|
+
return [0] * len(peak_x_values)
|
|
1233
|
+
|
|
1234
|
+
threshold = x_axis_span * proximity_ratio
|
|
1235
|
+
levels = []
|
|
1236
|
+
prev_x = None
|
|
1237
|
+
current_level = 0
|
|
1238
|
+
for x in peak_x_values:
|
|
1239
|
+
if prev_x is not None and abs(x - prev_x) < threshold:
|
|
1240
|
+
current_level = (current_level + 1) % max_levels
|
|
1241
|
+
else:
|
|
1242
|
+
current_level = 0
|
|
1243
|
+
levels.append(current_level)
|
|
1244
|
+
prev_x = x
|
|
1245
|
+
return levels
|
|
1246
|
+
|
|
1247
|
+
|
|
1248
|
+
def split_dataframe_by_column(df: Any, split_col: str) -> dict[str, Any]:
|
|
1249
|
+
"""区分列の値ごとに DataFrame を分ける(long 形式のデータ)。
|
|
1250
|
+
|
|
1251
|
+
グループはファイル中の初出順(測定順などの並びに意味があることが多い)。
|
|
1252
|
+
区分列が NaN の行は除き、その数を n_dropped で返す。各グループの index は 0 から振り直す。
|
|
1253
|
+
"""
|
|
1254
|
+
if split_col not in df.columns:
|
|
1255
|
+
raise ValueError(f"列 '{split_col}' がデータに存在しません。")
|
|
1256
|
+
|
|
1257
|
+
values = df[split_col]
|
|
1258
|
+
keep = values.notna()
|
|
1259
|
+
n_dropped = int((~keep).sum())
|
|
1260
|
+
filtered = df[keep]
|
|
1261
|
+
|
|
1262
|
+
groups = []
|
|
1263
|
+
for label, sub_df in filtered.groupby(split_col, sort=False, observed=True):
|
|
1264
|
+
groups.append((_format_group_label(label), sub_df.reset_index(drop=True)))
|
|
1265
|
+
|
|
1266
|
+
return {'groups': groups, 'n_dropped': n_dropped, 'n_groups': len(groups)}
|
|
1267
|
+
|
|
1268
|
+
|
|
1269
|
+
def _format_group_label(value: Any) -> str:
|
|
1270
|
+
"""numpy のスカラは str() すると 'np.float64(1.5)' になるので、Python の型にしてから文字列にする。"""
|
|
1271
|
+
if hasattr(value, 'item'):
|
|
1272
|
+
try:
|
|
1273
|
+
value = value.item()
|
|
1274
|
+
except (ValueError, TypeError):
|
|
1275
|
+
pass
|
|
1276
|
+
if isinstance(value, float):
|
|
1277
|
+
return f"{value:g}"
|
|
1278
|
+
return str(value)
|