D4Xgui 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,15 @@
1
+ [theme]
2
+ base="dark"
3
+ primaryColor= '#aa80f7'#aa80f7'#"#7070A0" # Dark Purple-Blue primaryColor="#3B4A59" # Dark Blue for primary elements
4
+ backgroundColor='#0E1117'#"#121212" # Very Dark Gray for background
5
+ secondaryBackgroundColor="#262730" # Darker Gray for secondary background
6
+ textColor="#FAFAFA" # Lightest Gray for text
7
+ font="sans serif"
8
+
9
+
10
+ [browser]
11
+ gatherUsageStats = false
12
+
13
+ [server]
14
+ showEmailPrompt = false
15
+ port = 1337
File without changes
D4Xgui/Welcome.py ADDED
@@ -0,0 +1,115 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ import os.path
4
+ from pathlib import Path
5
+ from typing import Dict, Any
6
+
7
+ import streamlit as st
8
+ import D47crunch
9
+
10
+ from tools.page_config import set_page_config
11
+ from tools import sidebar_logo
12
+ from tools.database import init_db
13
+ from tools.init_params import INIT_PARAMS
14
+
15
+
16
+
17
+ class WelcomePageManager:
18
+ """Manages the welcome page content and initialization."""
19
+
20
+ def __init__(self):
21
+ """Initialize the welcome page manager."""
22
+ self.app_dir = Path(__file__).parent.absolute()
23
+ self.session_state = st.session_state
24
+ self._setup_page()
25
+ self._initialize_database()
26
+ self._initialize_session_state()
27
+
28
+ def _setup_page(self) -> None:
29
+ """Set up the page configuration and sidebar."""
30
+ set_page_config(0)
31
+ sidebar_logo.add_logo()
32
+
33
+ def _initialize_database(self) -> None:
34
+ """Initialize the application database."""
35
+ init_db()
36
+
37
+ def _initialize_session_state(self) -> None:
38
+ """Initialize session state with default parameters if not present."""
39
+ if 'standards_nominal' not in self.session_state:
40
+ self.session_state['standards_nominal'] = {**INIT_PARAMS['standards_nominal']}
41
+ #self.session_state['working_gas'] = {**INIT_PARAMS['working_gas']}
42
+
43
+ def _get_info_content(self) -> str:
44
+ """Generate the welcome page information content.
45
+
46
+ Returns:
47
+ Formatted markdown string with welcome information.
48
+ """
49
+ return rf"""## Welcome to D4Xgui v1.0.0!
50
+
51
+ [D4Xgui](https://github.com/itsMig/D4Xgui) is developed to enable easy access to state-of-the-art CO₂ clumped isotope (∆₄₇, ∆₄₈ and ∆₄₉) data processing.
52
+ A recently developed optimizer algorithm allows pre-processing of mass spectrometric raw intensities utilizing a m/z47.5 half-mass Faraday cup correction to account for the effect of a negative pressure baseline, which is essential for accurate and highest precision clumped isotope analysis of CO₂ ([Bernecker et al., 2023](https://doi.org/10.1016/j.chemgeo.2023.121803)).
53
+ It is backed with the recently published processing tool [D47crunch (v.{D47crunch.__version__})](https://github.com/mdaeron/D47crunch) (following the methodology outlined in [Daeron, 2021](https://doi.org/10.1029/2020GC009592)), which allows standardization under consideration of full error propagation and has been used for the InterCarb community effort ([Bernasconi et al., 2021](https://doi.org/10.1029/2020GC009588)).
54
+ This web-app allows users to discover replicate- or sample-based processing results in interactive spreadsheets and plots.
55
+
56
+ <br>
57
+
58
+ Example data is accessible from the Data-IO page, or can be downloaded from [GitHub](https://github.com/itsMig/D4Xgui/tree/main/D4Xgui/static).
59
+
60
+ <br>
61
+
62
+ In order to process post-background corrected data (Data-IO page, Upload δ⁴⁵-δ⁴⁹ replicates tab), the following columns need to be provided (`.xlsx`, `.csv`):
63
+
64
+ | `UID` | `Sample` | `Session` | `Timetag` | `d45` | `d46` | `d47` | `d48` | `d49` |
65
+ |----|----|----|----|----|----|----|----|----|
66
+
67
+ $~$
68
+
69
+ Baseline correction can be performed using a m/z47.5 half-mass cup. For this purpose either a set of equilibrated gases (via heated-gas-line), or carbonate standards (via target values) is used to determine m/z47, m/z48 and m/z49-specific scaling factors. Please upload a cycle-based spreadsheet (`.xlsx`, `.csv`) including the following columns (Data-IO page, Upload m/z44-m/z49 intensities tab):
70
+
71
+ | `UID` | `Sample` | `Session` | `Timetag` | `Replicate` |
72
+ |----|----|----|----|----|
73
+
74
+ | `raw_r44` | `raw_r45` | `raw_r46` | `raw_r47` | `raw_r48` | `raw_r49` | `raw_r47.5` |
75
+ |----|----|----|----|----|----|----|
76
+
77
+ | `raw_s44` | `raw_s45` | `raw_s46` | `raw_s47` | `raw_s48` | `raw_s49` | `raw_s47.5` |
78
+ |----|----|----|----|----|----|----|
79
+
80
+
81
+ <br><br>
82
+ Please find the [code documentation here](https://itsmig.github.io/D4Xgui/index.html).
83
+ """
84
+
85
+ def _save_readme(self, content: str) -> None:
86
+ """Save the welcome content to README.md file.
87
+
88
+ Args:
89
+ content: Content to save to README.md.
90
+ """
91
+ try:
92
+ with open(os.path.join(self.app_dir,'../INSTALLATION.md', ),'r') as file:
93
+ content += f"\n" + file.read()
94
+
95
+ readme_path = Path('../README.md')
96
+ readme_path.write_text(content, encoding='utf-8')
97
+ except Exception as e:
98
+ st.warning(f"Could not save README.md: {e}")
99
+
100
+ def display_welcome_page(self) -> None:
101
+ """Display the welcome page content."""
102
+
103
+ info_content = self._get_info_content()
104
+ # self._save_readme(info_content)
105
+ st.markdown(info_content, unsafe_allow_html=True)
106
+
107
+
108
+ def main():
109
+ """Main function to run the welcome page."""
110
+ welcome_manager = WelcomePageManager()
111
+ welcome_manager.display_welcome_page()
112
+
113
+
114
+ if __name__ == "__main__":
115
+ main()
D4Xgui/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """D4Xgui - Processing app for mass-spectrometric CO₂ clumped isotope (∆₄₇, ∆₄₈ and ∆₄₉) data"""
2
+
3
+ __version__ = "1.0.0"
@@ -0,0 +1,508 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+
4
+ import os
5
+ import re
6
+ import base64
7
+ import io
8
+ from typing import List, Optional, Any
9
+
10
+ import pandas as pd
11
+ import streamlit as st
12
+ import plotly.express as px
13
+
14
+ from tools.authenticator import Authenticator
15
+ from tools.page_config import PageConfigManager
16
+ from tools.sidebar_logo import SidebarLogoManager
17
+ from tools.database import DatabaseManager
18
+ from tools.init_params import IsotopeStandards
19
+
20
+
21
+ class DataIOPage:
22
+ """Manages the Data I/O page of the D4Xgui application."""
23
+
24
+ def __init__(self):
25
+ """Initialize the DataIOPage."""
26
+ self.sss = st.session_state
27
+ self.db_manager = DatabaseManager()
28
+ self._setup_page()
29
+ self._initialize_session_state()
30
+
31
+ def _setup_page(self) -> None:
32
+ """Set up page configuration, logo, and authentication."""
33
+ page_config_manager = PageConfigManager()
34
+ page_config_manager.configure_page(page_number=1)
35
+
36
+ logo_manager = SidebarLogoManager()
37
+ logo_manager.add_logo()
38
+
39
+ if "PYTEST_CURRENT_TEST" not in os.environ:
40
+ authenticator = Authenticator()
41
+ if not authenticator.require_authentication():
42
+ st.stop()
43
+
44
+ def _initialize_session_state(self) -> None:
45
+ """Initialize session state with default parameters if not present."""
46
+ if 'standards_nominal' not in self.sss:
47
+ self.sss['standards_nominal'] = IsotopeStandards.get_standards()
48
+ #self.sss['working_gas'] = IsotopeStandards.get_working_gas()
49
+ if 'show_overwrite_button' not in self.sss:
50
+ self.sss.show_overwrite_button = False
51
+
52
+ def run(self) -> None:
53
+ """Run the main application page."""
54
+ st.title("Data I/O")
55
+ tab1, tab2 = st.tabs(['Upload data', 'Access database'])
56
+ with tab1:
57
+ self._upload_tab()
58
+ with tab2:
59
+ self._database_tab()
60
+
61
+ def _upload_tab(self) -> None:
62
+ """Render the data upload tab."""
63
+ st.header("Upload Data")
64
+ self._render_replicates_uploader()
65
+ self._render_intensities_uploader()
66
+
67
+ def _render_replicates_uploader(self) -> None:
68
+ """Render the UI for uploading pre-processed replicates."""
69
+ st.subheader("Pre-processed replicates")
70
+ st.markdown(r"Drag & Drop pre-processed $\delta^{45-49}$ data!")
71
+
72
+ col1, col2 = st.columns(2)
73
+ with col1:
74
+ uploaded_files = st.file_uploader(
75
+ "Drag and Drop .csv or .xls(x) file(s)",
76
+ accept_multiple_files=True,
77
+ type=["xlsx", "xls", "csv"],
78
+ key="uploaded_reps"
79
+ )
80
+ with col2:
81
+ if st.button("Load test data (replicates)", key="loadTest"):
82
+ self._load_test_files("reps")
83
+ if st.button("Reset data", key="reset_preprocessed"):
84
+ self._delete_data()
85
+
86
+ if uploaded_files:
87
+ self._read_preprocessed(uploaded_files)
88
+
89
+ if "input_rep" in self.sss and not self.sss.input_rep.empty:
90
+ self._handle_database_upload()
91
+ self.sss.input_rep = self._modify_uploaded_df([self.sss.input_rep])[0]
92
+ self._display_metrics(self.sss.input_rep)
93
+
94
+ def _render_intensities_uploader(self) -> None:
95
+ """Render the UI for uploading raw intensity data."""
96
+ st.subheader("Raw intensity data")
97
+ st.markdown("Drag & Drop sample and reference gas m/z$_{44-49}$ data!")
98
+
99
+ col1, col2 = st.columns(2)
100
+ with col1:
101
+ uploaded_files = st.file_uploader(
102
+ "Drag and Drop .csv, .xlsx or .did file(s)",
103
+ accept_multiple_files=True,
104
+ type=["xlsx", "xls", "csv", "did"],
105
+ key="uploaded_cycles"
106
+ )
107
+ with col2:
108
+ if st.button("Load test data (intensities)", key="loadTestRaw"):
109
+ self._load_test_files("raw")
110
+ if st.button("Reset raw data", key="reset_raw"):
111
+ self._delete_data()
112
+
113
+ if uploaded_files:
114
+ self._read_intensities(uploaded_files)
115
+
116
+ if "input_intensities" in self.sss and not self.sss.input_intensities.empty:
117
+ self._display_metrics(self.sss.input_intensities, is_intensity_data=True)
118
+
119
+ def _handle_database_upload(self) -> None:
120
+ """Handle the logic for uploading data to the database."""
121
+ if st.button("Upload Data to Database"):
122
+ with self.db_manager.get_connection() as conn:
123
+ existing = self._check_existing_timestamps(self.sss.input_rep, conn)
124
+ self.sss.show_overwrite_button = True
125
+ self.sss.existing_timestamps = existing
126
+ if not existing:
127
+ session = self.sss.input_rep['Session'].iloc[0]
128
+ rows = self.db_manager.upsert_dataframe_with_conflict_resolution(self.sss.input_rep, session)
129
+ st.success(f"Data uploaded successfully. {rows} rows affected.")
130
+ else:
131
+ st.warning(f"Found {len(existing)} existing timestamps in the database.")
132
+
133
+ if self.sss.get('show_overwrite_button', False):
134
+ if st.button("Confirm Overwrite"):
135
+ with self.db_manager.get_connection() as conn:
136
+ session = self.sss.input_rep['Session'].iloc[0]
137
+ cursor = conn.cursor()
138
+ cursor.executemany(
139
+ "DELETE FROM replicates WHERE Timetag = ?",
140
+ [(str(ts),) for ts in self.sss.existing_timestamps]
141
+ )
142
+ rows = self.db_manager.upsert_dataframe_with_conflict_resolution(self.sss.input_rep, session)
143
+ conn.commit()
144
+ st.success(f"Data overwritten successfully. {rows} rows affected.")
145
+ self.sss.show_overwrite_button = False
146
+ self.sss.existing_timestamps = []
147
+
148
+ def _database_tab(self) -> None:
149
+ """Render the database access tab."""
150
+ st.header("Access Database")
151
+ st.write("Use the filters below to query the database.")
152
+
153
+ sample_db = self._load_sample_database()
154
+
155
+ all_df = self.db_manager.get_dataframe()
156
+ all_sessions = sorted(all_df['Session'].unique().tolist()) if not all_df.empty else []
157
+ all_samples = sorted(all_df['Sample'].unique().tolist()) if not all_df.empty else []
158
+
159
+ col1, col2 = st.columns(2)
160
+ with col1:
161
+ selected_sessions = st.multiselect("Select Session(s)", all_sessions, default=all_sessions)
162
+ selected_samples = st.multiselect("Select Sample(s)", all_samples)
163
+
164
+ selected_project = []
165
+ selected_type = []
166
+ selected_mineralogy = []
167
+ selected_publication = []
168
+
169
+ if sample_db is not None:
170
+ selected_project, selected_type, selected_mineralogy, selected_publication = self._render_sample_db_filters(sample_db, col1, col2)
171
+
172
+ load_entire_sessions = st.checkbox("Load entire sessions when filtering", value=True)
173
+
174
+ if st.button("Apply Filters"):
175
+ self._apply_database_filters(
176
+ selected_sessions, selected_samples, sample_db, load_entire_sessions,
177
+ selected_project, selected_type, selected_mineralogy, selected_publication
178
+ )
179
+
180
+ def _load_sample_database(self) -> Optional[pd.DataFrame]:
181
+ """Load the sample database from Excel."""
182
+ sample_db_path = "static/SampleDatabase.xlsx"
183
+ try:
184
+ return pd.read_excel(sample_db_path, engine="openpyxl")
185
+ except FileNotFoundError:
186
+ st.warning(f"SampleDatabase not found at {sample_db_path}.")
187
+ st.page_link("pages/97_Database_Management.py", label="→ Database Management", icon="🔗")
188
+ return None
189
+
190
+ def _render_sample_db_filters(self, sample_db: pd.DataFrame, col1, col2) -> tuple:
191
+ """Render filters based on the sample database."""
192
+ with self.db_manager.get_connection() as conn:
193
+ replicates_df = pd.read_sql_query("SELECT * FROM replicates", conn)
194
+
195
+ sub_sample_db = sample_db[sample_db['Sample'].isin(replicates_df['Sample'])]
196
+
197
+ def get_split_values(column: str) -> List[str]:
198
+ values = sub_sample_db[column].dropna().astype(str)
199
+ split_values = [v.strip() for value in values for v in value.split(',')]
200
+ return sorted(set(split_values))
201
+
202
+ with col1:
203
+ selected_project = st.multiselect("Select Project(s)", get_split_values("Project"))
204
+ with col2:
205
+ selected_type = st.multiselect("Select Type(s)", get_split_values("Type"))
206
+ selected_mineralogy = st.multiselect("Select Mineralogy", get_split_values("Mineralogy"))
207
+ selected_publication = st.multiselect("Select Publication", get_split_values("Publication"))
208
+
209
+ return selected_project, selected_type, selected_mineralogy, selected_publication
210
+
211
+ def _apply_database_filters(
212
+ self, selected_sessions: List[str], selected_samples: List[str],
213
+ sample_db: Optional[pd.DataFrame], load_entire_sessions: bool,
214
+ selected_project: List[str], selected_type: List[str],
215
+ selected_mineralogy: List[str], selected_publication: List[str]
216
+ ) -> None:
217
+ """Apply filters and display data from the database."""
218
+ query = "SELECT * FROM replicates WHERE 1=1"
219
+ params = []
220
+ if selected_sessions:
221
+ query += f" AND Session IN ({','.join(['?'] * len(selected_sessions))})"
222
+ params.extend(selected_sessions)
223
+ if selected_samples:
224
+ query += f" AND Sample IN ({','.join(['?'] * len(selected_samples))})"
225
+ params.extend(selected_samples)
226
+
227
+ with self.db_manager.get_connection() as conn:
228
+ df = pd.read_sql_query(query, conn, params=params)
229
+
230
+ if sample_db is not None:
231
+ df = self._filter_by_sample_db(df, sample_db, selected_project, selected_type, selected_mineralogy, selected_publication)
232
+
233
+ if load_entire_sessions and not df.empty:
234
+ sessions_to_load = df['Session'].unique()
235
+ query = f"SELECT * FROM replicates WHERE Session IN ({','.join(['?'] * len(sessions_to_load))})"
236
+ with self.db_manager.get_connection() as conn:
237
+ df = pd.read_sql_query(query, conn, params=sessions_to_load.tolist())
238
+
239
+ self.sss.input_rep = self._modify_uploaded_df([df])[0]
240
+
241
+ if not df.empty:
242
+ self._display_filtered_data(df, sample_db)
243
+ else:
244
+ st.info("No entries found with your current filter settings.")
245
+
246
+ def _filter_by_sample_db(self, df: pd.DataFrame, sample_db: pd.DataFrame,
247
+ selected_project: List[str], selected_type: List[str],
248
+ selected_mineralogy: List[str], selected_publication: List[str]) -> pd.DataFrame:
249
+ """Filter DataFrame based on selections from the sample database."""
250
+ filtered_sample_db = sample_db.copy()
251
+
252
+ def filter_by_keywords(df_to_filter, column, keywords):
253
+ if not keywords:
254
+ return df_to_filter
255
+ pattern = '|'.join([re.escape(k.strip().lower()) for k in keywords])
256
+ return df_to_filter[df_to_filter[column].fillna('').str.lower().str.contains(pattern, regex=True)]
257
+
258
+ for field, selected in [('Project', selected_project), ('Type', selected_type),
259
+ ('Mineralogy', selected_mineralogy), ('Publication', selected_publication)]:
260
+ if selected:
261
+ filtered_sample_db = filter_by_keywords(filtered_sample_db, field, selected)
262
+
263
+ filtered_samples = filtered_sample_db['Sample'].unique()
264
+ return df[df['Sample'].isin(filtered_samples)]
265
+
266
+ @staticmethod
267
+ def _read_file(file: Any) -> pd.DataFrame:
268
+ """Read a file and return a DataFrame."""
269
+ filename = file.name
270
+ if filename.endswith((".xlsx", ".xls")):
271
+ return pd.read_excel(file, engine="openpyxl")
272
+ elif filename.endswith((".csv", ".txt")):
273
+ stringio = io.StringIO(file.getvalue().decode("utf-8"))
274
+ STR = stringio.read()
275
+ if '\t' in STR:
276
+ SEP = '\t'
277
+ elif ';' in STR:
278
+ SEP = ';'
279
+ elif ',' in STR:
280
+ SEP = ','
281
+ else:
282
+
283
+ SEP = None
284
+ st.info('Could not recognize column separator, please use `\\t` `;` or `,` !')
285
+ df=pd.read_csv(file, sep=SEP)
286
+ return df
287
+ raise ValueError(f"Unsupported file type: {filename}")
288
+
289
+ def _delete_data(self) -> None:
290
+ """Clear uploaded data from session state."""
291
+ for key in ['input_intensities', 'raw_files', 'input_rep','raw_data', 'scaling_factors', 'standards', 'correction_output_full_dataset','_06_filtered_reps','_06_filtered_summary','bg_success','correction_output_summary']:
292
+ if key in self.sss:
293
+ del self.sss[key]
294
+
295
+ def _modify_uploaded_df(self, dfs: List[pd.DataFrame]) -> List[pd.DataFrame]:
296
+ """Clean and modify uploaded DataFrames."""
297
+ for df in dfs:
298
+ df.drop_duplicates(inplace=True)
299
+ if "Sample" not in df.columns:
300
+ df["Sample"] = ""
301
+ df["Sample"] = df["Sample"].astype(str)
302
+
303
+ # Sanitize sample names
304
+ for char in ("+", ":", "(", ")", "&"):
305
+ if df["Sample"].str.contains(char, regex=False).any():
306
+ st.info(f"Sample names must not contain `{char}` --> removed!")
307
+ df["Sample"] = df["Sample"].str.replace(char, "", regex=False)
308
+
309
+ # Standardize sample names
310
+ rename_dict = {
311
+ **{f"ETH {i}": f"ETH-{i}" for i in range(1, 5)},
312
+ **{f"ETH{i}": f"ETH-{i}" for i in range(1, 5)},
313
+ "25G": "25C", "EG": "25C", "HG": "1000C",
314
+ "Heated": "1000C", "GU-1": "GU1"
315
+ }
316
+ df["Sample"].replace(rename_dict, inplace=True)
317
+
318
+ for key in ('Session', 'Sample'):
319
+ if key in df.columns:
320
+ df[key] = df[key].astype('string')
321
+
322
+ if "Timetag" not in df.columns:
323
+ datetime_columns = [key for key in df.columns if key.lower() in ("date", "time", "datetime")]
324
+ if datetime_columns:
325
+ df.rename(columns={datetime_columns[0]: "Timetag"}, inplace=True)
326
+ else:
327
+ st.error("No Timetag column provided! An artificial Timetag is created.")
328
+ d0 = pd.to_datetime('1900-01-01 00:00:00')
329
+ df['Timetag'] = [d0 + pd.Timedelta(days=i) for i in range(len(df))]
330
+
331
+ for col in ("Outlier", "Project", "Type"):
332
+ if col not in df.columns:
333
+ df[col] = None
334
+ return dfs
335
+
336
+ def _load_test_files(self, mode: str) -> None:
337
+ """Load example data files."""
338
+ self.sss.raw_files = mode != "reps"
339
+ path = (
340
+ "static/exampleReplicates_anonymized.xlsx" if mode == "reps"
341
+ else "static/exampleIntensities_anonymized.xlsx"
342
+ )
343
+ df = pd.read_excel(path, engine="openpyxl")
344
+ input_data = self._modify_uploaded_df([df])[0]
345
+
346
+ if mode == "reps":
347
+ self.sss.input_rep = input_data
348
+ else:
349
+ self.sss.input_intensities = input_data
350
+
351
+ def _read_intensities(self, uploaded_files: List[Any]) -> None:
352
+ """Read and process raw intensity data files."""
353
+ self.sss.raw_files = True
354
+ df_list = []
355
+
356
+ with st.spinner(text="Uploading..."):
357
+ for idx, file in enumerate(uploaded_files):
358
+ new_df = self._read_file(file)
359
+ for col in ["Session", "Type", "Project"]:
360
+ if col not in new_df.columns:
361
+ new_df[col] = idx
362
+ df_list.append(new_df)
363
+
364
+ df = pd.concat(df_list, ignore_index=True)
365
+
366
+ if "UID" in df and not df["UID"].is_unique:
367
+ st.error("Non-unique UIDs found! Please provide unique identifiers.")
368
+ st.stop()
369
+
370
+ columns_to_keep = [
371
+ "UID", "Sample", "Replicate", "Session", "Timetag",
372
+ "raw_s44", "raw_s45", "raw_s46", "raw_s47", "raw_s48", "raw_s49",
373
+ "raw_r44", "raw_r45", "raw_r46", "raw_r47", "raw_r48", "raw_r49"
374
+ ]
375
+
376
+ for mz in (47, 48):
377
+ if f"raw_s{mz}.5" in df.columns:
378
+ columns_to_keep.extend([f"raw_s{mz}.5", f"raw_r{mz}.5"])
379
+ else:
380
+ self.sss[f"half-mass-cup{mz}"] = False
381
+
382
+ df = self._modify_uploaded_df([df])[0]
383
+ existing_cols = [c for c in columns_to_keep if c in df.columns]
384
+ if len(existing_cols) < 5:
385
+ st.error("Uploaded intensity file(s) are missing required columns. Please provide raw_s44-49 and raw_r44-49 with Session, Sample, Replicate, Timetag.")
386
+ st.stop()
387
+ self.sss.input_intensities = df.loc[:, existing_cols]
388
+
389
+ def _read_preprocessed(self, uploaded_files: List[Any]) -> None:
390
+ """Read and process pre-processed replicate files."""
391
+ self.sss.raw_files = False
392
+ df_list = [self._read_file(f) for f in uploaded_files]
393
+ df = pd.concat(df_list, ignore_index=True)
394
+
395
+ df = self._modify_uploaded_df([df])[0]
396
+
397
+ if "UID" in df and not df["UID"].is_unique:
398
+ st.error("Non-unique UIDs found! Please provide unique identifiers.")
399
+ st.stop()
400
+
401
+ required = ["UID", "d45", "d46", "d47", "d48", "d49", "Sample", "Session"]
402
+ missing = [k for k in required if k not in df.columns]
403
+ if missing:
404
+ st.error(f"Please provide all required columns: {', '.join(missing)}")
405
+ st.stop()
406
+
407
+ self.sss.input_rep = df
408
+ st.info("Data loaded. Click 'Upload Data to Database' to save it.")
409
+
410
+ def _display_metrics(self, df: pd.DataFrame, is_intensity_data: bool = False) -> None:
411
+ """Display metrics and charts for the loaded data."""
412
+ st.dataframe(self._calculate_metrics(df, is_intensity_data), width="stretch")
413
+
414
+ time_col = 'Timetag' if "Timetag" in df.columns else 'datetime'
415
+ if is_intensity_data and "Replicate" in df.columns:
416
+ dist_samples = {
417
+ "Sample": df.groupby("Replicate")["Sample"].first(),
418
+ "Datetime": df.groupby("Replicate")[time_col].first(),
419
+ }
420
+ else:
421
+ dist_samples = {
422
+ "Sample": df["Sample"],
423
+ "Datetime": df[time_col],
424
+ }
425
+ st.scatter_chart(pd.DataFrame(dist_samples), y="Sample", x="Datetime", #width="stretch"
426
+ )
427
+
428
+ def _calculate_metrics(self, df: pd.DataFrame, is_intensity_data: bool) -> pd.DataFrame:
429
+ """Calculate metrics from the DataFrame."""
430
+ metrics = []
431
+ time_col = 'Timetag' if "Timetag" in df.columns else 'datetime'
432
+ for session, sdf in df.groupby("Session"):
433
+ sdf["Sample"] = sdf["Sample"].astype(str)
434
+ session_metrics = {
435
+ "Session": session,
436
+ "Duration": f"{sdf[time_col].min()} - {sdf[time_col].max()}",
437
+ "Unique Samples": sdf["Sample"].nunique(),
438
+ "Replicates": sdf["Replicate"].nunique() if is_intensity_data else len(sdf),
439
+ "Samples": sorted(sdf["Sample"].unique()),
440
+ }
441
+ if is_intensity_data:
442
+ session_metrics["Cycles"] = len(sdf)
443
+ metrics.append(session_metrics)
444
+ return pd.DataFrame(metrics)
445
+
446
+ def _check_existing_timestamps(self, df: pd.DataFrame, conn: Any) -> List[Any]:
447
+ """Check for existing timestamps in the database."""
448
+ cursor = conn.cursor()
449
+ existing = []
450
+ for timestamp in df['Timetag']:
451
+ cursor.execute("SELECT COUNT(*) FROM replicates WHERE Timetag = ?", (str(timestamp),))
452
+ if cursor.fetchone()[0] > 0:
453
+ existing.append(timestamp)
454
+ return existing
455
+
456
+ def _display_filtered_data(self, df: pd.DataFrame, sample_db: Optional[pd.DataFrame]) -> None:
457
+ """Display metrics and data for filtered database results."""
458
+ st.subheader("Data Metrics")
459
+ col1, col2 = st.columns(2)
460
+ with col1:
461
+ st.metric("Total Replicates", len(df))
462
+ st.metric("Unique Samples", df['Sample'].nunique(), help=' | '.join(sorted(df['Sample'].unique())))
463
+ st.metric("Unique Sessions", df['Session'].nunique(), help=' | '.join(sorted(df['Session'].unique())))
464
+ with col2:
465
+ tt_series = pd.to_datetime(df['Timetag'], errors='coerce')
466
+ min_dt = tt_series.min()
467
+ max_dt = tt_series.max()
468
+ if pd.isna(min_dt) or pd.isna(max_dt):
469
+ date_range = "—"
470
+ else:
471
+ date_range = f"{min_dt.strftime('%Y-%m-%d')} to {max_dt.strftime('%Y-%m-%d')}"
472
+ st.metric("Date Range", date_range)
473
+ if sample_db is not None:
474
+ self._display_sample_db_metrics(df, sample_db)
475
+
476
+ st.markdown(self._create_excel_download_link(df, "filtered_data.xlsx"), unsafe_allow_html=True)
477
+
478
+ st.subheader("Sample Distribution")
479
+ sample_counts = df['Sample'].value_counts()
480
+ fig = px.bar(x=sample_counts.index, y=sample_counts.values, labels={'x': 'Sample', 'y': 'Count'})
481
+ st.plotly_chart(fig)
482
+
483
+ st.subheader("Selected Data")
484
+ st.dataframe(df)
485
+
486
+ def _display_sample_db_metrics(self, df: pd.DataFrame, sample_db: pd.DataFrame) -> None:
487
+ """Display metrics derived from the sample database."""
488
+ filtered_samples = df['Sample'].unique()
489
+ sub_sample_db = sample_db[sample_db['Sample'].isin(filtered_samples)]
490
+
491
+ def count_unique_split(column: str) -> int:
492
+ return sub_sample_db[column].str.split(',').explode().str.strip().nunique()
493
+
494
+ st.metric("Total Projects", count_unique_split('Project'))
495
+ st.metric("Total Types", count_unique_split('Type'))
496
+
497
+ def _create_excel_download_link(self, df: pd.DataFrame, filename: str) -> str:
498
+ """Generate a download link for an Excel file."""
499
+ output = io.BytesIO()
500
+ with pd.ExcelWriter(output, engine='openpyxl') as writer:
501
+ df.to_excel(writer, index=False, sheet_name='Sheet1')
502
+ b64 = base64.b64encode(output.getvalue()).decode()
503
+ return f'<a href="data:application/vnd.openxmlformats-officedocument.spreadsheetml.sheet;base64,{b64}" download="{filename}">Download Excel File</a>'
504
+
505
+
506
+ if __name__ == "__main__":
507
+ page = DataIOPage()
508
+ page.run()