dimwit 0.2.7__tar.gz → 0.2.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: dimwit
3
- Version: 0.2.7
3
+ Version: 0.2.9
4
4
  Summary: A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics.
5
5
  Home-page: https://github.com/danielsoutar/dimwit
6
6
  Author: Daniel Soutar
@@ -13,6 +13,8 @@ Classifier: Programming Language :: Python :: 3.12
13
13
  Requires-Dist: matplotlib (>=3.8.3,<4.0.0)
14
14
  Requires-Dist: pandas (>=2.1.4,<3.0.0)
15
15
  Requires-Dist: requests (>=2.31.0,<3.0.0)
16
+ Requires-Dist: scipy (>=1.12.0,<2.0.0)
17
+ Requires-Dist: seaborn (>=0.13.2,<0.14.0)
16
18
  Project-URL: Repository, https://github.com/danielsoutar/dimwit
17
19
  Description-Content-Type: text/markdown
18
20
 
@@ -0,0 +1,6 @@
1
+ from dimwit.main import *
2
+ import dimwit.airflow
3
+ import dimwit.legacy
4
+ import dimwit.pomodoro
5
+ import dimwit.weight
6
+ import dimwit.air_pollution as ap
@@ -0,0 +1,277 @@
1
+ from dimwit.main import get_moving_average_trend, populate_with_events
2
+
3
+ import datetime as dat
4
+ import matplotlib.pyplot as plt
5
+ import numpy as np
6
+ from scipy.stats import norm
7
+ import seaborn as sb
8
+
9
+
10
+ def beginning_of_data():
11
+ uk_with_dst = dat.timezone(dat.timedelta(seconds=3600))
12
+ return dat.datetime(2023, 8, 21, 0, 0, tzinfo=uk_with_dst)
13
+
14
+
15
+ def get_events():
16
+ uk_with_dst = dat.timezone(dat.timedelta(seconds=3600))
17
+ dates = [
18
+ dat.datetime(2023, 9, 12, 13, 40, tzinfo=uk_with_dst),
19
+ dat.datetime(2024, 2, 4, 0, 0, tzinfo=dat.timezone.utc),
20
+ dat.datetime(2024, 2, 8, 0, 0, tzinfo=dat.timezone.utc),
21
+ ]
22
+ descs = [
23
+ "Started using inhaler",
24
+ "Only using inhaler as needed",
25
+ "Using inhaler unless healthy",
26
+ ]
27
+ return [(date, "gray", "--", desc) for date, desc in zip(dates, descs)]
28
+
29
+
30
+ def data_over_period(df, from_date, moving_average_window_size, events, title):
31
+ latest_df = df.loc[from_date:]
32
+ fig, ax = plt.subplots(1, 1, figsize=(12, 6))
33
+
34
+ x = list(latest_df.index)
35
+
36
+ ax.scatter(x, latest_df["Recording 1"], label="Point 1", alpha=0.3)
37
+ ax.scatter(x, latest_df["Recording 2"], label="Point 2", alpha=0.3)
38
+ ax.scatter(x, latest_df["Recording 3"], label="Point 3", alpha=0.3)
39
+
40
+ # Plot a line for the maximum of the points
41
+ ax.plot(x, latest_df["Max Point"], label="Max", color="red")
42
+
43
+ window_size = moving_average_window_size
44
+
45
+ # Calculate the rolling average over the max values array
46
+ # TODO: Figure out whether to use actual data rather than repeating first
47
+ # point (k - 1) / 2 times, where possible.
48
+ rolling_average = get_moving_average_trend(
49
+ np.array(latest_df["Max Point"]), window_size
50
+ )
51
+ ax.plot(
52
+ x,
53
+ rolling_average,
54
+ label=f"Rolling Avg ({window_size})",
55
+ color="green",
56
+ )
57
+
58
+ ax = populate_with_events(ax, events, x[0])
59
+
60
+ plt.legend()
61
+ plt.grid()
62
+ # Add a title to the entire figure
63
+ fig.suptitle(title, fontsize=16)
64
+
65
+ return fig, ax
66
+
67
+
68
+ def differences_over_period(
69
+ df,
70
+ from_date,
71
+ moving_average_window_size,
72
+ events,
73
+ title,
74
+ ):
75
+ latest_df = df.loc[from_date:]
76
+ fig, ax = plt.subplots(1, 1, figsize=(12, 6))
77
+
78
+ max_vals = df[["Recording 1", "Recording 2", "Recording 3"]].max(axis=1)
79
+ min_vals = df[["Recording 1", "Recording 2", "Recording 3"]].min(axis=1)
80
+ df["Delta"] = max_vals - min_vals
81
+
82
+ x = list(latest_df.index)
83
+
84
+ ax.scatter(x, df["Delta"], label="Max-Min Difference", alpha=0.3)
85
+
86
+ window_size = moving_average_window_size
87
+
88
+ # Calculate the rolling average over the max values array
89
+ # TODO: Figure out whether to use actual data rather than repeating first
90
+ # point (k - 1) / 2 times, where possible.
91
+ rolling_average = get_moving_average_trend(
92
+ np.array(df["Delta"]),
93
+ window_size,
94
+ )
95
+ ax.plot(
96
+ x,
97
+ rolling_average,
98
+ label=f"Rolling Avg ({window_size})",
99
+ color="green",
100
+ )
101
+
102
+ ax = populate_with_events(ax, events, x[0])
103
+
104
+ plt.legend()
105
+ plt.grid()
106
+ # Add a title to the entire figure
107
+ fig.suptitle(title, fontsize=16)
108
+
109
+ return fig, ax
110
+
111
+
112
+ def get_pretty_image(
113
+ image_arr,
114
+ title,
115
+ colour_bar_title,
116
+ palette="viridis",
117
+ foreground_colour="white",
118
+ background_colour="black",
119
+ ):
120
+ # 'flare_r' is a neat fire-y palette, as an alternative.
121
+ cmap = sb.color_palette(palette, as_cmap=True)
122
+
123
+ fig, ax = plt.subplots(figsize=(24, 4))
124
+ image = ax.imshow(
125
+ image_arr,
126
+ interpolation="nearest",
127
+ aspect="auto",
128
+ cmap=cmap,
129
+ )
130
+ colour_bar = plt.colorbar(image)
131
+
132
+ # set figure facecolor
133
+ ax.patch.set_facecolor(background_colour)
134
+
135
+ # set tick and ticklabel color
136
+ image.axes.get_xaxis().set_visible(False)
137
+ image.axes.get_yaxis().set_visible(False)
138
+
139
+ # set imshow outline
140
+ for spine in image.axes.spines.values():
141
+ spine.set_edgecolor(background_colour)
142
+
143
+ # set colorbar label plus label color
144
+ colour_bar.set_label(colour_bar_title, color=foreground_colour)
145
+
146
+ # set colorbar tick color
147
+ colour_bar.ax.yaxis.set_tick_params(color=foreground_colour)
148
+
149
+ # set colorbar edgecolor
150
+ colour_bar.outline.set_edgecolor(foreground_colour)
151
+
152
+ # set colorbar ticklabels
153
+ plt.setp(
154
+ plt.getp(colour_bar.ax.axes, "yticklabels"),
155
+ color=foreground_colour,
156
+ )
157
+
158
+ _ = ax.set_title(title, color=foreground_colour)
159
+ fig.patch.set_facecolor(background_colour)
160
+ return fig, ax
161
+
162
+
163
+ def generate_month_year_tuples(start=None, end=None):
164
+ if start is None:
165
+ start = (2023, 8)
166
+
167
+ if end is None:
168
+ end_date = dat.datetime.now()
169
+ end = (end_date.year, end_date.month)
170
+ else:
171
+ assert end[0] >= start[0]
172
+ if end[0] == start[0]:
173
+ assert end[1] > start[1]
174
+
175
+ month_names = {
176
+ 1: "Jan",
177
+ 2: "Feb",
178
+ 3: "Mar",
179
+ 4: "Apr",
180
+ 5: "May",
181
+ 6: "Jun",
182
+ 7: "Jul",
183
+ 8: "Aug",
184
+ 9: "Sep",
185
+ 10: "Oct",
186
+ 11: "Nov",
187
+ 12: "Dec",
188
+ }
189
+
190
+ tuples = []
191
+ current = start
192
+
193
+ while current != end:
194
+ current_year, current_month = current
195
+ tuples.append((current_year, current_month))
196
+ increment_year = current_month == 12
197
+ if increment_year:
198
+ next_year, next_month = current_year + 1, 1
199
+ else:
200
+ next_year, next_month = current_year, current_month + 1
201
+
202
+ current = (next_year, next_month)
203
+
204
+ tuples.append(end)
205
+
206
+ res = list(map(lambda t: (*t, f"{month_names[t[1]]} {str(t[0])}"), tuples))
207
+
208
+ return res
209
+
210
+
211
+ def get_hist_data_for_month(df, year, month, use_maxes=False):
212
+ utc = dat.timezone.utc
213
+
214
+ dt = dat.datetime(year, month, 1, 0, 0, tzinfo=utc)
215
+ if month == 12:
216
+ end_dt = dat.datetime(year + 1, 1, 1, 0, 0, tzinfo=utc)
217
+ else:
218
+ end_dt = dat.datetime(year, month + 1, 1, 0, 0, tzinfo=utc)
219
+
220
+ month_data = df[["Recording 1", "Recording 2", "Recording 3"]][dt:end_dt]
221
+ month_data = np.array(month_data).reshape((-1, 3))
222
+
223
+ if use_maxes:
224
+ month_maxes = np.max(month_data, axis=1).reshape((-1, 1))
225
+ return month_maxes
226
+
227
+ return month_data
228
+
229
+
230
+ def generate_overlaid_monthly_pdfs(df, xmin=500, xmax=800, num_samples=100):
231
+ fig, ax = plt.subplots(1, 1)
232
+
233
+ for year, month, name in generate_month_year_tuples():
234
+ all_samples_for_month = get_hist_data_for_month(df, year, month)
235
+ maxes_for_month = get_hist_data_for_month(
236
+ df,
237
+ year,
238
+ month,
239
+ use_maxes=True,
240
+ )
241
+ mean, std_dev = norm.fit(all_samples_for_month)
242
+ max_mean, _ = norm.fit(maxes_for_month)
243
+
244
+ pdf_x = np.linspace(xmin, xmax, num_samples)
245
+ pdf_y = norm.pdf(pdf_x, mean, std_dev)
246
+
247
+ month_label = f"{name} ({round(mean)}/{round(max_mean)})"
248
+ ax.plot(pdf_x, pdf_y, linewidth=2, label=month_label)
249
+
250
+ ax.legend()
251
+ ax.set_xlabel("Airflow (L/Min)")
252
+ ax.set_ylabel("Count")
253
+ ax.set_title("PDF of data, per month", size=16)
254
+ ax.set_xlim(xmin, xmax)
255
+ return fig, ax
256
+
257
+
258
+ def create_monthly_airflow_histograms(df, month_year_tuples):
259
+ # Create a figure with subplots
260
+ rows = len(month_year_tuples)
261
+ fig, axes = plt.subplots(rows, 1, figsize=(5, 10), sharex=True)
262
+
263
+ # Plot data for each week on separate axes
264
+ for i, ax in enumerate(axes):
265
+ year, month, name = month_year_tuples[i]
266
+ month_data = get_hist_data_for_month(df, year, month)
267
+ flattened_month_data = month_data.ravel()
268
+ name = name + f" ({len(flattened_month_data)} samples)"
269
+ ax.hist(flattened_month_data, label=f"{name}")
270
+ ax.set_title(f"{name}")
271
+ ax.set_ylim(0, 50)
272
+
273
+ # Add a title to the entire figure
274
+ fig.suptitle("Distribution of data, per month", fontsize=16)
275
+ fig.supxlabel("Airflow (L/Min)")
276
+ fig.supylabel("Count")
277
+ return fig, axes
@@ -356,6 +356,9 @@ def get_aggregated_event_counts(ts, events, period):
356
356
  return grouped_df
357
357
 
358
358
 
359
+ # TODO: Add 'get all events where equal to' function. Do not want a dense df.
360
+
361
+
359
362
  def populate_with_events(ax, events, from_date):
360
363
  for event in events:
361
364
  event_date, event_colour, event_style, event_label = event
@@ -0,0 +1,168 @@
1
+ import pandas as pd
2
+ from typing_extensions import NamedTuple
3
+
4
+
5
+ def is_business_day(date):
6
+ return bool(len(pd.bdate_range(date, date)))
7
+
8
+
9
+ def generate_cumulative_df(df, ref_policy, from_date=None, until_date=None):
10
+ from_date = df.index[0] if from_date is None else from_date
11
+ until_date = df.index[-1] if until_date is None else until_date
12
+ time_range_df = df.copy(deep=True).loc[from_date:until_date]
13
+
14
+ exclude_dates = []
15
+
16
+ resolution = ref_policy.pomodoro_length + ref_policy.break_length
17
+ current_time = pd.Timestamp.today().time().replace(second=0, microsecond=0)
18
+ start_of_business = current_time.replace(hour=9, minute=0)
19
+ close_of_business = current_time.replace(hour=17, minute=0)
20
+ ref_entries = get_reference_entries(
21
+ time_range_df.index,
22
+ resolution,
23
+ start_of_business,
24
+ close_of_business,
25
+ exclude_dates,
26
+ )
27
+
28
+ target_entries = get_target_entries(time_range_df.index, exclude_dates)
29
+
30
+ ref_pom_col = "Reference Pomodoro Lengths " + ref_policy.description
31
+ ref_break_col = "Reference Break Lengths " + ref_policy.description
32
+ target_pom_col = "Target Pomodoro Lengths (3x 45+15)"
33
+ target_break_col = "Target Break Lengths (3x 45+15)"
34
+
35
+ time_range_df[ref_pom_col] = 0.0
36
+ time_range_df[ref_break_col] = 0.0
37
+ time_range_df[target_pom_col] = 0.0
38
+ time_range_df[target_break_col] = 0.0
39
+
40
+ time_range_df.loc[ref_entries, ref_pom_col] = ref_policy.pomodoro_length
41
+ time_range_df.loc[ref_entries, ref_break_col] = ref_policy.break_length
42
+ time_range_df.loc[target_entries, target_pom_col] = 45.0
43
+ time_range_df.loc[target_entries, target_break_col] = 15.0
44
+
45
+ cumulative_df = time_range_df.cumsum()
46
+
47
+ return cumulative_df
48
+
49
+
50
+ def get_target_entries(index, exclude_dates):
51
+ valid_entries_for_target = []
52
+
53
+ set_times = set([9, 10, 11])
54
+
55
+ for i, timestamp in enumerate(index):
56
+ if timestamp in exclude_dates or not is_business_day(timestamp):
57
+ continue
58
+ ts: pd.Timestamp = timestamp
59
+ if ts.time().hour in set_times and ts.time().minute == 0:
60
+ valid_entries_for_target.append(ts)
61
+
62
+ return valid_entries_for_target
63
+
64
+
65
+ def get_reference_entries(
66
+ index,
67
+ resolution,
68
+ start_time,
69
+ end_time,
70
+ exclude_dates,
71
+ ):
72
+ valid_entries_for_reference = []
73
+
74
+ for i, timestamp in enumerate(index):
75
+ if timestamp in exclude_dates:
76
+ continue
77
+ if is_business_day(timestamp):
78
+ time = timestamp.time()
79
+ start_of_day, end_of_day = start_time, end_time
80
+ if time >= start_of_day and time < end_of_day and time.hour != 13:
81
+ # Way to fix this would be to check if the latest valid entry
82
+ # has the delta. Skip this step until len(valid_entries) > 0.
83
+ if len(valid_entries_for_reference) > 0:
84
+ delta = timestamp - valid_entries_for_reference[-1]
85
+ if delta < pd.Timedelta(minutes=resolution):
86
+ continue
87
+ valid_entries_for_reference.append(timestamp)
88
+
89
+ return valid_entries_for_reference
90
+
91
+
92
+ class ReferencePolicy(NamedTuple):
93
+ pomodoro_length: float
94
+ break_length: float
95
+ description: str
96
+
97
+
98
+ def get_kth_latest_monday(k=0):
99
+ today = pd.to_datetime("today", utc=True)
100
+ latest_monday = today - pd.Timedelta(days=today.weekday(), weeks=k)
101
+ latest_monday_start_of_day = latest_monday.replace(hour=8, minute=0)
102
+ return latest_monday_start_of_day
103
+
104
+
105
+ def get_kth_latest_sunday(k=0):
106
+ today = pd.to_datetime("today", utc=True)
107
+ today_k_weeks_ago = today - pd.Timedelta(weeks=k)
108
+ latest_sunday = today_k_weeks_ago + pd.Timedelta(
109
+ days=7 - today.weekday() - 1,
110
+ )
111
+ latest_sunday_end_of_day = latest_sunday.replace(hour=18, minute=0)
112
+ return latest_sunday_end_of_day
113
+
114
+
115
+ def generate_cumulative_df_for_kth_latest_week(df, k):
116
+ pomodoro_date_range = pd.date_range(
117
+ start=get_kth_latest_monday(k), freq="d", end=get_kth_latest_sunday(k)
118
+ )
119
+ start, end = pomodoro_date_range[0], pomodoro_date_range[-1]
120
+
121
+ # If end later than latest available entry in df (i.e. start of the week),
122
+ # then pad out df to generate zero entries.
123
+ # if end > df.index[-1]:
124
+ # num_entries = 0
125
+ # latest_val = df.index[-1]
126
+ # while end > latest_val:
127
+ # latest_val += df.index.freq
128
+ # num_entries += 1
129
+ # df = dimwit.pad_out_table(
130
+ # df, num_entries, df.index.freq, pad_before=False
131
+ # )
132
+
133
+ ref_policy = ReferencePolicy(45.0, 15.0, "(45+15)")
134
+ cum_df = generate_cumulative_df(df, ref_policy, start, end)
135
+
136
+ return cum_df, ref_policy, start, end
137
+
138
+
139
+ def plot_burn_up_plot_for_kth_latest_week(df, k):
140
+ result = generate_cumulative_df_for_kth_latest_week(df, k)
141
+ cum_df, ref_policy, start, end = result
142
+
143
+ ref_col = "Reference Pomodoro Lengths " + ref_policy.description
144
+ target_col = "Target Pomodoro Lengths (3x 45+15)"
145
+
146
+ ax = cum_df.plot(y=["pomodoro_lengths", ref_col, target_col])
147
+
148
+ total_pomodoro_length = cum_df["pomodoro_lengths"].iloc[-1]
149
+ total_ref_pomodoro_length = cum_df[ref_col].iloc[-1]
150
+ total_target_pomodoro_length = cum_df[target_col].iloc[-1]
151
+
152
+ ax.legend(
153
+ labels=[
154
+ f"Pomodoro Lengths (Mixed), total={total_pomodoro_length}",
155
+ target_col + f", total={total_target_pomodoro_length}",
156
+ ref_col + f", total={total_ref_pomodoro_length}",
157
+ ],
158
+ loc="upper right",
159
+ )
160
+ ax.set_ylabel("Cumulative time (minutes)")
161
+
162
+ start_str = start.to_pydatetime().strftime("%d/%m")
163
+ end_str = end.to_pydatetime().strftime("%d/%m")
164
+
165
+ ax.set_title(f"Work Pomodoros for week ({start_str}-{end_str})")
166
+ ax.set_ylim(0, 2100)
167
+
168
+ return ax
@@ -0,0 +1,65 @@
1
+ from dimwit import get_moving_average_trend, populate_with_events
2
+
3
+ from datetime import datetime, timedelta
4
+
5
+ import datetime as dat
6
+ import matplotlib.pyplot as plt
7
+ import numpy as np
8
+
9
+
10
+ def beginning_of_data():
11
+ dst = dat.timezone(dat.timedelta(seconds=3600))
12
+ return datetime(2023, 5, 20, 0, 0, tzinfo=dst)
13
+
14
+
15
+ def daterange(start_date, end_date, unit, step=1):
16
+ N = int((end_date - start_date) / timedelta(**{unit: 1}))
17
+ for n in range(0, N, step):
18
+ yield start_date + timedelta(**{unit: n})
19
+
20
+
21
+ def data_over_period(df, from_date, moving_average_window_size, events, title):
22
+ fig, ax = plt.subplots(1, 1, figsize=(12, 6))
23
+ current_target = 68.0
24
+
25
+ latest_df = df.loc[from_date:]
26
+ xs = latest_df.index
27
+ ys = np.array(latest_df["weight_kg"])
28
+
29
+ category_names = ["active-era-ref-scale", "external-scale", "historical"]
30
+ category_labels = ["Reference", "External", "Historical"]
31
+
32
+ for category, label in zip(category_names, category_labels):
33
+ category_df = latest_df.loc[latest_df["category"] == category]
34
+ category_xs = category_df.index
35
+ ax.scatter(category_xs, category_df["weight_kg"], label=label)
36
+
37
+ current_target_ys = [current_target] * len(xs)
38
+ ax.plot(
39
+ xs,
40
+ current_target_ys,
41
+ label=f"Current Target ({current_target}kg)",
42
+ )
43
+
44
+ window_size = moving_average_window_size
45
+
46
+ # Calculate the rolling average over the max values array
47
+ # TODO: Figure out whether to use actual data rather than repeating first
48
+ # point (k - 1) / 2 times, where possible.
49
+ rolling_average = get_moving_average_trend(ys, window_size)
50
+ ax.plot(
51
+ xs,
52
+ rolling_average,
53
+ label=f"Rolling Avg ({window_size})",
54
+ color="green",
55
+ )
56
+
57
+ ax = populate_with_events(ax, events, xs[0])
58
+
59
+ ax.set_ylabel("Weight (kg)")
60
+ plt.legend()
61
+ plt.grid()
62
+ # Add a title to the entire figure
63
+ fig.suptitle(title, fontsize=16)
64
+
65
+ return fig, ax
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "dimwit"
3
- version = "0.2.7"
3
+ version = "0.2.9"
4
4
  description = "A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics."
5
5
  authors = ["Daniel Soutar <danielsoutar144@gmail.com>"]
6
6
  readme = "README.md"
@@ -11,6 +11,8 @@ python = "^3.10"
11
11
  pandas = "^2.1.4"
12
12
  requests = "^2.31.0"
13
13
  matplotlib = "^3.8.3"
14
+ seaborn = "^0.13.2"
15
+ scipy = "^1.12.0"
14
16
 
15
17
 
16
18
  [build-system]
@@ -1,3 +0,0 @@
1
- from dimwit.main import *
2
- import dimwit.legacy as legacy
3
- import dimwit.air_pollution as ap
File without changes
File without changes
File without changes