dimwit 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dimwit-0.3.0 → dimwit-0.3.2}/PKG-INFO +1 -1
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/main.py +112 -113
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit.egg-info/PKG-INFO +1 -1
- {dimwit-0.3.0 → dimwit-0.3.2}/pyproject.toml +1 -1
- {dimwit-0.3.0 → dimwit-0.3.2}/README.md +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/__init__.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/air_pollution.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/airflow.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/legacy.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/pomodoro.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/sleep.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit/weight.py +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit.egg-info/SOURCES.txt +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit.egg-info/dependency_links.txt +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit.egg-info/requires.txt +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/dimwit.egg-info/top_level.txt +0 -0
- {dimwit-0.3.0 → dimwit-0.3.2}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dimwit
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics.
|
|
5
5
|
Author-email: Daniel Soutar <danielsoutar144@gmail.com>
|
|
6
6
|
Project-URL: repository, https://github.com/danielsoutar/dimwit
|
|
@@ -314,119 +314,6 @@ def group_data(
|
|
|
314
314
|
return grouped_ts, grouped_events
|
|
315
315
|
|
|
316
316
|
|
|
317
|
-
# This enables reading from a (chunked) database with good compression and
|
|
318
|
-
# substantially faster reads for older data, and only a small amount of syncing
|
|
319
|
-
# required with the slower database.
|
|
320
|
-
# This does assume the presence of a 'date' column in the un-chunked database,
|
|
321
|
-
# however. This enables the filtering on the slower database.
|
|
322
|
-
def get_notion_pages_from_db_sync_with_chunked(
|
|
323
|
-
db_id: str,
|
|
324
|
-
chunked_db_id: str,
|
|
325
|
-
headers: NotionHeaders,
|
|
326
|
-
extract_chunked_func: Callable,
|
|
327
|
-
decompression_map: dict,
|
|
328
|
-
compression_map: dict,
|
|
329
|
-
extract_func: Callable,
|
|
330
|
-
compress_func: Callable,
|
|
331
|
-
patch_chunked_page: Callable,
|
|
332
|
-
post_new_chunked_page: Callable,
|
|
333
|
-
):
|
|
334
|
-
# Read all the data from chunked_db in ascending order of id.
|
|
335
|
-
chunked_pages = get_notion_pages_from_db(
|
|
336
|
-
chunked_db_id,
|
|
337
|
-
headers,
|
|
338
|
-
sort_column="id",
|
|
339
|
-
sort_direction="ascending",
|
|
340
|
-
)
|
|
341
|
-
|
|
342
|
-
# Get the latest date from the chunked data.
|
|
343
|
-
latest_chunked_page = [chunked_pages[-1]] if chunked_pages else []
|
|
344
|
-
chunked_page_ts, chunked_page_events = get_all_entries_chunked(
|
|
345
|
-
latest_chunked_page, extract_chunked_func, decompression_map
|
|
346
|
-
)
|
|
347
|
-
latest_chunked_date: dat.datetime | None = (
|
|
348
|
-
chunked_page_ts[-1] if chunked_page_ts else None
|
|
349
|
-
)
|
|
350
|
-
chunked_page_id: str | None = chunked_pages[-1]["id"] if chunked_pages else None
|
|
351
|
-
print(f"Latest chunked timestamp: {latest_chunked_date}")
|
|
352
|
-
print(f"Latest chunked page ID: {chunked_page_id = }")
|
|
353
|
-
|
|
354
|
-
# Read all the data from slower_db s.t. date >= latest chunked date.
|
|
355
|
-
# Typicall this should be a small amount of data - only on an empty chunked
|
|
356
|
-
# database should this pay the full cost of loading/syncing all the data.
|
|
357
|
-
filter_by = None
|
|
358
|
-
if latest_chunked_date:
|
|
359
|
-
filter_by = {
|
|
360
|
-
"property": "date",
|
|
361
|
-
"date": {"on_or_after": latest_chunked_date.isoformat()},
|
|
362
|
-
}
|
|
363
|
-
unchunked_pages = get_notion_pages_from_db(
|
|
364
|
-
db_id,
|
|
365
|
-
headers,
|
|
366
|
-
sort_column="date",
|
|
367
|
-
sort_direction="ascending",
|
|
368
|
-
filter_by=filter_by,
|
|
369
|
-
)
|
|
370
|
-
unchunked_ts, unchunked_events = get_all_entries(
|
|
371
|
-
unchunked_pages,
|
|
372
|
-
extract_func,
|
|
373
|
-
)
|
|
374
|
-
|
|
375
|
-
if not latest_chunked_date:
|
|
376
|
-
# Useful debugging utility when creating a new chunked DB.
|
|
377
|
-
unchunked_event_freqs = {}
|
|
378
|
-
for event in unchunked_events:
|
|
379
|
-
if event not in unchunked_event_freqs:
|
|
380
|
-
unchunked_event_freqs[event] = 0
|
|
381
|
-
unchunked_event_freqs[event] += 1
|
|
382
|
-
print(f"{unchunked_event_freqs = }")
|
|
383
|
-
|
|
384
|
-
earliest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[0]
|
|
385
|
-
latest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[-1]
|
|
386
|
-
print(f"Earliest unchunked timestamp: {earliest_unchunked}")
|
|
387
|
-
print(f"Latest unchunked timestamp: {latest_unchunked}")
|
|
388
|
-
|
|
389
|
-
# De-duplicate the data and transform into standardised chunked pages (i.e.
|
|
390
|
-
# update the latest chunked page and add any new chunked pages).
|
|
391
|
-
if latest_chunked_date:
|
|
392
|
-
dedup_chunked_pairs = [
|
|
393
|
-
(t, e)
|
|
394
|
-
for t, e in zip(chunked_page_ts, chunked_page_events)
|
|
395
|
-
if t < latest_chunked_date
|
|
396
|
-
]
|
|
397
|
-
dedup_ts, dedup_events = zip(*dedup_chunked_pairs)
|
|
398
|
-
chunked_page_ts = list(dedup_ts)
|
|
399
|
-
chunked_page_events = list(dedup_events)
|
|
400
|
-
|
|
401
|
-
ts = chunked_page_ts + unchunked_ts
|
|
402
|
-
events = chunked_page_events + unchunked_events
|
|
403
|
-
# Group data for better compression.
|
|
404
|
-
grouped_ts, grouped_events = group_data(ts, events)
|
|
405
|
-
new_chunked_pages = compress_event_data(
|
|
406
|
-
grouped_ts,
|
|
407
|
-
grouped_events,
|
|
408
|
-
compression_map,
|
|
409
|
-
chunked_db_id,
|
|
410
|
-
)
|
|
411
|
-
|
|
412
|
-
# PATCH/POST the latest chunked page and any new chunked pages.
|
|
413
|
-
for idx, chunked_page in enumerate(new_chunked_pages):
|
|
414
|
-
if idx == 0 and chunked_page_id:
|
|
415
|
-
patch_chunked_page(
|
|
416
|
-
headers,
|
|
417
|
-
chunked_page_id,
|
|
418
|
-
chunked_page,
|
|
419
|
-
)
|
|
420
|
-
continue
|
|
421
|
-
post_new_chunked_page(
|
|
422
|
-
headers,
|
|
423
|
-
chunked_page,
|
|
424
|
-
)
|
|
425
|
-
|
|
426
|
-
# Return the chunked pages.
|
|
427
|
-
return chunked_pages[:-1] + new_chunked_pages
|
|
428
|
-
|
|
429
|
-
|
|
430
317
|
def patch_chunked_page(
|
|
431
318
|
headers: NotionHeaders,
|
|
432
319
|
page_id: str,
|
|
@@ -633,6 +520,118 @@ def get_all_entries_chunked(pages, add_data_entry, decompression_map):
|
|
|
633
520
|
return timestamps, data
|
|
634
521
|
|
|
635
522
|
|
|
523
|
+
# This enables reading from a (chunked) database with good compression and
|
|
524
|
+
# substantially faster reads for older data, and only a small amount of syncing
|
|
525
|
+
# required with the slower database.
|
|
526
|
+
# This does assume the presence of a 'date' column in the un-chunked database,
|
|
527
|
+
# however. This enables the filtering on the slower database.
|
|
528
|
+
def get_notion_pages_from_db_sync_with_chunked(
|
|
529
|
+
db_id: str,
|
|
530
|
+
chunked_db_id: str,
|
|
531
|
+
headers: NotionHeaders,
|
|
532
|
+
decompression_map: dict,
|
|
533
|
+
compression_map: dict,
|
|
534
|
+
extract_chunked_func: Callable = extract_chunked_categorical_entry,
|
|
535
|
+
extract_func: Callable = extract_categorical_entry,
|
|
536
|
+
patch_chunked_page_func: Callable = patch_chunked_page,
|
|
537
|
+
post_new_chunked_page_func: Callable = post_new_chunked_page,
|
|
538
|
+
):
|
|
539
|
+
# Read all the data from chunked_db in ascending order of id.
|
|
540
|
+
chunked_pages = get_notion_pages_from_db(
|
|
541
|
+
chunked_db_id,
|
|
542
|
+
headers,
|
|
543
|
+
sort_column="id",
|
|
544
|
+
sort_direction="ascending",
|
|
545
|
+
)
|
|
546
|
+
|
|
547
|
+
# Get the latest date from the chunked data.
|
|
548
|
+
latest_chunked_page = [chunked_pages[-1]] if chunked_pages else []
|
|
549
|
+
chunked_page_ts, chunked_page_events = get_all_entries_chunked(
|
|
550
|
+
latest_chunked_page, extract_chunked_func, decompression_map
|
|
551
|
+
)
|
|
552
|
+
latest_chunked_date: dat.datetime | None = (
|
|
553
|
+
chunked_page_ts[-1] if chunked_page_ts else None
|
|
554
|
+
)
|
|
555
|
+
chunked_page_id: str | None = chunked_pages[-1]["id"] if chunked_pages else None
|
|
556
|
+
print(f"Latest chunked timestamp: {latest_chunked_date}")
|
|
557
|
+
print(f"Latest chunked page ID: {chunked_page_id = }")
|
|
558
|
+
|
|
559
|
+
# Read all the data from slower_db s.t. date >= latest chunked date.
|
|
560
|
+
# Typicall this should be a small amount of data - only on an empty chunked
|
|
561
|
+
# database should this pay the full cost of loading/syncing all the data.
|
|
562
|
+
filter_by = None
|
|
563
|
+
if latest_chunked_date:
|
|
564
|
+
filter_by = {
|
|
565
|
+
"property": "date",
|
|
566
|
+
"date": {"on_or_after": latest_chunked_date.isoformat()},
|
|
567
|
+
}
|
|
568
|
+
unchunked_pages = get_notion_pages_from_db(
|
|
569
|
+
db_id,
|
|
570
|
+
headers,
|
|
571
|
+
sort_column="date",
|
|
572
|
+
sort_direction="ascending",
|
|
573
|
+
filter_by=filter_by,
|
|
574
|
+
)
|
|
575
|
+
unchunked_ts, unchunked_events = get_all_entries(
|
|
576
|
+
unchunked_pages,
|
|
577
|
+
extract_func,
|
|
578
|
+
)
|
|
579
|
+
|
|
580
|
+
if not latest_chunked_date:
|
|
581
|
+
# Useful debugging utility when creating a new chunked DB.
|
|
582
|
+
unchunked_event_freqs = {}
|
|
583
|
+
for event in unchunked_events:
|
|
584
|
+
if event not in unchunked_event_freqs:
|
|
585
|
+
unchunked_event_freqs[event] = 0
|
|
586
|
+
unchunked_event_freqs[event] += 1
|
|
587
|
+
print(f"{unchunked_event_freqs = }")
|
|
588
|
+
|
|
589
|
+
earliest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[0]
|
|
590
|
+
latest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[-1]
|
|
591
|
+
print(f"Earliest unchunked timestamp: {earliest_unchunked}")
|
|
592
|
+
print(f"Latest unchunked timestamp: {latest_unchunked}")
|
|
593
|
+
|
|
594
|
+
# De-duplicate the data and transform into standardised chunked pages (i.e.
|
|
595
|
+
# update the latest chunked page and add any new chunked pages).
|
|
596
|
+
if latest_chunked_date:
|
|
597
|
+
dedup_chunked_pairs = [
|
|
598
|
+
(t, e)
|
|
599
|
+
for t, e in zip(chunked_page_ts, chunked_page_events)
|
|
600
|
+
if t < latest_chunked_date
|
|
601
|
+
]
|
|
602
|
+
dedup_ts, dedup_events = zip(*dedup_chunked_pairs)
|
|
603
|
+
chunked_page_ts = list(dedup_ts)
|
|
604
|
+
chunked_page_events = list(dedup_events)
|
|
605
|
+
|
|
606
|
+
ts = chunked_page_ts + unchunked_ts
|
|
607
|
+
events = chunked_page_events + unchunked_events
|
|
608
|
+
# Group data for better compression.
|
|
609
|
+
grouped_ts, grouped_events = group_data(ts, events)
|
|
610
|
+
new_chunked_pages = compress_event_data(
|
|
611
|
+
grouped_ts,
|
|
612
|
+
grouped_events,
|
|
613
|
+
compression_map,
|
|
614
|
+
chunked_db_id,
|
|
615
|
+
)
|
|
616
|
+
|
|
617
|
+
# PATCH/POST the latest chunked page and any new chunked pages.
|
|
618
|
+
for idx, chunked_page in enumerate(new_chunked_pages):
|
|
619
|
+
if idx == 0 and chunked_page_id:
|
|
620
|
+
patch_chunked_page_func(
|
|
621
|
+
headers,
|
|
622
|
+
chunked_page_id,
|
|
623
|
+
chunked_page,
|
|
624
|
+
)
|
|
625
|
+
continue
|
|
626
|
+
post_new_chunked_page_func(
|
|
627
|
+
headers,
|
|
628
|
+
chunked_page,
|
|
629
|
+
)
|
|
630
|
+
|
|
631
|
+
# Return the chunked pages.
|
|
632
|
+
return chunked_pages[:-1] + new_chunked_pages
|
|
633
|
+
|
|
634
|
+
|
|
636
635
|
def get_n_weeks_ago(n):
|
|
637
636
|
now = dat.datetime.now().astimezone()
|
|
638
637
|
current_week_start = now - dat.timedelta(days=now.weekday())
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dimwit
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics.
|
|
5
5
|
Author-email: Daniel Soutar <danielsoutar144@gmail.com>
|
|
6
6
|
Project-URL: repository, https://github.com/danielsoutar/dimwit
|
|
@@ -14,7 +14,7 @@ dependencies = [
|
|
|
14
14
|
"tqdm>=4.67.1",
|
|
15
15
|
]
|
|
16
16
|
name = "dimwit"
|
|
17
|
-
version = "0.3.
|
|
17
|
+
version = "0.3.2"
|
|
18
18
|
description = "A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics."
|
|
19
19
|
readme = "README.md"
|
|
20
20
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|