dimwit 0.3.0__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dimwit
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Summary: A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics.
5
5
  Author-email: Daniel Soutar <danielsoutar144@gmail.com>
6
6
  Project-URL: repository, https://github.com/danielsoutar/dimwit
@@ -314,119 +314,6 @@ def group_data(
314
314
  return grouped_ts, grouped_events
315
315
 
316
316
 
317
- # This enables reading from a (chunked) database with good compression and
318
- # substantially faster reads for older data, and only a small amount of syncing
319
- # required with the slower database.
320
- # This does assume the presence of a 'date' column in the un-chunked database,
321
- # however. This enables the filtering on the slower database.
322
- def get_notion_pages_from_db_sync_with_chunked(
323
- db_id: str,
324
- chunked_db_id: str,
325
- headers: NotionHeaders,
326
- extract_chunked_func: Callable,
327
- decompression_map: dict,
328
- compression_map: dict,
329
- extract_func: Callable,
330
- compress_func: Callable,
331
- patch_chunked_page: Callable,
332
- post_new_chunked_page: Callable,
333
- ):
334
- # Read all the data from chunked_db in ascending order of id.
335
- chunked_pages = get_notion_pages_from_db(
336
- chunked_db_id,
337
- headers,
338
- sort_column="id",
339
- sort_direction="ascending",
340
- )
341
-
342
- # Get the latest date from the chunked data.
343
- latest_chunked_page = [chunked_pages[-1]] if chunked_pages else []
344
- chunked_page_ts, chunked_page_events = get_all_entries_chunked(
345
- latest_chunked_page, extract_chunked_func, decompression_map
346
- )
347
- latest_chunked_date: dat.datetime | None = (
348
- chunked_page_ts[-1] if chunked_page_ts else None
349
- )
350
- chunked_page_id: str | None = chunked_pages[-1]["id"] if chunked_pages else None
351
- print(f"Latest chunked timestamp: {latest_chunked_date}")
352
- print(f"Latest chunked page ID: {chunked_page_id = }")
353
-
354
- # Read all the data from slower_db s.t. date >= latest chunked date.
355
- # Typicall this should be a small amount of data - only on an empty chunked
356
- # database should this pay the full cost of loading/syncing all the data.
357
- filter_by = None
358
- if latest_chunked_date:
359
- filter_by = {
360
- "property": "date",
361
- "date": {"on_or_after": latest_chunked_date.isoformat()},
362
- }
363
- unchunked_pages = get_notion_pages_from_db(
364
- db_id,
365
- headers,
366
- sort_column="date",
367
- sort_direction="ascending",
368
- filter_by=filter_by,
369
- )
370
- unchunked_ts, unchunked_events = get_all_entries(
371
- unchunked_pages,
372
- extract_func,
373
- )
374
-
375
- if not latest_chunked_date:
376
- # Useful debugging utility when creating a new chunked DB.
377
- unchunked_event_freqs = {}
378
- for event in unchunked_events:
379
- if event not in unchunked_event_freqs:
380
- unchunked_event_freqs[event] = 0
381
- unchunked_event_freqs[event] += 1
382
- print(f"{unchunked_event_freqs = }")
383
-
384
- earliest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[0]
385
- latest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[-1]
386
- print(f"Earliest unchunked timestamp: {earliest_unchunked}")
387
- print(f"Latest unchunked timestamp: {latest_unchunked}")
388
-
389
- # De-duplicate the data and transform into standardised chunked pages (i.e.
390
- # update the latest chunked page and add any new chunked pages).
391
- if latest_chunked_date:
392
- dedup_chunked_pairs = [
393
- (t, e)
394
- for t, e in zip(chunked_page_ts, chunked_page_events)
395
- if t < latest_chunked_date
396
- ]
397
- dedup_ts, dedup_events = zip(*dedup_chunked_pairs)
398
- chunked_page_ts = list(dedup_ts)
399
- chunked_page_events = list(dedup_events)
400
-
401
- ts = chunked_page_ts + unchunked_ts
402
- events = chunked_page_events + unchunked_events
403
- # Group data for better compression.
404
- grouped_ts, grouped_events = group_data(ts, events)
405
- new_chunked_pages = compress_event_data(
406
- grouped_ts,
407
- grouped_events,
408
- compression_map,
409
- chunked_db_id,
410
- )
411
-
412
- # PATCH/POST the latest chunked page and any new chunked pages.
413
- for idx, chunked_page in enumerate(new_chunked_pages):
414
- if idx == 0 and chunked_page_id:
415
- patch_chunked_page(
416
- headers,
417
- chunked_page_id,
418
- chunked_page,
419
- )
420
- continue
421
- post_new_chunked_page(
422
- headers,
423
- chunked_page,
424
- )
425
-
426
- # Return the chunked pages.
427
- return chunked_pages[:-1] + new_chunked_pages
428
-
429
-
430
317
  def patch_chunked_page(
431
318
  headers: NotionHeaders,
432
319
  page_id: str,
@@ -633,6 +520,118 @@ def get_all_entries_chunked(pages, add_data_entry, decompression_map):
633
520
  return timestamps, data
634
521
 
635
522
 
523
+ # This enables reading from a (chunked) database with good compression and
524
+ # substantially faster reads for older data, and only a small amount of syncing
525
+ # required with the slower database.
526
+ # This does assume the presence of a 'date' column in the un-chunked database,
527
+ # however. This enables the filtering on the slower database.
528
+ def get_notion_pages_from_db_sync_with_chunked(
529
+ db_id: str,
530
+ chunked_db_id: str,
531
+ headers: NotionHeaders,
532
+ decompression_map: dict,
533
+ compression_map: dict,
534
+ extract_chunked_func: Callable = extract_chunked_categorical_entry,
535
+ extract_func: Callable = extract_categorical_entry,
536
+ patch_chunked_page_func: Callable = patch_chunked_page,
537
+ post_new_chunked_page_func: Callable = post_new_chunked_page,
538
+ ):
539
+ # Read all the data from chunked_db in ascending order of id.
540
+ chunked_pages = get_notion_pages_from_db(
541
+ chunked_db_id,
542
+ headers,
543
+ sort_column="id",
544
+ sort_direction="ascending",
545
+ )
546
+
547
+ # Get the latest date from the chunked data.
548
+ latest_chunked_page = [chunked_pages[-1]] if chunked_pages else []
549
+ chunked_page_ts, chunked_page_events = get_all_entries_chunked(
550
+ latest_chunked_page, extract_chunked_func, decompression_map
551
+ )
552
+ latest_chunked_date: dat.datetime | None = (
553
+ chunked_page_ts[-1] if chunked_page_ts else None
554
+ )
555
+ chunked_page_id: str | None = chunked_pages[-1]["id"] if chunked_pages else None
556
+ print(f"Latest chunked timestamp: {latest_chunked_date}")
557
+ print(f"Latest chunked page ID: {chunked_page_id = }")
558
+
559
+ # Read all the data from slower_db s.t. date >= latest chunked date.
560
+ # Typicall this should be a small amount of data - only on an empty chunked
561
+ # database should this pay the full cost of loading/syncing all the data.
562
+ filter_by = None
563
+ if latest_chunked_date:
564
+ filter_by = {
565
+ "property": "date",
566
+ "date": {"on_or_after": latest_chunked_date.isoformat()},
567
+ }
568
+ unchunked_pages = get_notion_pages_from_db(
569
+ db_id,
570
+ headers,
571
+ sort_column="date",
572
+ sort_direction="ascending",
573
+ filter_by=filter_by,
574
+ )
575
+ unchunked_ts, unchunked_events = get_all_entries(
576
+ unchunked_pages,
577
+ extract_func,
578
+ )
579
+
580
+ if not latest_chunked_date:
581
+ # Useful debugging utility when creating a new chunked DB.
582
+ unchunked_event_freqs = {}
583
+ for event in unchunked_events:
584
+ if event not in unchunked_event_freqs:
585
+ unchunked_event_freqs[event] = 0
586
+ unchunked_event_freqs[event] += 1
587
+ print(f"{unchunked_event_freqs = }")
588
+
589
+ earliest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[0]
590
+ latest_unchunked = None if len(unchunked_ts) == 0 else unchunked_ts[-1]
591
+ print(f"Earliest unchunked timestamp: {earliest_unchunked}")
592
+ print(f"Latest unchunked timestamp: {latest_unchunked}")
593
+
594
+ # De-duplicate the data and transform into standardised chunked pages (i.e.
595
+ # update the latest chunked page and add any new chunked pages).
596
+ if latest_chunked_date:
597
+ dedup_chunked_pairs = [
598
+ (t, e)
599
+ for t, e in zip(chunked_page_ts, chunked_page_events)
600
+ if t < latest_chunked_date
601
+ ]
602
+ dedup_ts, dedup_events = zip(*dedup_chunked_pairs)
603
+ chunked_page_ts = list(dedup_ts)
604
+ chunked_page_events = list(dedup_events)
605
+
606
+ ts = chunked_page_ts + unchunked_ts
607
+ events = chunked_page_events + unchunked_events
608
+ # Group data for better compression.
609
+ grouped_ts, grouped_events = group_data(ts, events)
610
+ new_chunked_pages = compress_event_data(
611
+ grouped_ts,
612
+ grouped_events,
613
+ compression_map,
614
+ chunked_db_id,
615
+ )
616
+
617
+ # PATCH/POST the latest chunked page and any new chunked pages.
618
+ for idx, chunked_page in enumerate(new_chunked_pages):
619
+ if idx == 0 and chunked_page_id:
620
+ patch_chunked_page_func(
621
+ headers,
622
+ chunked_page_id,
623
+ chunked_page,
624
+ )
625
+ continue
626
+ post_new_chunked_page_func(
627
+ headers,
628
+ chunked_page,
629
+ )
630
+
631
+ # Return the chunked pages.
632
+ return chunked_pages[:-1] + new_chunked_pages
633
+
634
+
636
635
  def get_n_weeks_ago(n):
637
636
  now = dat.datetime.now().astimezone()
638
637
  current_week_start = now - dat.timedelta(days=now.weekday())
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dimwit
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Summary: A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics.
5
5
  Author-email: Daniel Soutar <danielsoutar144@gmail.com>
6
6
  Project-URL: repository, https://github.com/danielsoutar/dimwit
@@ -14,7 +14,7 @@ dependencies = [
14
14
  "tqdm>=4.67.1",
15
15
  ]
16
16
  name = "dimwit"
17
- version = "0.3.0"
17
+ version = "0.3.2"
18
18
  description = "A package containing various functions and classes for fetching, transforming, and visualising data for personal metrics."
19
19
  readme = "README.md"
20
20
 
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes