library 3.2.2__tar.gz → 3.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (160) hide show
  1. {library-3.2.2 → library-3.2.3}/.github/README.md +7 -36
  2. {library-3.2.2 → library-3.2.3}/PKG-INFO +8 -37
  3. {library-3.2.2 → library-3.2.3}/library/__main__.py +1 -3
  4. {library-3.2.2 → library-3.2.3}/library/createdb/fs_add.py +2 -2
  5. {library-3.2.2 → library-3.2.3}/library/createdb/fs_add_metadata.py +8 -2
  6. {library-3.2.2 → library-3.2.3}/library/createdb/getty_add.py +26 -4
  7. {library-3.2.2 → library-3.2.3}/library/createdb/hn_add.py +2 -2
  8. {library-3.2.2 → library-3.2.3}/library/editdb/dedupe_db.py +24 -4
  9. {library-3.2.2 → library-3.2.3}/library/editdb/dedupe_media.py +1 -6
  10. {library-3.2.2 → library-3.2.3}/library/files/similar_files.py +4 -4
  11. {library-3.2.2 → library-3.2.3}/library/folders/big_dirs.py +1 -0
  12. {library-3.2.2 → library-3.2.3}/library/folders/merge_mv.py +2 -1
  13. {library-3.2.2 → library-3.2.3}/library/folders/similar_folders.py +4 -3
  14. {library-3.2.2 → library-3.2.3}/library/mediadb/download.py +21 -17
  15. {library-3.2.2 → library-3.2.3}/library/mediadb/download_status.py +3 -2
  16. {library-3.2.2 → library-3.2.3}/library/mediafiles/process_ffmpeg.py +44 -23
  17. {library-3.2.2 → library-3.2.3}/library/mediafiles/process_image.py +10 -3
  18. {library-3.2.2 → library-3.2.3}/library/mediafiles/process_media.py +59 -3
  19. {library-3.2.2 → library-3.2.3}/library/mediafiles/process_text.py +10 -5
  20. {library-3.2.2 → library-3.2.3}/library/misc/export_text.py +1 -4
  21. {library-3.2.2 → library-3.2.3}/library/playback/play_actions.py +5 -2
  22. {library-3.2.2 → library-3.2.3}/library/playback/playback_control.py +3 -3
  23. {library-3.2.2 → library-3.2.3}/library/playback/surf.py +13 -8
  24. {library-3.2.2 → library-3.2.3}/library/playback/torrents_info.py +1 -1
  25. {library-3.2.2 → library-3.2.3}/library/tablefiles/eda.py +1 -1
  26. {library-3.2.2 → library-3.2.3}/library/tablefiles/mcda.py +1 -1
  27. {library-3.2.2 → library-3.2.3}/library/text/cluster_sort.py +36 -44
  28. {library-3.2.2 → library-3.2.3}/library/text/extract_text.py +2 -7
  29. {library-3.2.2 → library-3.2.3}/library/usage.py +6 -26
  30. {library-3.2.2 → library-3.2.3}/library/utils/arg_utils.py +0 -12
  31. {library-3.2.2 → library-3.2.3}/library/utils/arggroups.py +7 -4
  32. {library-3.2.2 → library-3.2.3}/library/utils/argparse_utils.py +0 -9
  33. {library-3.2.2 → library-3.2.3}/library/utils/db_utils.py +1 -18
  34. {library-3.2.2 → library-3.2.3}/library/utils/devices.py +20 -15
  35. {library-3.2.2 → library-3.2.3}/library/utils/filter_engine.py +0 -3
  36. {library-3.2.2 → library-3.2.3}/library/utils/log_utils.py +0 -14
  37. {library-3.2.2 → library-3.2.3}/library/utils/mpv_utils.py +12 -1
  38. {library-3.2.2 → library-3.2.3}/library/utils/path_utils.py +0 -17
  39. {library-3.2.2 → library-3.2.3}/library/utils/pd_utils.py +0 -5
  40. {library-3.2.2 → library-3.2.3}/library/utils/shell_utils.py +1 -1
  41. {library-3.2.2 → library-3.2.3}/library/utils/strings.py +0 -1
  42. {library-3.2.2 → library-3.2.3}/pyproject.toml +1 -1
  43. library-3.2.2/library/files/llm_map.py +0 -121
  44. {library-3.2.2 → library-3.2.3}/LICENSE +0 -0
  45. {library-3.2.2 → library-3.2.3}/library/__init__.py +0 -0
  46. {library-3.2.2 → library-3.2.3}/library/assets/__init__.py +0 -0
  47. {library-3.2.2 → library-3.2.3}/library/assets/calibre.css +0 -0
  48. {library-3.2.2 → library-3.2.3}/library/assets/kotobago.png +0 -0
  49. {library-3.2.2 → library-3.2.3}/library/createdb/__init__.py +0 -0
  50. {library-3.2.2 → library-3.2.3}/library/createdb/av.py +0 -0
  51. {library-3.2.2 → library-3.2.3}/library/createdb/computer_info.py +0 -0
  52. {library-3.2.2 → library-3.2.3}/library/createdb/computers_add.py +0 -0
  53. {library-3.2.2 → library-3.2.3}/library/createdb/gallery_add.py +0 -0
  54. {library-3.2.2 → library-3.2.3}/library/createdb/gallery_backend.py +0 -0
  55. {library-3.2.2 → library-3.2.3}/library/createdb/links_add.py +0 -0
  56. {library-3.2.2 → library-3.2.3}/library/createdb/nicotine_import.py +0 -0
  57. {library-3.2.2 → library-3.2.3}/library/createdb/places_import.py +0 -0
  58. {library-3.2.2 → library-3.2.3}/library/createdb/reddit_add.py +0 -0
  59. {library-3.2.2 → library-3.2.3}/library/createdb/row_add.py +0 -0
  60. {library-3.2.2 → library-3.2.3}/library/createdb/site_add.py +0 -0
  61. {library-3.2.2 → library-3.2.3}/library/createdb/substack.py +0 -0
  62. {library-3.2.2 → library-3.2.3}/library/createdb/subtitle.py +0 -0
  63. {library-3.2.2 → library-3.2.3}/library/createdb/tables_add.py +0 -0
  64. {library-3.2.2 → library-3.2.3}/library/createdb/tabs_add.py +0 -0
  65. {library-3.2.2 → library-3.2.3}/library/createdb/tildes.py +0 -0
  66. {library-3.2.2 → library-3.2.3}/library/createdb/torrents_add.py +0 -0
  67. {library-3.2.2 → library-3.2.3}/library/createdb/tube_add.py +0 -0
  68. {library-3.2.2 → library-3.2.3}/library/createdb/tube_backend.py +0 -0
  69. {library-3.2.2 → library-3.2.3}/library/createdb/web_add.py +0 -0
  70. {library-3.2.2 → library-3.2.3}/library/data/__init__.py +0 -0
  71. {library-3.2.2 → library-3.2.3}/library/data/dictionary.pkl +0 -0
  72. {library-3.2.2 → library-3.2.3}/library/data/ffmpeg_errors.py +0 -0
  73. {library-3.2.2 → library-3.2.3}/library/data/http_errors.py +0 -0
  74. {library-3.2.2 → library-3.2.3}/library/data/imagemagick_errors.py +0 -0
  75. {library-3.2.2 → library-3.2.3}/library/data/unar_errors.py +0 -0
  76. {library-3.2.2 → library-3.2.3}/library/data/wordbank.py +0 -0
  77. {library-3.2.2 → library-3.2.3}/library/data/yt_dlp_errors.py +0 -0
  78. {library-3.2.2 → library-3.2.3}/library/editdb/__init__.py +0 -0
  79. {library-3.2.2 → library-3.2.3}/library/editdb/merge_online_local.py +0 -0
  80. {library-3.2.2 → library-3.2.3}/library/editdb/mpv_watchlater.py +0 -0
  81. {library-3.2.2 → library-3.2.3}/library/editdb/pushshift.py +0 -0
  82. {library-3.2.2 → library-3.2.3}/library/editdb/reddit_selftext.py +0 -0
  83. {library-3.2.2 → library-3.2.3}/library/files/__init__.py +0 -0
  84. {library-3.2.2 → library-3.2.3}/library/files/christen.py +0 -0
  85. {library-3.2.2 → library-3.2.3}/library/files/sample_compare.py +0 -0
  86. {library-3.2.2 → library-3.2.3}/library/files/sample_hash.py +0 -0
  87. {library-3.2.2 → library-3.2.3}/library/folders/__init__.py +0 -0
  88. {library-3.2.2 → library-3.2.3}/library/folders/filter_src.py +0 -0
  89. {library-3.2.2 → library-3.2.3}/library/folders/mergerfs_cp.py +0 -0
  90. {library-3.2.2 → library-3.2.3}/library/folders/mount_stats.py +0 -0
  91. {library-3.2.2 → library-3.2.3}/library/folders/move_list.py +0 -0
  92. {library-3.2.2 → library-3.2.3}/library/folders/scatter.py +0 -0
  93. {library-3.2.2 → library-3.2.3}/library/fsdb/__init__.py +0 -0
  94. {library-3.2.2 → library-3.2.3}/library/fsdb/disk_usage.py +0 -0
  95. {library-3.2.2 → library-3.2.3}/library/fsdb/filesystem.py +0 -0
  96. {library-3.2.2 → library-3.2.3}/library/fsdb/folder_stats.py +0 -0
  97. {library-3.2.2 → library-3.2.3}/library/fsdb/search_db.py +0 -0
  98. {library-3.2.2 → library-3.2.3}/library/lb.py +0 -0
  99. {library-3.2.2 → library-3.2.3}/library/mediadb/__init__.py +0 -0
  100. {library-3.2.2 → library-3.2.3}/library/mediadb/block.py +0 -0
  101. {library-3.2.2 → library-3.2.3}/library/mediadb/db_history.py +0 -0
  102. {library-3.2.2 → library-3.2.3}/library/mediadb/db_media.py +0 -0
  103. {library-3.2.2 → library-3.2.3}/library/mediadb/db_playlists.py +0 -0
  104. {library-3.2.2 → library-3.2.3}/library/mediadb/history.py +0 -0
  105. {library-3.2.2 → library-3.2.3}/library/mediadb/history_add.py +0 -0
  106. {library-3.2.2 → library-3.2.3}/library/mediadb/optimize_db.py +0 -0
  107. {library-3.2.2 → library-3.2.3}/library/mediadb/playlists.py +0 -0
  108. {library-3.2.2 → library-3.2.3}/library/mediadb/redownload.py +0 -0
  109. {library-3.2.2 → library-3.2.3}/library/mediadb/search.py +0 -0
  110. {library-3.2.2 → library-3.2.3}/library/mediadb/stats.py +3 -3
  111. {library-3.2.2 → library-3.2.3}/library/mediafiles/__init__.py +0 -0
  112. {library-3.2.2 → library-3.2.3}/library/mediafiles/images_to_pdf.py +0 -0
  113. {library-3.2.2 → library-3.2.3}/library/mediafiles/media_check.py +0 -0
  114. {library-3.2.2 → library-3.2.3}/library/mediafiles/pdf_edit.py +0 -0
  115. {library-3.2.2 → library-3.2.3}/library/mediafiles/torrents_dump.py +0 -0
  116. {library-3.2.2 → library-3.2.3}/library/mediafiles/torrents_start.py +0 -0
  117. {library-3.2.2 → library-3.2.3}/library/mediafiles/unardel.py +0 -0
  118. {library-3.2.2 → library-3.2.3}/library/misc/__init__.py +0 -0
  119. {library-3.2.2 → library-3.2.3}/library/misc/dedupe_czkawka.py +0 -0
  120. {library-3.2.2 → library-3.2.3}/library/misc/search_help.py +0 -0
  121. {library-3.2.2 → library-3.2.3}/library/multidb/__init__.py +0 -0
  122. {library-3.2.2 → library-3.2.3}/library/multidb/allocate_torrents.py +0 -0
  123. {library-3.2.2 → library-3.2.3}/library/multidb/copy_play_counts.py +0 -0
  124. {library-3.2.2 → library-3.2.3}/library/multidb/merge_dbs.py +0 -0
  125. {library-3.2.2 → library-3.2.3}/library/playback/__init__.py +0 -0
  126. {library-3.2.2 → library-3.2.3}/library/playback/links_open.py +0 -0
  127. {library-3.2.2 → library-3.2.3}/library/playback/media_player.py +0 -0
  128. {library-3.2.2 → library-3.2.3}/library/playback/media_printer.py +0 -0
  129. {library-3.2.2 → library-3.2.3}/library/playback/post_actions.py +0 -0
  130. {library-3.2.2 → library-3.2.3}/library/playback/tabs_open.py +0 -0
  131. {library-3.2.2 → library-3.2.3}/library/playback/torrents_remaining.py +0 -0
  132. {library-3.2.2 → library-3.2.3}/library/readme.py +0 -0
  133. {library-3.2.2 → library-3.2.3}/library/tablefiles/__init__.py +0 -0
  134. {library-3.2.2 → library-3.2.3}/library/tablefiles/columns.py +0 -0
  135. {library-3.2.2 → library-3.2.3}/library/tablefiles/incremental_diff.py +0 -0
  136. {library-3.2.2 → library-3.2.3}/library/tablefiles/markdown_tables.py +0 -0
  137. {library-3.2.2 → library-3.2.3}/library/tablefiles/plot.py +0 -0
  138. {library-3.2.2 → library-3.2.3}/library/text/__init__.py +0 -0
  139. {library-3.2.2 → library-3.2.3}/library/text/combinations.py +0 -0
  140. {library-3.2.2 → library-3.2.3}/library/text/expand_links.py +0 -0
  141. {library-3.2.2 → library-3.2.3}/library/text/extract_links.py +0 -0
  142. {library-3.2.2 → library-3.2.3}/library/text/json_keys_rename.py +0 -0
  143. {library-3.2.2 → library-3.2.3}/library/text/markdown_links.py +0 -0
  144. {library-3.2.2 → library-3.2.3}/library/text/nouns.py +0 -0
  145. {library-3.2.2 → library-3.2.3}/library/text/regex_sort.py +0 -0
  146. {library-3.2.2 → library-3.2.3}/library/text/timestamps.py +0 -0
  147. {library-3.2.2 → library-3.2.3}/library/utils/__init__.py +0 -0
  148. {library-3.2.2 → library-3.2.3}/library/utils/consts.py +0 -0
  149. {library-3.2.2 → library-3.2.3}/library/utils/date_utils.py +0 -0
  150. {library-3.2.2 → library-3.2.3}/library/utils/file_utils.py +0 -0
  151. {library-3.2.2 → library-3.2.3}/library/utils/gui.py +0 -0
  152. {library-3.2.2 → library-3.2.3}/library/utils/iterables.py +0 -0
  153. {library-3.2.2 → library-3.2.3}/library/utils/nums.py +0 -0
  154. {library-3.2.2 → library-3.2.3}/library/utils/objects.py +0 -0
  155. {library-3.2.2 → library-3.2.3}/library/utils/printing.py +0 -0
  156. {library-3.2.2 → library-3.2.3}/library/utils/processes.py +0 -0
  157. {library-3.2.2 → library-3.2.3}/library/utils/remote_processes.py +0 -0
  158. {library-3.2.2 → library-3.2.3}/library/utils/sql_utils.py +0 -0
  159. {library-3.2.2 → library-3.2.3}/library/utils/sqlgroups.py +0 -0
  160. {library-3.2.2 → library-3.2.3}/library/utils/web.py +0 -0
@@ -99,7 +99,7 @@ To stop playing press Ctrl+C in either the terminal or mpv
99
99
  <details><summary>List all subcommands</summary>
100
100
 
101
101
  $ library
102
- library (v3.2.002; 103 subcommands)
102
+ library (v3.2.003; 102 subcommands)
103
103
 
104
104
  Create database subcommands:
105
105
  ╭─────────────────┬──────────────────────────────────────────╮
@@ -197,8 +197,6 @@ To stop playing press Ctrl+C in either the terminal or mpv
197
197
  │ filesystem │ Find files by mimetype and size │
198
198
  ├────────────────┼─────────────────────────────────────────────────────┤
199
199
  │ similar-files │ Find similar files based on filename and size │
200
- ├────────────────┼─────────────────────────────────────────────────────┤
201
- │ llm-map │ Run LLMs across multiple files │
202
200
  ╰────────────────┴─────────────────────────────────────────────────────╯
203
201
 
204
202
  Tabular data subcommands:
@@ -1614,11 +1612,11 @@ BTW, for some cols like time_deleted you'll need to specify a where clause so th
1614
1612
  <details><summary>Find similar folders based on folder name, size, and count</summary>
1615
1613
 
1616
1614
  $ library similar-folders -h
1617
- usage: library similar-folders PATH ...
1615
+ usage: library similar-folders (--filter-names | --filter-counts | --filter-sizes | --filter-durations) PATH ...
1618
1616
 
1619
1617
  Find similar folders based on foldernames, similar size, and similar number of files
1620
1618
 
1621
- library similar-folders ~/d/
1619
+ library similar-folders --filter-names --filter-counts --filter-sizes ~/d/
1622
1620
 
1623
1621
  group /home/xk/d/dump/datasets/*vector total_size median_size files
1624
1622
  ---------------------------------------------- ------------ ------------- -------
@@ -1730,11 +1728,11 @@ BTW, for some cols like time_deleted you'll need to specify a where clause so th
1730
1728
  <details><summary>Find similar files based on filename and size</summary>
1731
1729
 
1732
1730
  $ library similar-files -h
1733
- usage: library similar-files PATH ...
1731
+ usage: library similar-files (--filter-names | --filter-sizes | --filter-durations) PATH ...
1734
1732
 
1735
1733
  Find similar files using filenames and size
1736
1734
 
1737
- library similar-files ~/d/
1735
+ library similar-files --filter-names --filter-sizes ~/d/
1738
1736
 
1739
1737
  Find similar files based on ONLY foldernames, using the full path
1740
1738
 
@@ -1752,34 +1750,6 @@ BTW, for some cols like time_deleted you'll need to specify a where clause so th
1752
1750
  library similar-files --filter-names --filter-durations --estimated-duplicates 3 .
1753
1751
 
1754
1752
 
1755
- </details>
1756
-
1757
- ###### llm-map
1758
-
1759
- <details><summary>Run LLMs across multiple files</summary>
1760
-
1761
- $ library llm-map -h
1762
- usage: library llm-map LLAMA_FILE [paths ...] [--llama-args LLAMA_ARGS] [--prompt STR] [--text [INT]] [--rename]
1763
-
1764
- Run a llamafile with a prompt including path names and file contents
1765
-
1766
- Rename files based on file contents
1767
-
1768
- library llm-map ./gemma2.llamafile ~/Downloads/booka.pdf --rename --text
1769
-
1770
- cat llm_map_renames.csv
1771
- Path,Output
1772
- /home/xk/Downloads/booka.pdf,/home/xk/Downloads/Mining_Massive_Datasets.pdf
1773
-
1774
- Using GGUF files
1775
-
1776
- wget https://github.com/Mozilla-Ocho/llamafile/releases/download/0.8.9/llamafile-0.8.9
1777
- chmod +x ~/Downloads/llamafile-0.8.9
1778
- mv ~/Downloads/llamafile-0.8.9 ~/.local/bin/llamafile # move it somewhere in your $PATH
1779
-
1780
- library llm-map --model ~/Downloads/llava-v1.5-7b-Q4_K.gguf --image-model ~/Downloads/llava-v1.5-7b-mmproj-Q4_0.gguf --prompt 'what do you see?' ~/Downloads/comp_*.jpg
1781
-
1782
-
1783
1753
  </details>
1784
1754
 
1785
1755
  ### Tabular data subcommands
@@ -2487,9 +2457,10 @@ Inspired somewhat by https://nikkhokkho.sourceforge.io/?page=FileOptimizer
2487
2457
  <details><summary>Download media</summary>
2488
2458
 
2489
2459
  $ library download -h
2490
- usage: library download DATABASE [--prefix /mnt/d/] --video [--subs] [--auto-subs] [--small] | --audio | --photos [--safe]
2460
+ usage: library download DATABASE [--prefix /mnt/d/] [--video (default) | --audio | --image | --filesystem] [--safe]
2491
2461
 
2492
2462
  Files will be saved to <prefix>/<extractor>/. The default prefix is the current working directory.
2463
+ The default download profile is video. `--safe` supports audio, video, and image profiles only.
2493
2464
 
2494
2465
  By default things will download in a random order
2495
2466
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: library
3
- Version: 3.2.2
3
+ Version: 3.2.3
4
4
  Summary: xk media library
5
5
  Author-Email: Jacob Chapman <7908073+chapmanjacobd@users.noreply.github.com>
6
6
  License: BSD 3-Clause License
@@ -207,7 +207,7 @@ To stop playing press Ctrl+C in either the terminal or mpv
207
207
  <details><summary>List all subcommands</summary>
208
208
 
209
209
  $ library
210
- library (v3.2.002; 103 subcommands)
210
+ library (v3.2.003; 102 subcommands)
211
211
 
212
212
  Create database subcommands:
213
213
  ╭─────────────────┬──────────────────────────────────────────╮
@@ -305,8 +305,6 @@ To stop playing press Ctrl+C in either the terminal or mpv
305
305
  │ filesystem │ Find files by mimetype and size │
306
306
  ├────────────────┼─────────────────────────────────────────────────────┤
307
307
  │ similar-files │ Find similar files based on filename and size │
308
- ├────────────────┼─────────────────────────────────────────────────────┤
309
- │ llm-map │ Run LLMs across multiple files │
310
308
  ╰────────────────┴─────────────────────────────────────────────────────╯
311
309
 
312
310
  Tabular data subcommands:
@@ -1722,11 +1720,11 @@ BTW, for some cols like time_deleted you'll need to specify a where clause so th
1722
1720
  <details><summary>Find similar folders based on folder name, size, and count</summary>
1723
1721
 
1724
1722
  $ library similar-folders -h
1725
- usage: library similar-folders PATH ...
1723
+ usage: library similar-folders (--filter-names | --filter-counts | --filter-sizes | --filter-durations) PATH ...
1726
1724
 
1727
1725
  Find similar folders based on foldernames, similar size, and similar number of files
1728
1726
 
1729
- library similar-folders ~/d/
1727
+ library similar-folders --filter-names --filter-counts --filter-sizes ~/d/
1730
1728
 
1731
1729
  group /home/xk/d/dump/datasets/*vector total_size median_size files
1732
1730
  ---------------------------------------------- ------------ ------------- -------
@@ -1838,11 +1836,11 @@ BTW, for some cols like time_deleted you'll need to specify a where clause so th
1838
1836
  <details><summary>Find similar files based on filename and size</summary>
1839
1837
 
1840
1838
  $ library similar-files -h
1841
- usage: library similar-files PATH ...
1839
+ usage: library similar-files (--filter-names | --filter-sizes | --filter-durations) PATH ...
1842
1840
 
1843
1841
  Find similar files using filenames and size
1844
1842
 
1845
- library similar-files ~/d/
1843
+ library similar-files --filter-names --filter-sizes ~/d/
1846
1844
 
1847
1845
  Find similar files based on ONLY foldernames, using the full path
1848
1846
 
@@ -1860,34 +1858,6 @@ BTW, for some cols like time_deleted you'll need to specify a where clause so th
1860
1858
  library similar-files --filter-names --filter-durations --estimated-duplicates 3 .
1861
1859
 
1862
1860
 
1863
- </details>
1864
-
1865
- ###### llm-map
1866
-
1867
- <details><summary>Run LLMs across multiple files</summary>
1868
-
1869
- $ library llm-map -h
1870
- usage: library llm-map LLAMA_FILE [paths ...] [--llama-args LLAMA_ARGS] [--prompt STR] [--text [INT]] [--rename]
1871
-
1872
- Run a llamafile with a prompt including path names and file contents
1873
-
1874
- Rename files based on file contents
1875
-
1876
- library llm-map ./gemma2.llamafile ~/Downloads/booka.pdf --rename --text
1877
-
1878
- cat llm_map_renames.csv
1879
- Path,Output
1880
- /home/xk/Downloads/booka.pdf,/home/xk/Downloads/Mining_Massive_Datasets.pdf
1881
-
1882
- Using GGUF files
1883
-
1884
- wget https://github.com/Mozilla-Ocho/llamafile/releases/download/0.8.9/llamafile-0.8.9
1885
- chmod +x ~/Downloads/llamafile-0.8.9
1886
- mv ~/Downloads/llamafile-0.8.9 ~/.local/bin/llamafile # move it somewhere in your $PATH
1887
-
1888
- library llm-map --model ~/Downloads/llava-v1.5-7b-Q4_K.gguf --image-model ~/Downloads/llava-v1.5-7b-mmproj-Q4_0.gguf --prompt 'what do you see?' ~/Downloads/comp_*.jpg
1889
-
1890
-
1891
1861
  </details>
1892
1862
 
1893
1863
  ### Tabular data subcommands
@@ -2595,9 +2565,10 @@ Inspired somewhat by https://nikkhokkho.sourceforge.io/?page=FileOptimizer
2595
2565
  <details><summary>Download media</summary>
2596
2566
 
2597
2567
  $ library download -h
2598
- usage: library download DATABASE [--prefix /mnt/d/] --video [--subs] [--auto-subs] [--small] | --audio | --photos [--safe]
2568
+ usage: library download DATABASE [--prefix /mnt/d/] [--video (default) | --audio | --image | --filesystem] [--safe]
2599
2569
 
2600
2570
  Files will be saved to <prefix>/<extractor>/. The default prefix is the current working directory.
2571
+ The default download profile is video. `--safe` supports audio, video, and image profiles only.
2601
2572
 
2602
2573
  By default things will download in a random order
2603
2574
 
@@ -5,7 +5,7 @@ from tabulate import tabulate
5
5
  from library.utils import argparse_utils, iterables
6
6
  from library.utils.log_utils import log
7
7
 
8
- __version__ = "3.2.002"
8
+ __version__ = "3.2.003"
9
9
 
10
10
  progs = {
11
11
  "Create database subcommands": {
@@ -58,7 +58,6 @@ progs = {
58
58
  "sample_compare": "Compare files using sample-hash and other shortcuts",
59
59
  "filesystem": "Find files by mimetype and size",
60
60
  "similar_files": "Find similar files based on filename and size",
61
- "llm_map": "Run LLMs across multiple files",
62
61
  },
63
62
  "Tabular data subcommands": {
64
63
  "eda": "Exploratory Data Analysis on table-like files",
@@ -197,7 +196,6 @@ modules = {
197
196
  "library.files.sample_compare.sample_compare": ["cmp"],
198
197
  "library.files.sample_hash.sample_hash": ["hash", "hash-file"],
199
198
  "library.files.similar_files.similar_files": [],
200
- "library.files.llm_map.llm_map": [],
201
199
  "library.folders.move_list.move_list": ["mv-list"],
202
200
  "library.folders.merge_mv.merge_mv": ["mv"],
203
201
  "library.folders.merge_mv.merge_cp": ["cp"],
@@ -264,7 +264,7 @@ def scan_path(args, path_str: str) -> int:
264
264
  while m is None:
265
265
  m = extract_metadata(args, new_files.pop())
266
266
 
267
- extract_chunk(args, [m])
267
+ extract_chunk(args, iterables.conform(m))
268
268
  db_utils.optimize(args)
269
269
  del args.playlist_path
270
270
 
@@ -301,7 +301,7 @@ def scan_path(args, path_str: str) -> int:
301
301
  playlist_path=path, **{k: v for k, v in args.__dict__.items() if k not in {"db"}}
302
302
  )
303
303
  metadata = parallel.map(partial(extract_metadata, mp_args), chunk_paths)
304
- metadata = list(filter(None, metadata))
304
+ metadata = iterables.conform(metadata)
305
305
  extract_chunk(args, metadata)
306
306
  print()
307
307
 
@@ -1,4 +1,4 @@
1
- import os, re
1
+ import argparse, os, re
2
2
  from multiprocessing import TimeoutError as mp_TimeoutError
3
3
  from pathlib import Path
4
4
  from timeit import default_timer as timer
@@ -39,7 +39,7 @@ munge_book_tags_slow = processes.with_timeout(350)(munge_book_tags)
39
39
  munge_book_tags_fast = processes.with_timeout(70)(munge_book_tags)
40
40
 
41
41
 
42
- def extract_metadata(mp_args, path) -> dict[str, str | int | None] | None:
42
+ def extract_metadata(mp_args, path) -> dict[str, str | int | None] | list[dict[str, str | int | None]] | None:
43
43
  try:
44
44
  path.encode()
45
45
  except UnicodeEncodeError:
@@ -127,11 +127,17 @@ def extract_metadata(mp_args, path) -> dict[str, str | int | None] | None:
127
127
  )
128
128
  if result is None:
129
129
  return None
130
+ if isinstance(result, list):
131
+ processed_args = argparse.Namespace(**(vars(mp_args) | {"process": False}))
132
+ return [m for output_path in result if (m := extract_metadata(processed_args, output_path))]
130
133
  path = m["path"] = str(result)
131
134
  elif objects.is_profile(mp_args, DBType.video) and Path(path).suffix not in [".av1.mkv"]:
132
135
  result = process_ffmpeg.process_path(mp_args, path)
133
136
  if result is None:
134
137
  return None
138
+ if isinstance(result, list):
139
+ processed_args = argparse.Namespace(**(vars(mp_args) | {"process": False}))
140
+ return [m for output_path in result if (m := extract_metadata(processed_args, output_path))]
135
141
  path = m["path"] = str(result)
136
142
  elif objects.is_profile(mp_args, DBType.image) and Path(path).suffix not in [".avif", ".avifs"]:
137
143
  result = process_image.process_path(mp_args, path)
@@ -2,10 +2,13 @@ import sqlite3
2
2
 
3
3
  from library import usage
4
4
  from library.data.http_errors import HTTPStatus
5
- from library.utils import arggroups, argparse_utils, iterables, web
5
+ from library.mediadb import db_media, db_playlists
6
+ from library.utils import arggroups, argparse_utils, consts, iterables, web
6
7
  from library.utils.log_utils import log
7
8
  from library.utils.objects import traverse_obj
8
9
 
10
+ GETTY_COLLECTION_URL = "https://data.getty.edu/museum/collection/"
11
+
9
12
 
10
13
  def parse_args():
11
14
  parser = argparse_utils.ArgumentParser(usage=usage.getty_add)
@@ -15,6 +18,7 @@ def parse_args():
15
18
  arggroups.database(parser)
16
19
  args = parser.parse_args()
17
20
  arggroups.args_post(args, parser, create_db=True)
21
+ args.profile = consts.DBType.image
18
22
 
19
23
  web.requests_session(args) # prepare requests session
20
24
 
@@ -163,7 +167,7 @@ def objects_extract(args, j):
163
167
  )
164
168
 
165
169
  d = {
166
- "path": image_path or None,
170
+ "path": image_path or j["id"],
167
171
  "title": j["_label"],
168
172
  "types": "; ".join(set(d["_label"] for d in j["classified_as"]) - ignore_types),
169
173
  "description": description,
@@ -199,6 +203,12 @@ def update_objects(args):
199
203
  ]
200
204
 
201
205
  print("Fetching", len(unknown_objects), "unknown objects")
206
+ playlists_id = db_playlists.add(
207
+ args,
208
+ GETTY_COLLECTION_URL,
209
+ {"title": "Getty Museum collection"},
210
+ extractor_key="Getty",
211
+ )
202
212
 
203
213
  for unknown_object in unknown_objects:
204
214
  log.debug("Fetching %s...", unknown_object)
@@ -206,14 +216,26 @@ def update_objects(args):
206
216
  page_data = getty_fetch(unknown_object)
207
217
  if page_data:
208
218
  images = objects_extract(args, page_data)
209
- args.db["media"].insert_all(images, alter=True, pk="id")
219
+ for image in images:
220
+ db_media.add(args, {"playlists_id": playlists_id, **image})
210
221
  else:
211
- args.db["media"].insert({"title": "404 Not Found", "object_path": unknown_object}, alter=True, pk="id")
222
+ db_media.add(
223
+ args,
224
+ {
225
+ "playlists_id": playlists_id,
226
+ "path": unknown_object,
227
+ "title": "404 Not Found",
228
+ "object_path": unknown_object,
229
+ },
230
+ )
212
231
 
213
232
 
214
233
  def getty_add():
215
234
  args = parse_args()
216
235
 
236
+ db_playlists.create(args)
237
+ db_media.create(args)
238
+
217
239
  update_activity_stream(args)
218
240
 
219
241
  """
@@ -1,4 +1,4 @@
1
- import argparse, asyncio, queue, sqlite3, threading
1
+ import argparse, asyncio, importlib, queue, sqlite3, threading
2
2
 
3
3
  from library import usage
4
4
  from library.utils import arggroups, argparse_utils, db_utils, objects, web
@@ -103,7 +103,7 @@ async def run(args, db_queue):
103
103
  def hacker_news_add() -> None:
104
104
  args = parse_args(usage=usage.hn_add)
105
105
  try:
106
- import aiohttp
106
+ importlib.import_module("aiohttp")
107
107
  except ModuleNotFoundError:
108
108
  log.error("aiohttp is required for hn_extract. Install with pip install aiohttp or pip install library[deluxe]")
109
109
  raise
@@ -36,11 +36,31 @@ def dedupe_rows(args, tablename, primary_keys, business_keys):
36
36
  with args.db.conn:
37
37
  args.db.conn.execute(
38
38
  f"""
39
- DELETE FROM {tablename}
40
- WHERE ({','.join(primary_keys)}) NOT IN (
41
- SELECT {','.join(f"MIN({col}) AS {col}" for col in primary_keys)}
39
+ WITH ranked AS (
40
+ SELECT
41
+ rowid
42
+ , {','.join(business_keys)}
43
+ , ROW_NUMBER() OVER (
44
+ PARTITION BY {','.join(business_keys)}
45
+ ORDER BY {','.join(primary_keys)}, rowid
46
+ ) AS duplicate_number
47
+ , COUNT(*) OVER (
48
+ PARTITION BY {','.join(business_keys)}, {','.join(primary_keys)}
49
+ ) AS primary_key_count
42
50
  FROM {tablename}
43
- GROUP BY {','.join(business_keys)}
51
+ ), candidates AS (
52
+ SELECT
53
+ rowid
54
+ , duplicate_number
55
+ , MAX(primary_key_count) OVER (
56
+ PARTITION BY {','.join(business_keys)}
57
+ ) AS max_primary_key_count
58
+ FROM ranked
59
+ )
60
+ DELETE FROM {tablename} WHERE rowid IN (
61
+ SELECT rowid
62
+ FROM candidates
63
+ WHERE duplicate_number > 1 AND max_primary_key_count = 1
44
64
  )
45
65
  """,
46
66
  )
@@ -1,4 +1,4 @@
1
- import argparse, difflib, os, re, shlex, tempfile
1
+ import argparse, difflib, os, shlex, tempfile
2
2
  from collections import defaultdict
3
3
  from concurrent.futures import ThreadPoolExecutor
4
4
  from copy import deepcopy
@@ -395,11 +395,6 @@ def get_fs_duplicates(args) -> list[dict]:
395
395
  return dup_media
396
396
 
397
397
 
398
- def filter_split_files(paths):
399
- pattern = r"\.\d{3,5}\."
400
- return filter(lambda x: not re.search(pattern, x), paths)
401
-
402
-
403
398
  def dedupe_media() -> None:
404
399
  args = parse_args()
405
400
 
@@ -27,13 +27,13 @@ def parse_args():
27
27
 
28
28
  arggroups.paths_or_stdin(parser)
29
29
  args = parser.parse_intermixed_args()
30
- arggroups.args_post(args, parser)
31
30
 
31
+ if not args.filter_names and not args.filter_sizes and not args.filter_durations:
32
+ parser.error("specify at least one filter mode: --filter-names, --filter-sizes, or --filter-durations")
33
+
34
+ arggroups.args_post(args, parser)
32
35
  arggroups.files_post(args)
33
36
  arggroups.similar_files_post(args)
34
- if not args.filter_names and not args.filter_sizes and not args.filter_durations:
35
- print("Nothing to do")
36
- raise NotImplementedError
37
37
 
38
38
  return args
39
39
 
@@ -91,6 +91,7 @@ def group_files_by_parent(args, media) -> list[dict]:
91
91
  for parent, media in list(p_media.items()):
92
92
  d[parent] = {
93
93
  "total": len(media),
94
+ "folders": 0,
94
95
  "duration": sum(m.get("duration") or 0 for m in media if not bool(m.get("time_deleted"))),
95
96
  "median_duration": nums.safe_median(m.get("duration") for m in media if not bool(m.get("time_deleted"))),
96
97
  "size": sum(m.get("size") or 0 for m in media if not bool(m.get("time_deleted"))),
@@ -217,7 +217,8 @@ def gen_src_dest(args, sources: Iterable[str], destination: str, shortcut_allowe
217
217
  log.debug("rglob-file relpath %s", relpath)
218
218
  if args.modify_depth:
219
219
  rel_p = Path(relpath)
220
- parts = rel_p.parent.parts[args.modify_depth]
220
+ parts = rel_p.parent.parts
221
+ parts = parts[args.modify_depth]
221
222
  relpath = os.path.join(*parts, rel_p.name)
222
223
  log.debug("rglob-file modify_depth %s %s", parts, relpath)
223
224
 
@@ -48,12 +48,13 @@ def parse_args():
48
48
 
49
49
  arggroups.paths_or_stdin(parser)
50
50
  args = parser.parse_intermixed_args()
51
- arggroups.args_post(args, parser)
52
51
 
53
52
  if not args.filter_names and not args.filter_counts and not args.filter_sizes and not args.filter_durations:
54
- print("Nothing to do")
55
- raise NotImplementedError
53
+ parser.error(
54
+ "specify at least one filter mode: --filter-names, --filter-counts, --filter-sizes, or --filter-durations"
55
+ )
56
56
 
57
+ arggroups.args_post(args, parser)
57
58
  if args.filter_counts and not any([args.folder_counts, args.file_counts, args.folder_sizes]):
58
59
  args.file_counts = ["+2"]
59
60
 
@@ -61,15 +61,15 @@ def parse_args():
61
61
 
62
62
  parser.set_defaults(fts=False)
63
63
  args, unk = parser.parse_known_intermixed_args()
64
- arggroups.args_post(args, parser, create_db=args.database and args.database.endswith(consts.SQLITE_EXTENSIONS))
65
64
 
66
65
  if unk and not args.profile in (DBType.video, DBType.audio):
67
66
  parser.error(f"unrecognized arguments: {' '.join(unk)}")
68
- args.unk = unk
69
67
 
70
- if not args.profile and not args.print:
71
- log.error("Download profile must be specified. Use one of: --video OR --audio OR --image OR --filesystem")
72
- raise SystemExit(1)
68
+ if args.safe and args.profile == DBType.filesystem:
69
+ parser.error("--safe is supported only with --audio, --video, or --image")
70
+
71
+ arggroups.args_post(args, parser, create_db=args.database and args.database.endswith(consts.SQLITE_EXTENSIONS))
72
+ args.unk = unk
73
73
 
74
74
  arggroups.sql_fs_post(args)
75
75
  arggroups.filter_links_post(args)
@@ -212,6 +212,7 @@ def download(args=None) -> None:
212
212
  local_path = None
213
213
  error = str(excinfo)
214
214
 
215
+ local_paths = [local_path]
215
216
  if local_path and args.process:
216
217
  extension = local_path.rsplit(".", 1)[-1].lower()
217
218
  if extension in consts.AUDIO_ONLY_EXTENSIONS | consts.VIDEO_EXTENSIONS:
@@ -220,23 +221,26 @@ def download(args=None) -> None:
220
221
  result = process_image.process_path(args, local_path)
221
222
 
222
223
  if result is not None:
223
- local_path = str(result)
224
+ local_paths = process_ffmpeg.result_paths(result)
224
225
 
225
226
  is_not_found = error is not None and "HTTPNotFound" in error
226
227
  if error is not None and "HTTPNotFound" not in error:
227
228
  any_error = True
228
229
 
229
- db_media.download_add(
230
- args,
231
- webpath=original_path,
232
- info=m,
233
- local_path=local_path,
234
- error=error,
235
- mark_deleted=is_not_found,
236
- delete_webpath_entry=(
237
- not any_error if i == len(dl_paths) - 1 else False
238
- ), # only check after last download link was saved
239
- )
230
+ for output_index, local_path in enumerate(local_paths):
231
+ db_media.download_add(
232
+ args,
233
+ webpath=original_path,
234
+ info=m,
235
+ local_path=local_path,
236
+ error=error,
237
+ mark_deleted=is_not_found,
238
+ delete_webpath_entry=(
239
+ not any_error
240
+ if i == len(dl_paths) - 1 and output_index == len(local_paths) - 1
241
+ else False
242
+ ), # only check after last downloaded output was saved
243
+ )
240
244
  else:
241
245
  raise NotImplementedError
242
246
 
@@ -44,13 +44,14 @@ def download_status() -> None:
44
44
 
45
45
  for m in media:
46
46
  extractor_key = m.get("extractor_key", "Playlist-less media")
47
+ time_downloaded = m.get("time_downloaded") or 0
47
48
 
48
49
  if "download_attempts" in m and (m["download_attempts"] or 0) >= args.download_retries:
49
50
  extractor_stats[extractor_key]["retries_exceeded"] += 1
50
- elif (m.get("time_downloaded") or 0) > 0 or (
51
+ elif time_downloaded > 0 or (
51
52
  not m["path"].startswith("http") and (m.get("webpath") or "").startswith("http")
52
53
  ):
53
- if (m["time_downloaded"] + retry_delay) >= consts.APPLICATION_START:
54
+ if (time_downloaded + retry_delay) >= consts.APPLICATION_START:
54
55
  extractor_stats[extractor_key]["downloaded_recently"] += 1
55
56
  elif m["path"].startswith("http"):
56
57
  if "time_modified" in m: