bookery-cli 2026.10.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. bookery_cli-2026.10.1/PKG-INFO +543 -0
  2. bookery_cli-2026.10.1/README.md +512 -0
  3. bookery_cli-2026.10.1/pyproject.toml +90 -0
  4. bookery_cli-2026.10.1/pyproject.toml.orig +82 -0
  5. bookery_cli-2026.10.1/src/bookery/__init__.py +6 -0
  6. bookery_cli-2026.10.1/src/bookery/__main__.py +7 -0
  7. bookery_cli-2026.10.1/src/bookery/cli/__init__.py +95 -0
  8. bookery_cli-2026.10.1/src/bookery/cli/_dispatch.py +42 -0
  9. bookery_cli-2026.10.1/src/bookery/cli/_match_helpers.py +201 -0
  10. bookery_cli-2026.10.1/src/bookery/cli/_pdf_support.py +21 -0
  11. bookery_cli-2026.10.1/src/bookery/cli/commands/__init__.py +2 -0
  12. bookery_cli-2026.10.1/src/bookery/cli/commands/add_cmd.py +482 -0
  13. bookery_cli-2026.10.1/src/bookery/cli/commands/authors_cmd.py +426 -0
  14. bookery_cli-2026.10.1/src/bookery/cli/commands/collection_cmd.py +506 -0
  15. bookery_cli-2026.10.1/src/bookery/cli/commands/convert_cmd.py +161 -0
  16. bookery_cli-2026.10.1/src/bookery/cli/commands/genre_cmd.py +179 -0
  17. bookery_cli-2026.10.1/src/bookery/cli/commands/info_cmd.py +392 -0
  18. bookery_cli-2026.10.1/src/bookery/cli/commands/inventory_cmd.py +130 -0
  19. bookery_cli-2026.10.1/src/bookery/cli/commands/ls_cmd.py +137 -0
  20. bookery_cli-2026.10.1/src/bookery/cli/commands/mark_cmd.py +159 -0
  21. bookery_cli-2026.10.1/src/bookery/cli/commands/match_cmd.py +213 -0
  22. bookery_cli-2026.10.1/src/bookery/cli/commands/prune_cmd.py +172 -0
  23. bookery_cli-2026.10.1/src/bookery/cli/commands/rematch_cmd.py +300 -0
  24. bookery_cli-2026.10.1/src/bookery/cli/commands/remove_cmd.py +175 -0
  25. bookery_cli-2026.10.1/src/bookery/cli/commands/reveal_cmd.py +103 -0
  26. bookery_cli-2026.10.1/src/bookery/cli/commands/search_cmd.py +49 -0
  27. bookery_cli-2026.10.1/src/bookery/cli/commands/series_cmd.py +438 -0
  28. bookery_cli-2026.10.1/src/bookery/cli/commands/serve_cmd.py +44 -0
  29. bookery_cli-2026.10.1/src/bookery/cli/commands/sync_cmd.py +279 -0
  30. bookery_cli-2026.10.1/src/bookery/cli/commands/tag_cmd.py +93 -0
  31. bookery_cli-2026.10.1/src/bookery/cli/commands/vault_export_cmd.py +335 -0
  32. bookery_cli-2026.10.1/src/bookery/cli/commands/verify_cmd.py +57 -0
  33. bookery_cli-2026.10.1/src/bookery/cli/deprecation.py +207 -0
  34. bookery_cli-2026.10.1/src/bookery/cli/options.py +125 -0
  35. bookery_cli-2026.10.1/src/bookery/cli/review.py +227 -0
  36. bookery_cli-2026.10.1/src/bookery/collections/__init__.py +29 -0
  37. bookery_cli-2026.10.1/src/bookery/collections/lucene_compose.py +42 -0
  38. bookery_cli-2026.10.1/src/bookery/collections/query.py +331 -0
  39. bookery_cli-2026.10.1/src/bookery/convert/__init__.py +2 -0
  40. bookery_cli-2026.10.1/src/bookery/convert/assemble.py +117 -0
  41. bookery_cli-2026.10.1/src/bookery/convert/assets/kobo.css +48 -0
  42. bookery_cli-2026.10.1/src/bookery/convert/cache.py +47 -0
  43. bookery_cli-2026.10.1/src/bookery/convert/errors.py +42 -0
  44. bookery_cli-2026.10.1/src/bookery/convert/extract.py +128 -0
  45. bookery_cli-2026.10.1/src/bookery/convert/llm.py +145 -0
  46. bookery_cli-2026.10.1/src/bookery/convert/preflight.py +51 -0
  47. bookery_cli-2026.10.1/src/bookery/convert/types.py +52 -0
  48. bookery_cli-2026.10.1/src/bookery/core/__init__.py +0 -0
  49. bookery_cli-2026.10.1/src/bookery/core/book_lookup.py +82 -0
  50. bookery_cli-2026.10.1/src/bookery/core/config.py +279 -0
  51. bookery_cli-2026.10.1/src/bookery/core/converter.py +175 -0
  52. bookery_cli-2026.10.1/src/bookery/core/coverfetch.py +66 -0
  53. bookery_cli-2026.10.1/src/bookery/core/dedup.py +118 -0
  54. bookery_cli-2026.10.1/src/bookery/core/enrichment.py +140 -0
  55. bookery_cli-2026.10.1/src/bookery/core/filecopy.py +24 -0
  56. bookery_cli-2026.10.1/src/bookery/core/genre_applier.py +121 -0
  57. bookery_cli-2026.10.1/src/bookery/core/importer.py +257 -0
  58. bookery_cli-2026.10.1/src/bookery/core/pathformat.py +153 -0
  59. bookery_cli-2026.10.1/src/bookery/core/pdf_converter.py +68 -0
  60. bookery_cli-2026.10.1/src/bookery/core/pipeline.py +371 -0
  61. bookery_cli-2026.10.1/src/bookery/core/prune.py +90 -0
  62. bookery_cli-2026.10.1/src/bookery/core/remove.py +195 -0
  63. bookery_cli-2026.10.1/src/bookery/core/scanner.py +181 -0
  64. bookery_cli-2026.10.1/src/bookery/core/text_sort.py +57 -0
  65. bookery_cli-2026.10.1/src/bookery/core/vault/__init__.py +2 -0
  66. bookery_cli-2026.10.1/src/bookery/core/vault/assemble.py +245 -0
  67. bookery_cli-2026.10.1/src/bookery/core/vault/epub.py +117 -0
  68. bookery_cli-2026.10.1/src/bookery/core/vault/frontmatter.py +60 -0
  69. bookery_cli-2026.10.1/src/bookery/core/vault/image.py +64 -0
  70. bookery_cli-2026.10.1/src/bookery/core/vault/index.py +49 -0
  71. bookery_cli-2026.10.1/src/bookery/core/vault/note.py +66 -0
  72. bookery_cli-2026.10.1/src/bookery/core/vault/walker.py +85 -0
  73. bookery_cli-2026.10.1/src/bookery/core/vault/wikilink.py +31 -0
  74. bookery_cli-2026.10.1/src/bookery/core/verifier.py +69 -0
  75. bookery_cli-2026.10.1/src/bookery/db/__init__.py +16 -0
  76. bookery_cli-2026.10.1/src/bookery/db/catalog.py +2183 -0
  77. bookery_cli-2026.10.1/src/bookery/db/connection.py +103 -0
  78. bookery_cli-2026.10.1/src/bookery/db/hashing.py +32 -0
  79. bookery_cli-2026.10.1/src/bookery/db/mapping.py +158 -0
  80. bookery_cli-2026.10.1/src/bookery/db/schema.py +356 -0
  81. bookery_cli-2026.10.1/src/bookery/db/status.py +74 -0
  82. bookery_cli-2026.10.1/src/bookery/device/__init__.py +0 -0
  83. bookery_cli-2026.10.1/src/bookery/device/errors.py +32 -0
  84. bookery_cli-2026.10.1/src/bookery/device/kepub_cache.py +159 -0
  85. bookery_cli-2026.10.1/src/bookery/device/kepubify.py +60 -0
  86. bookery_cli-2026.10.1/src/bookery/device/kobo.py +904 -0
  87. bookery_cli-2026.10.1/src/bookery/device/kobo_backup.py +89 -0
  88. bookery_cli-2026.10.1/src/bookery/device/kobo_reader.py +208 -0
  89. bookery_cli-2026.10.1/src/bookery/device/kobo_writer.py +482 -0
  90. bookery_cli-2026.10.1/src/bookery/formats/__init__.py +2 -0
  91. bookery_cli-2026.10.1/src/bookery/formats/epub.py +593 -0
  92. bookery_cli-2026.10.1/src/bookery/formats/mobi.py +534 -0
  93. bookery_cli-2026.10.1/src/bookery/metadata/__init__.py +15 -0
  94. bookery_cli-2026.10.1/src/bookery/metadata/author_names.py +97 -0
  95. bookery_cli-2026.10.1/src/bookery/metadata/cache.py +83 -0
  96. bookery_cli-2026.10.1/src/bookery/metadata/candidate.py +25 -0
  97. bookery_cli-2026.10.1/src/bookery/metadata/consensus.py +319 -0
  98. bookery_cli-2026.10.1/src/bookery/metadata/genres.py +320 -0
  99. bookery_cli-2026.10.1/src/bookery/metadata/googlebooks.py +288 -0
  100. bookery_cli-2026.10.1/src/bookery/metadata/hardcover.py +249 -0
  101. bookery_cli-2026.10.1/src/bookery/metadata/http.py +231 -0
  102. bookery_cli-2026.10.1/src/bookery/metadata/normalizer.py +323 -0
  103. bookery_cli-2026.10.1/src/bookery/metadata/openlibrary.py +373 -0
  104. bookery_cli-2026.10.1/src/bookery/metadata/openlibrary_parser.py +268 -0
  105. bookery_cli-2026.10.1/src/bookery/metadata/provider.py +26 -0
  106. bookery_cli-2026.10.1/src/bookery/metadata/registry.py +52 -0
  107. bookery_cli-2026.10.1/src/bookery/metadata/scoring.py +106 -0
  108. bookery_cli-2026.10.1/src/bookery/metadata/series_heuristic.py +66 -0
  109. bookery_cli-2026.10.1/src/bookery/metadata/title_correspondence.py +125 -0
  110. bookery_cli-2026.10.1/src/bookery/metadata/types.py +48 -0
  111. bookery_cli-2026.10.1/src/bookery/plugins/__init__.py +0 -0
  112. bookery_cli-2026.10.1/src/bookery/py.typed +0 -0
  113. bookery_cli-2026.10.1/src/bookery/util/__init__.py +0 -0
  114. bookery_cli-2026.10.1/src/bookery/util/file_manager.py +114 -0
  115. bookery_cli-2026.10.1/src/bookery/util/text.py +79 -0
  116. bookery_cli-2026.10.1/src/bookery/web/__init__.py +46 -0
  117. bookery_cli-2026.10.1/src/bookery/web/browse.py +258 -0
  118. bookery_cli-2026.10.1/src/bookery/web/candidate_payload.py +63 -0
  119. bookery_cli-2026.10.1/src/bookery/web/covers.py +117 -0
  120. bookery_cli-2026.10.1/src/bookery/web/diff.py +124 -0
  121. bookery_cli-2026.10.1/src/bookery/web/routes.py +1768 -0
  122. bookery_cli-2026.10.1/src/bookery/web/static/fonts/Fraunces-OFL.txt +93 -0
  123. bookery_cli-2026.10.1/src/bookery/web/static/fonts/fraunces-latin-wght-italic.woff2 +0 -0
  124. bookery_cli-2026.10.1/src/bookery/web/static/fonts/fraunces-latin-wght-normal.woff2 +0 -0
  125. bookery_cli-2026.10.1/src/bookery/web/static/htmx.min.js +1 -0
  126. bookery_cli-2026.10.1/src/bookery/web/static/pico.min.css +4 -0
  127. bookery_cli-2026.10.1/src/bookery/web/static/style.css +1327 -0
  128. bookery_cli-2026.10.1/src/bookery/web/templates/_book_card.html +36 -0
  129. bookery_cli-2026.10.1/src/bookery/web/templates/_book_list.html +174 -0
  130. bookery_cli-2026.10.1/src/bookery/web/templates/_book_subhead.html +7 -0
  131. bookery_cli-2026.10.1/src/bookery/web/templates/_collection_delete_confirm.html +23 -0
  132. bookery_cli-2026.10.1/src/bookery/web/templates/_collection_detail.html +52 -0
  133. bookery_cli-2026.10.1/src/bookery/web/templates/_collection_form.html +105 -0
  134. bookery_cli-2026.10.1/src/bookery/web/templates/_collection_preview.html +21 -0
  135. bookery_cli-2026.10.1/src/bookery/web/templates/_collection_query_error.html +3 -0
  136. bookery_cli-2026.10.1/src/bookery/web/templates/_collection_query_field.html +5 -0
  137. bookery_cli-2026.10.1/src/bookery/web/templates/_collections_list.html +18 -0
  138. bookery_cli-2026.10.1/src/bookery/web/templates/_columns_menu.html +35 -0
  139. bookery_cli-2026.10.1/src/bookery/web/templates/_delete_confirm.html +60 -0
  140. bookery_cli-2026.10.1/src/bookery/web/templates/_detail.html +11 -0
  141. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_classification.html +37 -0
  142. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_collections.html +55 -0
  143. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_description.html +8 -0
  144. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_file.html +21 -0
  145. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_header.html +37 -0
  146. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_identity.html +24 -0
  147. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_publication.html +11 -0
  148. bookery_cli-2026.10.1/src/bookery/web/templates/_detail_reading.html +40 -0
  149. bookery_cli-2026.10.1/src/bookery/web/templates/_edit_form.html +80 -0
  150. bookery_cli-2026.10.1/src/bookery/web/templates/_enrich_candidate_error.html +33 -0
  151. bookery_cli-2026.10.1/src/bookery/web/templates/_enrich_candidate_row.html +28 -0
  152. bookery_cli-2026.10.1/src/bookery/web/templates/_enrich_candidates.html +20 -0
  153. bookery_cli-2026.10.1/src/bookery/web/templates/_enrich_diff.html +134 -0
  154. bookery_cli-2026.10.1/src/bookery/web/templates/_enrich_search.html +58 -0
  155. bookery_cli-2026.10.1/src/bookery/web/templates/_field_diff_row.html +67 -0
  156. bookery_cli-2026.10.1/src/bookery/web/templates/_filter_chips.html +37 -0
  157. bookery_cli-2026.10.1/src/bookery/web/templates/_provenance.html +21 -0
  158. bookery_cli-2026.10.1/src/bookery/web/templates/_status_filter.html +38 -0
  159. bookery_cli-2026.10.1/src/bookery/web/templates/_table.html +22 -0
  160. bookery_cli-2026.10.1/src/bookery/web/templates/base.html +47 -0
  161. bookery_cli-2026.10.1/src/bookery/web/templates/collection_detail.html +15 -0
  162. bookery_cli-2026.10.1/src/bookery/web/templates/collection_form.html +19 -0
  163. bookery_cli-2026.10.1/src/bookery/web/templates/collections_list.html +24 -0
  164. bookery_cli-2026.10.1/src/bookery/web/templates/detail.html +15 -0
  165. bookery_cli-2026.10.1/src/bookery/web/templates/edit.html +17 -0
  166. bookery_cli-2026.10.1/src/bookery/web/templates/enrich.html +17 -0
  167. bookery_cli-2026.10.1/src/bookery/web/templates/enrich_candidate.html +21 -0
  168. bookery_cli-2026.10.1/src/bookery/web/templates/list.html +46 -0
  169. bookery_cli-2026.10.1/src/bookery/web/templates/series_detail.html +28 -0
  170. bookery_cli-2026.10.1/src/bookery/web/templates/series_list.html +41 -0
@@ -0,0 +1,543 @@
1
+ Metadata-Version: 2.4
2
+ Name: bookery-cli
3
+ Version: 2026.10.1
4
+ Summary: CLI-first ebook library manager with Kobo sync, inspired by beets and Calibre
5
+ Keywords: ebook,epub,kobo,calibre,library,metadata
6
+ Author: Joe Cotellese
7
+ Author-email: Joe Cotellese <bookery@cotellese.me>
8
+ License-Expression: MIT
9
+ Classifier: Environment :: Console
10
+ Classifier: Environment :: Web Environment
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Requires-Dist: click>=8.1
14
+ Requires-Dist: ebooklib>=0.18
15
+ Requires-Dist: flask>=3.0
16
+ Requires-Dist: httpx>=0.27
17
+ Requires-Dist: luqum>=0.13
18
+ Requires-Dist: mobi>=0.4
19
+ Requires-Dist: openai>=1.0
20
+ Requires-Dist: pdfplumber>=0.11
21
+ Requires-Dist: pydantic>=2.0
22
+ Requires-Dist: pypdf>=5.0
23
+ Requires-Dist: pyyaml>=6.0
24
+ Requires-Dist: rich>=13.0
25
+ Requires-Dist: wordninja>=2.0
26
+ Requires-Python: >=3.12
27
+ Project-URL: Homepage, https://github.com/JoeCotellese/bookery
28
+ Project-URL: Repository, https://github.com/JoeCotellese/bookery
29
+ Project-URL: Issues, https://github.com/JoeCotellese/bookery/issues
30
+ Description-Content-Type: text/markdown
31
+
32
+ # Bookery
33
+
34
+ A CLI-first ebook library manager inspired by [beets](https://beets.io/) and [Calibre](https://calibre-ebook.com/). Fix your metadata, organize your library, sync to your Kobo — all from the terminal.
35
+
36
+ Bookery takes the "metadata-first" approach that made beets great for music and applies it to ebooks. It matches your EPUBs against online sources, lets you review and correct metadata interactively, and keeps your originals untouched.
37
+
38
+ ## Status
39
+
40
+ **Active development** — daily-driver usable. EPUB metadata extraction, Kindle (MOBI/AZW/AZW3) and PDF-to-EPUB conversion, multi-provider matching (Open Library + Google Books) with consensus merging, a SQLite catalog with per-field provenance, non-destructive write-back, Kobo device sync, a local web UI for browsing/editing, and Obsidian vault-export are all working. Plugin architecture is the main item still on the roadmap.
41
+
42
+ See [docs/roadmap.md](docs/roadmap.md) for the full plan.
43
+
44
+ ## ⚠️ Back up your Kobo first
45
+
46
+ Bookery's library write-back is non-destructive, but **Kobo sync rewrites EPUB
47
+ files and the device database** (`.kobo/KoboReader.sqlite`). **Back up both the
48
+ files on your Kobo and that database before syncing, and keep those backups.**
49
+ I daily-drive this tool, but you run it at your own risk — there is no warranty,
50
+ and any data loss is on you, not the project. PRs to make it safer are always
51
+ welcome.
52
+
53
+ ## Features
54
+
55
+ - **EPUB metadata extraction** — reads title, author, ISBN, language, publisher, description, cover, and identifiers from any EPUB
56
+ - **Kindle-to-EPUB conversion** — converts DRM-free MOBI, AZW, and AZW3 (KF8) files to EPUB, preserving metadata, images, cover art, and chapter structure (via NCX TOC). `add`, `convert`, and `info` all accept these suffixes in any case. DRM-protected files are rejected with a message naming the file.
57
+ - **PDF-to-EPUB conversion** — `bookery add` detects text-based PDFs, extracts their structure with pdfplumber + a local LLM (LM Studio), and produces a reflowable EPUB. Scanned PDFs are refused (OCR not yet supported).
58
+ - **Kobo sync** — `bookery sync kobo` walks the catalog, converts each EPUB to `.kepub.epub` via `kepubify`, and copies the result to a mounted Kobo. The library itself stays format-canonical (EPUB only); kepub is generated on demand at sync time and cached so re-syncs are free when nothing has changed. Books with series metadata group by series on device: sync writes `Series`/`SeriesNumber` into the Kobo database directly (the same mechanism Calibre's device driver uses — current firmware ignores series metadata inside sideloaded EPUBs). Newly copied books group on the sync after the device has indexed them.
59
+ - **Collections** — group books into named lists, either static (hand-picked) or rule-based (membership derived live from a query like `genre:"Science Fiction"` or `series:Dune`, so it stays current as the library grows). See `bookery collections`.
60
+ - **Collection shelves on device** — the same `bookery sync kobo` mirrors each collection to a Kobo shelf (`Shelf`/`ShelfContent`). Bookery owns only shelves whose `InternalName` is `bookery-<collection_id>`; a user-created shelf that shares a name is skipped, never overwritten. Unchanged shelves are skipped on re-sync (membership hash), and a shelf is removed once its collection is deleted. `bookery collections show <id> --sync-status` reports per-device shelf state.
61
+ - **Multi-provider metadata matching** — Open Library and Google Books in a consensus merger that prefers values agreed on by ≥2 providers and falls back to a priority order otherwise. ISBN-10/13 lookups are normalized and provider responses are cached.
62
+ - **Per-field provenance & locking** — every cataloged field records which provider supplied it and when. User edits are stamped as `user` and locked against `rematch`; individual fields can be locked/unlocked explicitly with `bookery info --lock` / `--unlock`.
63
+ - **Interactive review** — presents candidates in a Rich table, lets you accept, compare details, look up by URL, or skip
64
+ - **Smart normalization** — splits mangled filenames like `SteveBerry-TheTemplarLegacy` into clean search queries, detects embedded author names
65
+ - **SQLite catalog** — imports books into a local database for querying, tagging, and integrity checks
66
+ - **Web UI** — `bookery serve` launches a local browser UI for paginated/sortable browsing of the catalog, filter chips, cover thumbnails, a responsive mobile card layout, per-book detail/edit pages, collection create/edit (including raw rule strings, plus an append-only query builder for composing clauses without memorizing the syntax) with inline validation and non-persisting query match preview, search-active-providers, and "apply candidate metadata with diff" rematch flow.
67
+ - **Genre management** — `bookery genre` auto-maps raw provider subjects to a canonical genre vocabulary, with `assign` / `apply` / `auto` / `stats` / `unmatched` for curation.
68
+ - **Obsidian vault-export** — `bookery vault-export` turns an Obsidian vault into a single EPUB with a hierarchical folder/note TOC, A-Z buckets within each folder, leading-article-aware filing ("The Loop" files under L), resolved `[[wiki-links]]` and `![[image]]` embeds, and an optional tag index. Can auto-catalog the export so it ships on the next Kobo sync.
69
+ - **Non-destructive** — metadata writes always go to a copy; original file contents are never modified. `add` copies each source into your library by default (`--move` deletes the source after a successful catalog insert)
70
+
71
+ ## Installation
72
+
73
+ Requires Python 3.12+.
74
+
75
+ ### Install it (to use bookery)
76
+
77
+ Bookery is published on PyPI as `bookery-cli` (the `bookery` name was taken).
78
+ The command it installs is still `bookery`.
79
+
80
+ ```bash
81
+ uv tool install bookery-cli
82
+ # or
83
+ pipx install bookery-cli
84
+ ```
85
+
86
+ `uv tool install` puts `bookery` in its own isolated environment and drops a
87
+ `bookery` command on your `PATH` (in `~/.local/bin`) — no virtualenv to create
88
+ or activate. Just run `bookery` from anywhere. If the command isn't found after
89
+ install, run `uv tool update-shell` once and restart your shell.
90
+
91
+ Upgrade later with `uv tool upgrade bookery-cli`; remove with `uv tool uninstall bookery-cli`.
92
+ If you installed from GitHub before the PyPI release, run `uv tool uninstall bookery` first.
93
+
94
+ To run unreleased code from `main`, install straight from GitHub:
95
+
96
+ ```bash
97
+ uv tool install git+https://github.com/joecotellese/bookery.git
98
+ ```
99
+
100
+ ### Develop it (to hack on bookery)
101
+
102
+ ```bash
103
+ git clone https://github.com/joecotellese/bookery.git
104
+ cd bookery
105
+ uv sync
106
+ ```
107
+
108
+ `uv sync` installs into the project's `.venv`. Run the CLI with
109
+ `uv run bookery ...`, or activate the venv first
110
+ (`source .venv/bin/activate`) and call `bookery` directly.
111
+
112
+ ### Optional: PDF conversion
113
+
114
+ The PDF path in `bookery add` routes the document
115
+ through a single semantic LLM call that reassembles articles/chapters
116
+ into a clean EPUB. You'll need an OpenAI-compatible endpoint with a
117
+ model strong enough to return structured JSON.
118
+
119
+ **Local (default)** — [LM Studio](https://lmstudio.ai) with a long-
120
+ context instruct model (known-good: **Qwen 2.5 7B Instruct 1M**).
121
+ Load the model with ≥16k context, enable the local server, then point
122
+ bookery at it via `~/.bookery/config.toml`:
123
+
124
+ ```toml
125
+ [convert.semantic]
126
+ provider = "lm-studio"
127
+ model = "qwen2.5-7b-instruct-1m"
128
+ base_url = "http://localhost:1234/v1"
129
+ api_key_env = "" # empty for local; no key needed
130
+ prompt_version = 1
131
+ llm_max_retries = 2
132
+ ```
133
+
134
+ **Cloud** — swap `provider`, `model`, `base_url`, and point
135
+ `api_key_env` at the env var holding your key. Never write the key
136
+ into `config.toml`:
137
+
138
+ ```toml
139
+ [convert.semantic]
140
+ provider = "openai"
141
+ model = "gpt-5.4-nano"
142
+ base_url = "https://api.openai.com/v1"
143
+ api_key_env = "OPENAI_API_KEY"
144
+ prompt_version = 1
145
+ ```
146
+
147
+ See [`docs/config.example.toml`](docs/config.example.toml) for a fully
148
+ annotated config with LM Studio, Moonshot/Kimi, and OpenAI examples.
149
+
150
+ Semantic responses are cached under `~/.bookery/data/convert_cache.db`
151
+ so re-runs on the same PDF skip the LLM. Safe to delete at any time;
152
+ bumping `prompt_version` invalidates only stale entries.
153
+
154
+ ## Quick Start
155
+
156
+ ```bash
157
+ # Inspect a single EPUB (loose file on disk)
158
+ bookery info ~/Books/some-book.epub
159
+
160
+ # Scan a directory and report format coverage
161
+ bookery inventory ~/Books/
162
+
163
+ # Convert Kindle files (.mobi, .azw, .azw3) to EPUB
164
+ bookery convert ~/Books/ -o ~/Books-epub/
165
+
166
+ # Match EPUBs against Open Library and write corrected copies
167
+ bookery match ~/Books/ -o ~/Books-fixed/
168
+
169
+ # Auto-accept high-confidence matches (no prompts)
170
+ bookery match ~/Books/ -o ~/Books-fixed/ --yes
171
+
172
+ # Add a single EPUB to the catalog (copies into ~/.library/ by default)
173
+ bookery --db ~/library.db add ~/Books/dune.epub
174
+
175
+ # Add an entire directory of EPUBs (recursive)
176
+ bookery --db ~/library.db add ~/Books/
177
+
178
+ # Add and remove the sources after they land in the library
179
+ bookery --db ~/library.db add ~/Downloads/ --move
180
+
181
+ # Search and browse the catalog
182
+ bookery search "martian"
183
+ bookery ls --tag fiction
184
+ bookery info 42
185
+ ```
186
+
187
+ ### Common options
188
+
189
+ - **`--db PATH`** is a top-level option; put it before the subcommand so every command picks it up: `bookery --db ~/library.db ls`. Subcommand-level `--db` still works and overrides the global.
190
+ - **`-y/--yes`** auto-accepts high-confidence matches without prompting (valid on `add`, `match`, `rematch`, `convert`). `-q/--quiet` is a deprecated alias.
191
+ - **`-t/--threshold`** sets the confidence cutoff for auto-accept. The default comes from `[matching].auto_accept_threshold` in `~/.bookery/config.toml` (default `0.8`):
192
+
193
+ ```toml
194
+ [matching]
195
+ auto_accept_threshold = 0.85
196
+ cache_ttl_days = 30 # metadata response cache TTL (default 30)
197
+ providers = ["openlibrary", "googlebooks", "hardcover"] # priority order; default ["openlibrary"]
198
+ min_request_interval = 0.1 # seconds between provider HTTP requests (default 0.1)
199
+ fetch_covers = true # download & embed candidate covers during matching (default true)
200
+ ```
201
+
202
+ - **`--no-cache`** on `match`/`rematch` bypasses the on-disk metadata response cache and forces fresh provider lookups. Cached responses live at `{data_dir}/metadata_cache.db` and expire after `[matching].cache_ttl_days`.
203
+ - **`--no-covers`** on `add`/`match`/`rematch` skips downloading the accepted candidate's cover image (useful offline or for throttled batch runs). The default comes from `[matching].fetch_covers` (default `true`). When enabled, the cover is embedded in the same write as the text metadata; a failed download is non-fatal — the text still applies and one warning line is printed.
204
+ - **`[matching].providers`** selects and orders metadata sources. With a single entry the named provider is used directly; with two or more, results are merged by a consensus step that prefers values agreed on by ≥2 providers and falls back to the priority order otherwise. Supported: `openlibrary`, `googlebooks`, `hardcover`. Exception: `series`, `series_index`, `rating`, and `ratings_count` prefer Hardcover's value regardless of provider order (it is the only source with reliable series positions); ≥2-provider agreement still wins the vote, but Hardcover's spelling of an agreed series name is used. Title-pattern heuristics fill series data that providers leave empty and are stamped source=`heuristic`, so they never outrank provider or user values.
205
+ - **Google Books API key** — set the `GOOGLE_BOOKS_API_KEY` environment variable to authenticate Google Books requests. Without it, requests use the shared anonymous per-IP quota, which a full-library `rematch` exhausts quickly (HTTP 429). Create a free key in the [Google Cloud Console](https://console.cloud.google.com/): create/select a project, enable the **Books API**, then **Credentials → Create credentials → API key**. Then `export GOOGLE_BOOKS_API_KEY=AIza...`. The key is read from the environment only — never written to config.
206
+ - **Hardcover API key** — set the `HARDCOVER_API_KEY` environment variable to enable the `hardcover` provider. Hardcover is the strongest source for series name/position and community ratings, but it requires a token: create a free account at [hardcover.app](https://hardcover.app), then copy the token from **Account Settings → Hardcover API**. Then `export HARDCOVER_API_KEY=...`. The key is read from the environment only — never written to config. Without it, a configured `hardcover` provider logs one warning and returns no results. Rate limit is 60 requests/minute; tokens expire January 1st each year.
207
+ - **`[matching].min_request_interval`** is the minimum seconds between provider HTTP requests (default `0.1`). Raise it to throttle a bulk `rematch` below a provider's rate limit; on a `429` the client also honors the response's `Retry-After` header (capped at 60s) before retrying.
208
+ - **Per-field provenance** is recorded for every cataloged book in the `book_field_provenance` table. Use `bookery info <id> --provenance` to see which source supplied each field and when it was fetched. Use `bookery info <id> --set field=value` to hand-edit a value (it's stamped as `user` and locked against overwrite), and `--lock field` / `--unlock field` to gate fields against `rematch`.
209
+
210
+ ## Commands
211
+
212
+ ### Metadata & Matching
213
+
214
+ | Command | Description |
215
+ |---------|-------------|
216
+ | `match <path> -o <dir>` | Match metadata for loose EPUB files (not yet in the catalog) and write corrected copies |
217
+ | `rematch [book_id]` | Re-run matching on cataloged books and update the database |
218
+ | `series ls` | List series in the catalog with book counts and missing-position gaps |
219
+ | `series backfill [book_id\|--all\|--tag]` | Fill missing series/series_index from providers (title-pattern heuristics as fallback) and rewrite library EPUBs whose series meta is stale (`--dry-run` to preview) |
220
+
221
+ A provider's series is only taken from a candidate whose title actually
222
+ corresponds to the book being backfilled. A match confidence is computed across
223
+ title, author, ISBN and language, so a book whose own title carries little
224
+ searchable signal (`Book 17 - Remnant`) could clear the threshold against an
225
+ entirely unrelated book and inherit *its* series. Books matched by ISBN skip
226
+ this check — the ISBN already settles identity. When a series is declined, the
227
+ run says which candidate was rejected and why rather than reporting a bare "no
228
+ series found". A series position that merely repeats a `Book N` prefix from the
229
+ book's own title is dropped as well: a missing position is better than a wrong
230
+ one.
231
+
232
+ ### Conversion
233
+
234
+ | Command | Description |
235
+ |---------|-------------|
236
+ | `convert <path> -o <dir>` | Convert Kindle files (MOBI/AZW/AZW3, DRM-free) to EPUB format (supports `--match` to chain into matching). Exits 1 if any file fails. |
237
+ | `vault-export --vault <path> -o <file>` | Export an Obsidian vault to a single EPUB with clickable TOC, resolved wiki-links, and an optional tag index. Requires [pandoc](https://pandoc.org). |
238
+
239
+ ### Library Catalog
240
+
241
+ | Command | Description |
242
+ |---------|-------------|
243
+ | `add <path>` | Add a single EPUB/PDF/Kindle file (MOBI, AZW, AZW3) or a directory of EPUBs to the library (copies into `library_root`, catalogs). Files default to `--match`; directories default to `--no-match`. Supports `--move` (ignored for PDF and Kindle input), `--convert` (also picks up Kindle files in a directory), `--force-duplicates`, `-o/--output-dir`. `import` is a deprecated alias. |
244
+ | `remove <id>...` | Delete one or more books from the catalog and disk (`-y` skips prompt; `--keep-file` keeps the file) |
245
+ | `prune` | Remove catalog rows whose underlying files are missing |
246
+ | `ls` | List all books in the catalog (filter with `--series` or `--tag`) |
247
+ | `info <id-or-path>` | Show metadata for a cataloged book by ID, or for a loose EPUB on disk. Catalog mode supports `--provenance`, `--set field=value`, `--lock field`, `--unlock field`. `inspect` is a deprecated alias. |
248
+ | `search <query>` | Search the catalog by title, author, or description |
249
+ | `inventory <path>` | Scan a directory tree and report ebook format coverage |
250
+ | `reveal <query>` | Open the on-disk folder for a book (`--print` to print the path instead). `folder` is a deprecated alias. |
251
+
252
+ ### Organization
253
+
254
+ | Command | Description |
255
+ |---------|-------------|
256
+ | `tag add <id> <tag>` | Add a tag to a book |
257
+ | `tag rm <id> <tag>` | Remove a tag from a book |
258
+ | `tag ls` | List all tags with book counts |
259
+ | `genre assign <id> <genre>` | Assign a canonical genre to a book |
260
+ | `genre auto-assign` | Auto-assign genres from subjects for cataloged books |
261
+ | `genre auto` | Auto-map provider subjects to canonical genres across the catalog |
262
+ | `genre ls` | List all canonical genres with book counts |
263
+ | `genre stats` | Show the most common subjects that don't map to a canonical genre |
264
+ | `genre unmatched` | Show books with subjects but no genre assigned |
265
+ | `mark finished <id>` | Mark a book as finished (also `mark reading <id>`, `mark unread <id>`; supports `--bulk-from FILE`) |
266
+ | `authors list` | List authors with book counts (`--duplicates` shows only multi-spelling clusters; `--needs-review` lists malformed names for manual merge) |
267
+ | `authors normalize` | Reorder `Surname, Given` → `Given Surname` (dry-run by default; `--apply` to write; `--include-reversed` for names with no twin) |
268
+ | `authors merge <forms…> --into <canonical>` | Merge arbitrary spellings into one canonical name (dry-run by default; `--apply` to write) |
269
+ | `authors fix-sort` | Backfill `opf:file-as` so devices sort authors by surname (dry-run by default; `--apply` to write) |
270
+ | `verify` | Check for missing or changed files (supports `--check-hash`) |
271
+
272
+ ### Collections
273
+
274
+ Collections group books into named lists. A collection is either **static**
275
+ (hand-picked membership) or **rule-based** (membership derived live from a
276
+ query). The two are mutually exclusive — a rule-based collection holds no
277
+ hand-picked rows, and its members are recomputed on every read, so it stays
278
+ current automatically as the library changes.
279
+
280
+ | Command | Description |
281
+ |---------|-------------|
282
+ | `collections create <name>` | Create a static collection (`-d/--description` optional) |
283
+ | `collections create <name> --query '<rule>'` | Create a rule-based collection, e.g. `--query 'genre:"Science Fiction"'` |
284
+ | `collections ls` | List collections with live book counts |
285
+ | `collections show <id>` | Show a collection's books; rule-based collections also show the rule and live match count (`--sync-status` for per-device shelf state) |
286
+ | `collections add-books <id> <book_id>...` | Add books to a static collection |
287
+ | `collections remove-books <id> <book_id>...` | Remove books from a static collection |
288
+ | `collections edit <id> --query '<rule>'` | Convert a static collection to rule-based (discards hand-picked books) |
289
+ | `collections edit <id> --clear-query` | Convert a rule-based collection to static, snapshotting current members |
290
+ | `collections preview --query '<rule>'` | Show which books a rule matches, without saving |
291
+ | `collections query-help` | Print the full query reference (fields, operators, dates, examples) |
292
+ | `collections rename <id> <new_name>` | Rename a collection |
293
+ | `collections rm <id>` | Delete a collection (books are not deleted) |
294
+
295
+ Converting between the two kinds is one-way destructive in one direction:
296
+ setting a rule on a static collection **discards** its hand-picked books, while
297
+ clearing a rule **snapshots** the currently-matching books into a static list.
298
+ In the web edit form these conversions are gated — when a static collection with
299
+ hand-picked books gains a rule, or any rule is cleared, the form warns with the
300
+ exact count of books affected and requires a second confirm before writing.
301
+
302
+ From the web UI (`bookery serve`), reach collections via the **Collections** link
303
+ in the header. A book's detail page has a **Collections** section that lists the
304
+ static collections it belongs to (each with a **Remove**), an **Add to
305
+ collection** picker (static collections only — rule-based membership is derived),
306
+ and a **New collection from this book** link that seeds a rule from the book's
307
+ series or author. Deleting a collection from its detail page asks for
308
+ confirmation first; your books are never deleted.
309
+
310
+ #### Rule query language
311
+
312
+ A rule query is a [Lucene](https://lucene.apache.org/)-style expression over a
313
+ whitelisted set of fields, parsed permissively and validated restrictively — an
314
+ unknown field, bad value, or unsupported shape is rejected with a message naming
315
+ what's allowed. Run `bookery collections query-help` for the in-terminal
316
+ reference.
317
+
318
+ | Field | Matches |
319
+ |-------|---------|
320
+ | `id` | exact book id |
321
+ | `title` | exact, phrase, or `prefix*` (left-anchored) |
322
+ | `author` | substring (contains) |
323
+ | `series` | exact |
324
+ | `genre` | exact canonical genre |
325
+ | `tag` | exact |
326
+ | `language` | exact |
327
+ | `publisher` | exact |
328
+ | `subject` | substring (contains) |
329
+ | `isbn` | exact |
330
+ | `year` | publication year — `=`, range, or comparison |
331
+ | `rating` | `=`, range, or comparison |
332
+ | `added` | date added (ISO `YYYY-MM-DD`) — `=`, range, or comparison |
333
+
334
+ Operators: `AND`, `OR`, `NOT`, grouping with `( )`, and `+`/`-` prefixes
335
+ (require/exclude). The numeric/date fields (`year`, `rating`, `added`) also
336
+ accept ranges `[a TO b]` (inclusive), `{a TO b}` (exclusive), open-ended `*`,
337
+ and comparisons `>=` `<=` `>` `<`. Fuzzy (`~`) and boost (`^`) are not
338
+ supported.
339
+
340
+ ```text
341
+ series:Dune
342
+ genre:"Science Fiction" AND year:[2020 TO *]
343
+ rating:>=4
344
+ author:"Ursula K. Le Guin" NOT tag:reread
345
+ ```
346
+
347
+ ### Web UI
348
+
349
+ | Command | Description |
350
+ |---------|-------------|
351
+ | `serve` | Launch the local web UI (`--host`, `--port`; default `127.0.0.1:5000`) |
352
+
353
+ The web UI provides paginated and sortable browsing, filter chips, cover thumbnails, a responsive mobile card layout, per-book detail and edit pages, a "search active providers" flow that lets you apply candidate metadata with a side-by-side diff, and inline delete. A **Columns** control on the book list toggles the optional Series, ISBN, Language, Publisher, Added, and Enriched columns (Title, Author, and cover always show); the choice is remembered per browser. The default view shows Title, Author, Added, and Enriched. The diff panel has a checkbox per changed field (all checked by default), so you can apply one, some, or all of a candidate's fields — including a cover-only apply; unchecked fields keep their current values. Subjects are among the applyable fields: applying them re-derives the book's genre(s) from the provider's subjects, the same auto-mapping `bookery genre` performs on the CLI. When you apply a candidate, its cover image (if the provider offers one and the cover row is checked) is fetched and embedded into the non-destructive EPUB copy alongside the text fields; a cover-fetch failure is non-fatal — the text metadata still applies and the result message notes the cover was skipped.
354
+
355
+ Series get their own read-only browsing surface: a **Series** item in the masthead leads to an index of every series with book counts (and a hint when positions are missing), each series page lists its books ordered by position, and a book's detail page shows its series as "Series #1 of 14" linking back to the series page. The book list's Series column is sortable, grouping a sorted library by series with unpositioned and unseriesed books sinking to the bottom. Series metadata comes from `bookery series backfill` (see Metadata & Matching above).
356
+
357
+ Collections can be created and edited directly from the web UI. "New collection" (on the collections page) opens a form for the name, an optional description, and an optional raw rule query (the same Lucene subset as `collections create --query`); leave the query blank for a hand-picked static collection. Each collection's detail page has an "Edit" affordance for changing its name, description, and (for an already rule-based collection) its rule. Invalid input — a blank or duplicate name, or an unparseable rule — is reported inline on the form with the field whitelist hint, never as an error page. A "Preview matches" button on the form resolves the rule without saving and shows the true match count plus a capped sample, so you can see exactly which books a rule selects before committing to it.
358
+
359
+ For beginners who don't want to memorize the query syntax, the form also has a collapsible **Query builder** (collapsed by default). Pick a field from the whitelist, type a value, and "Add condition (AND)" appends a correctly-quoted `field:"value"` clause to the rule textarea — quoting is handled server-side so the composed clause is always valid. The builder only ever appends; the raw textarea remains the single source of truth, so advanced users can keep hand-editing it (OR/NOT, grouping, ranges, comparisons) and ignore the builder entirely.
360
+
361
+ ### Device sync
362
+
363
+ | Command | Description |
364
+ |---------|-------------|
365
+ | `sync kobo` | Convert library EPUBs to `.kepub.epub`, copy to a mounted Kobo, and mirror collections to device shelves |
366
+ | `sync kobo --target <path>` | Override auto-detection with an explicit mount point |
367
+ | `sync kobo --dry-run` | Show what would be copied without touching the device |
368
+ | `sync kobo --no-kepub` | Send plain EPUBs untouched — skip kepub conversion (and the `kepubify` dependency) |
369
+ | `collections show <id> --sync-status` | Show per-device shelf sync state for a collection |
370
+
371
+ Conversion to `.kepub.epub` requires the
372
+ [`kepubify`](https://pgaskin.net/kepubify/) binary on `PATH`
373
+ (`brew install kepubify` on macOS). Files are written to
374
+ `<kobo>/Bookery/Author/Title/Title.kepub.epub` — the dedicated `Bookery/`
375
+ subdirectory keeps synced content visibly separate from Calibre
376
+ sideloads, Kobo store purchases, and library borrows. Pass `--no-kepub`
377
+ (or set `kepub = false` under `[sync.kobo]`) to copy plain EPUBs as
378
+ `Title.epub` instead; kepub adds richer on-device progress/stats, but
379
+ Kobo reads plain sideloaded EPUBs fine, and this path needs no `kepubify`.
380
+ Sync is currently **additive**: existing files on the device are never
381
+ deleted — including when you switch a book's format, so the old
382
+ `.kepub.epub`/`.epub` is left in place. A SQLite cache at
383
+ `{data_dir}/kepub_cache.db` keyed on the source EPUB hash plus the
384
+ `kepubify` version makes re-syncs effectively free when nothing has
385
+ changed.
386
+
387
+ ```toml
388
+ [sync.kobo]
389
+ kepub = false # default true; send plain EPUBs when false
390
+ ```
391
+
392
+ Readers sort authors by the EPUB's `opf:file-as` key. If a device files
393
+ an author under their given name (e.g. "Brandon" instead of "Sanderson"),
394
+ the library copies are missing that key — run `bookery authors fix-sort`
395
+ to backfill a surname-first `file-as`, then re-sync.
396
+
397
+ Some devices (Kobo) ignore `file-as` and sort by the raw `dc:creator`
398
+ text instead, so a name stored as `Cussler, Clive` files under "C". Fix
399
+ the catalog with `bookery authors normalize` (reorders `Surname, Given`
400
+ to `Given Surname`) or `bookery authors merge` (collapses typos and
401
+ duplicate spellings into one canonical name). Both are dry-run by default
402
+ and back the database up before `--apply`. The corrected name flows into
403
+ the EPUB's `dc:creator` on the next write-back and re-sync. Use
404
+ `bookery authors list --duplicates` first to see what would change.
405
+
406
+ ### The `vault-export` workflow
407
+
408
+ Turn an Obsidian vault into a single EPUB — one chapter per note, with a
409
+ clickable TOC, resolved `[[wiki-links]]` and `![[image]]` embeds, and an
410
+ optional tag index at the end. Requires [pandoc](https://pandoc.org)
411
+ (`brew install pandoc` on macOS).
412
+
413
+ Run with flags:
414
+
415
+ ```bash
416
+ bookery vault-export --vault ~/obsidian-vault -o vault.epub \
417
+ --include-folder "3_Permanent Notes" --include-folder "2_Literature Notes" \
418
+ --index --exclude-tag type/meeting
419
+ ```
420
+
421
+ `--folder` is accepted as a deprecated alias for `--include-folder` and will
422
+ be removed in a future release.
423
+
424
+ Or set defaults once in `~/.bookery/config.toml` and just run
425
+ `bookery vault-export -o vault.epub`:
426
+
427
+ ```toml
428
+ [vault_export]
429
+ vault_path = "~/obsidian-vault"
430
+ folders = ["3_Permanent Notes", "2_Literature Notes"]
431
+ include_index = true
432
+ index_exclude_prefixes = ["type/"] # hide tags like `type/permanent` from index
433
+ index_min_count = 1
434
+ exclude_tags = ["type/meeting"] # drop notes with these exact frontmatter tags
435
+ default_author = "Your Name"
436
+ uuid_mode = "stable" # "stable" keeps the same dc:identifier
437
+ # across re-exports so Kobo updates in place.
438
+ # Pass `--random-ids` on the command line
439
+ # for a fresh identifier per export.
440
+ catalog = true # auto-add the EPUB to the bookery library
441
+ # so it ships on the next `sync kobo`
442
+ ```
443
+
444
+ `exclude_tags` matches the full tag string exactly (`type/meeting` skips
445
+ notes tagged `type/meeting` but not `type/permanent`). Callouts, block
446
+ references, note embeds (`![[note]]`), and Dataview queries are **not**
447
+ resolved in this version.
448
+
449
+ With `catalog = true` (or `--catalog` on the command line), the export is
450
+ imported into the library and then deployed to your reader on the next
451
+ `bookery sync kobo` — no separate `bookery add` step. A vault export is a
452
+ point-in-time snapshot, so re-running replaces the prior catalog row and
453
+ EPUB rather than piling up a new copy each day.
454
+
455
+ ### The `match` workflow
456
+
457
+ When you run `bookery match`, Bookery will:
458
+
459
+ 1. Extract metadata from each EPUB
460
+ 2. Normalize mangled titles (CamelCase splitting, word segmentation)
461
+ 3. Search Open Library by ISBN first, then fall back to title/author
462
+ 4. Present you with scored candidates to review
463
+ 5. Write your chosen metadata to a copy of the file
464
+
465
+ During review, you can:
466
+ - **[1-N]** Accept a candidate
467
+ - **[v1-vN]** View a side-by-side comparison
468
+ - **[u]** Look up a specific Open Library URL
469
+ - **[s]** Skip this book
470
+ - **[k]** Keep the original metadata
471
+
472
+ ## Project Structure
473
+
474
+ ```
475
+ src/bookery/
476
+ cli/ # Click commands
477
+ core/ # Pipeline logic (non-destructive write-back, conversion)
478
+ formats/ # Format handlers (EPUB via ebooklib, MOBI via KindleUnpack)
479
+ metadata/ # Matching engine
480
+ openlibrary.py # Open Library API provider
481
+ scoring.py # Weighted confidence scoring
482
+ normalizer.py # Title/author normalization (CamelCase, wordninja)
483
+ candidate.py # MetadataCandidate dataclass
484
+ provider.py # MetadataProvider protocol
485
+ db/ # SQLite catalog (schema, CRUD, search, migrations)
486
+ ```
487
+
488
+ ## Development
489
+
490
+ ```bash
491
+ # Install dev dependencies
492
+ uv sync
493
+
494
+ # Run tests
495
+ uv run pytest
496
+
497
+ # Run linter
498
+ uv run ruff check src/ tests/
499
+
500
+ # Run a specific test file
501
+ uv run pytest tests/unit/test_scoring.py -v
502
+ ```
503
+
504
+ Tests are isolated from your real `~/.bookery/library.db` by an autouse
505
+ guardrail. If you previously ran the suite without this guardrail and your
506
+ catalog now reports `source missing: /private/var/folders/.../pytest-of-...`,
507
+ see ["Recovering from test pollution"](CONTRIBUTING.md#recovering-from-test-pollution)
508
+ in `CONTRIBUTING.md`.
509
+
510
+ ## Roadmap
511
+
512
+ Bookery is being built in phases:
513
+
514
+ 1. ~~EPUB metadata read/write + CLI skeleton~~
515
+ 2. ~~Multi-provider matching (Open Library + Google Books) with consensus merger and interactive review~~
516
+ 3. ~~SQLite catalog, import pipeline, per-field provenance, query commands~~
517
+ 4. ~~MOBI-to-EPUB conversion~~
518
+ 5. ~~PDF-to-EPUB conversion (semantic, local-LLM)~~
519
+ 6. ~~Kobo device sync~~
520
+ 7. ~~Obsidian vault-export to EPUB~~
521
+ 8. ~~Web UI for browse/search/edit~~
522
+ 9. **Next:** Plugin architecture (format and provider plugins)
523
+ 10. Polish (covers cache, performance benchmarks, config polish)
524
+
525
+ See [docs/roadmap.md](docs/roadmap.md) for detailed checklists.
526
+
527
+ ## Design Principles
528
+
529
+ - **Metadata-first** — get the data right before organizing files
530
+ - **Non-destructive** — original file contents are never modified; metadata writes go to a copy in the library
531
+ - **CLI-first** — every feature is a terminal command; no GUI required
532
+ - **Extensible** — providers, formats, and devices will be pluggable
533
+ - **Respectful** — rate-limited API usage, no scraping
534
+
535
+ ## Contributing
536
+
537
+ Bookery is AI-friendly and developed with the help of [Claude Code](https://docs.anthropic.com/en/docs/claude-code). AI-assisted contributions are welcome alongside traditional ones.
538
+
539
+ See [CONTRIBUTING.md](CONTRIBUTING.md) for coding standards, testing requirements, and how to get started.
540
+
541
+ ## License
542
+
543
+ [MIT](LICENSE)