zotero-command-line 2.8.12__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (162) hide show
  1. zotero_command_line-2.8.12/CHANGELOG.md +466 -0
  2. zotero_command_line-2.8.12/LICENSE +21 -0
  3. zotero_command_line-2.8.12/MANIFEST.in +4 -0
  4. zotero_command_line-2.8.12/PKG-INFO +274 -0
  5. zotero_command_line-2.8.12/README.md +212 -0
  6. zotero_command_line-2.8.12/pyproject.toml +147 -0
  7. zotero_command_line-2.8.12/setup.cfg +4 -0
  8. zotero_command_line-2.8.12/src/zotero_cli/__init__.py +2 -0
  9. zotero_command_line-2.8.12/src/zotero_cli/api/__init__.py +0 -0
  10. zotero_command_line-2.8.12/src/zotero_cli/api/dependencies.py +68 -0
  11. zotero_command_line-2.8.12/src/zotero_cli/api/main.py +107 -0
  12. zotero_command_line-2.8.12/src/zotero_cli/api/routes/__init__.py +0 -0
  13. zotero_command_line-2.8.12/src/zotero_cli/api/routes/collections.py +30 -0
  14. zotero_command_line-2.8.12/src/zotero_cli/api/routes/items.py +85 -0
  15. zotero_command_line-2.8.12/src/zotero_cli/api/routes/jobs.py +57 -0
  16. zotero_command_line-2.8.12/src/zotero_cli/cli/__init__.py +0 -0
  17. zotero_command_line-2.8.12/src/zotero_cli/cli/base.py +53 -0
  18. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/__init__.py +27 -0
  19. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/collection_cmd.py +528 -0
  20. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/import_cmd.py +305 -0
  21. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/init_cmd.py +160 -0
  22. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/item_cmd.py +1303 -0
  23. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/rag_cmd.py +457 -0
  24. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/report_cmd.py +560 -0
  25. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/search_cmd.py +97 -0
  26. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/serve_cmd.py +94 -0
  27. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/__init__.py +23 -0
  28. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/decide_cmd.py +72 -0
  29. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/dedupe_cmd.py +181 -0
  30. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/extraction_cmd.py +50 -0
  31. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/list_cmd.py +321 -0
  32. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/load_cmd.py +116 -0
  33. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/promote_cmd.py +88 -0
  34. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/reconcile_cmd.py +165 -0
  35. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/report_cmd.py +532 -0
  36. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/screen_cmd.py +40 -0
  37. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/sdb_cmd.py +152 -0
  38. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/snowball_cmd.py +199 -0
  39. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr/source_cmd.py +276 -0
  40. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/slr_cmd.py +242 -0
  41. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/storage_cmd.py +68 -0
  42. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/system_cmd.py +785 -0
  43. zotero_command_line-2.8.12/src/zotero_cli/cli/commands/tag_cmd.py +118 -0
  44. zotero_command_line-2.8.12/src/zotero_cli/cli/main.py +119 -0
  45. zotero_command_line-2.8.12/src/zotero_cli/cli/presenters/item_list_presenter.py +185 -0
  46. zotero_command_line-2.8.12/src/zotero_cli/cli/presenters/rag_presenter.py +90 -0
  47. zotero_command_line-2.8.12/src/zotero_cli/cli/tui/__init__.py +0 -0
  48. zotero_command_line-2.8.12/src/zotero_cli/cli/tui/components.py +34 -0
  49. zotero_command_line-2.8.12/src/zotero_cli/cli/tui/extraction_tui.py +133 -0
  50. zotero_command_line-2.8.12/src/zotero_cli/cli/tui/factory.py +35 -0
  51. zotero_command_line-2.8.12/src/zotero_cli/cli/tui/screening_tui.py +183 -0
  52. zotero_command_line-2.8.12/src/zotero_cli/cli/tui/snowball_tui.py +252 -0
  53. zotero_command_line-2.8.12/src/zotero_cli/core/__init__.py +0 -0
  54. zotero_command_line-2.8.12/src/zotero_cli/core/config.py +354 -0
  55. zotero_command_line-2.8.12/src/zotero_cli/core/exceptions.py +26 -0
  56. zotero_command_line-2.8.12/src/zotero_cli/core/interfaces.py +610 -0
  57. zotero_command_line-2.8.12/src/zotero_cli/core/logging_config.py +170 -0
  58. zotero_command_line-2.8.12/src/zotero_cli/core/models.py +120 -0
  59. zotero_command_line-2.8.12/src/zotero_cli/core/services/__init__.py +0 -0
  60. zotero_command_line-2.8.12/src/zotero_cli/core/services/arxiv_query_parser.py +138 -0
  61. zotero_command_line-2.8.12/src/zotero_cli/core/services/attachment_service.py +336 -0
  62. zotero_command_line-2.8.12/src/zotero_cli/core/services/audit_service.py +94 -0
  63. zotero_command_line-2.8.12/src/zotero_cli/core/services/backup_service.py +210 -0
  64. zotero_command_line-2.8.12/src/zotero_cli/core/services/collection_service.py +334 -0
  65. zotero_command_line-2.8.12/src/zotero_cli/core/services/diagnostics_service.py +135 -0
  66. zotero_command_line-2.8.12/src/zotero_cli/core/services/duplicate_service.py +327 -0
  67. zotero_command_line-2.8.12/src/zotero_cli/core/services/embedding_provider.py +195 -0
  68. zotero_command_line-2.8.12/src/zotero_cli/core/services/enrichment_service.py +132 -0
  69. zotero_command_line-2.8.12/src/zotero_cli/core/services/export_service.py +105 -0
  70. zotero_command_line-2.8.12/src/zotero_cli/core/services/extraction_service.py +271 -0
  71. zotero_command_line-2.8.12/src/zotero_cli/core/services/graph_service.py +60 -0
  72. zotero_command_line-2.8.12/src/zotero_cli/core/services/identity_manager.py +43 -0
  73. zotero_command_line-2.8.12/src/zotero_cli/core/services/import_service.py +49 -0
  74. zotero_command_line-2.8.12/src/zotero_cli/core/services/job_queue_service.py +64 -0
  75. zotero_command_line-2.8.12/src/zotero_cli/core/services/llm_provider.py +122 -0
  76. zotero_command_line-2.8.12/src/zotero_cli/core/services/merge_plan_io.py +199 -0
  77. zotero_command_line-2.8.12/src/zotero_cli/core/services/merge_service.py +413 -0
  78. zotero_command_line-2.8.12/src/zotero_cli/core/services/metadata_aggregator.py +164 -0
  79. zotero_command_line-2.8.12/src/zotero_cli/core/services/network_gateway.py +186 -0
  80. zotero_command_line-2.8.12/src/zotero_cli/core/services/pdf_finder_service.py +131 -0
  81. zotero_command_line-2.8.12/src/zotero_cli/core/services/purge_service.py +284 -0
  82. zotero_command_line-2.8.12/src/zotero_cli/core/services/rag_service.py +543 -0
  83. zotero_command_line-2.8.12/src/zotero_cli/core/services/report_service.py +219 -0
  84. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/__init__.py +0 -0
  85. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/arxiv.py +49 -0
  86. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/bdtd.py +153 -0
  87. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/generic_scraper.py +88 -0
  88. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/openalex.py +63 -0
  89. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/semantic_scholar.py +63 -0
  90. zotero_command_line-2.8.12/src/zotero_cli/core/services/resolvers/unpaywall.py +55 -0
  91. zotero_command_line-2.8.12/src/zotero_cli/core/services/restore_service.py +308 -0
  92. zotero_command_line-2.8.12/src/zotero_cli/core/services/sandbox_service.py +105 -0
  93. zotero_command_line-2.8.12/src/zotero_cli/core/services/screening_service.py +284 -0
  94. zotero_command_line-2.8.12/src/zotero_cli/core/services/screening_state.py +78 -0
  95. zotero_command_line-2.8.12/src/zotero_cli/core/services/sdb/sdb_service.py +229 -0
  96. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/citation_service.py +36 -0
  97. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/csv_inbound.py +267 -0
  98. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/dedupe_service.py +229 -0
  99. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/integrity.py +99 -0
  100. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/orchestrator.py +263 -0
  101. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/snapshot.py +60 -0
  102. zotero_command_line-2.8.12/src/zotero_cli/core/services/slr/status_service.py +312 -0
  103. zotero_command_line-2.8.12/src/zotero_cli/core/services/snapshot_service.py +145 -0
  104. zotero_command_line-2.8.12/src/zotero_cli/core/services/snowball_graph.py +282 -0
  105. zotero_command_line-2.8.12/src/zotero_cli/core/services/snowball_ingestion.py +180 -0
  106. zotero_command_line-2.8.12/src/zotero_cli/core/services/snowball_worker.py +229 -0
  107. zotero_command_line-2.8.12/src/zotero_cli/core/services/storage_service.py +162 -0
  108. zotero_command_line-2.8.12/src/zotero_cli/core/services/sync_service.py +158 -0
  109. zotero_command_line-2.8.12/src/zotero_cli/core/services/tag_service.py +68 -0
  110. zotero_command_line-2.8.12/src/zotero_cli/core/services/transfer_service.py +83 -0
  111. zotero_command_line-2.8.12/src/zotero_cli/core/services/verify_service.py +122 -0
  112. zotero_command_line-2.8.12/src/zotero_cli/core/strategies.py +103 -0
  113. zotero_command_line-2.8.12/src/zotero_cli/core/utils/archive_safety.py +57 -0
  114. zotero_command_line-2.8.12/src/zotero_cli/core/utils/csv_safety.py +61 -0
  115. zotero_command_line-2.8.12/src/zotero_cli/core/utils/normalization.py +47 -0
  116. zotero_command_line-2.8.12/src/zotero_cli/core/utils/safe_tempfile.py +30 -0
  117. zotero_command_line-2.8.12/src/zotero_cli/core/utils/sdb_parser.py +57 -0
  118. zotero_command_line-2.8.12/src/zotero_cli/core/utils/slugify.py +12 -0
  119. zotero_command_line-2.8.12/src/zotero_cli/core/utils/terminal_safety.py +54 -0
  120. zotero_command_line-2.8.12/src/zotero_cli/core/utils/url_safety.py +386 -0
  121. zotero_command_line-2.8.12/src/zotero_cli/core/zotero_item.py +86 -0
  122. zotero_command_line-2.8.12/src/zotero_cli/infra/__init__.py +0 -0
  123. zotero_command_line-2.8.12/src/zotero_cli/infra/ai_provider_factory.py +148 -0
  124. zotero_command_line-2.8.12/src/zotero_cli/infra/arxiv_lib.py +92 -0
  125. zotero_command_line-2.8.12/src/zotero_cli/infra/base_api_client.py +109 -0
  126. zotero_command_line-2.8.12/src/zotero_cli/infra/bdtd_api.py +328 -0
  127. zotero_command_line-2.8.12/src/zotero_cli/infra/bibtex_lib.py +135 -0
  128. zotero_command_line-2.8.12/src/zotero_cli/infra/canonical_csv_lib.py +60 -0
  129. zotero_command_line-2.8.12/src/zotero_cli/infra/core_api.py +143 -0
  130. zotero_command_line-2.8.12/src/zotero_cli/infra/crossref_api.py +77 -0
  131. zotero_command_line-2.8.12/src/zotero_cli/infra/dblp_api.py +81 -0
  132. zotero_command_line-2.8.12/src/zotero_cli/infra/doaj_api.py +148 -0
  133. zotero_command_line-2.8.12/src/zotero_cli/infra/eric_api.py +92 -0
  134. zotero_command_line-2.8.12/src/zotero_cli/infra/factory.py +478 -0
  135. zotero_command_line-2.8.12/src/zotero_cli/infra/hal_api.py +89 -0
  136. zotero_command_line-2.8.12/src/zotero_cli/infra/http_client.py +224 -0
  137. zotero_command_line-2.8.12/src/zotero_cli/infra/ieee_csv_lib.py +46 -0
  138. zotero_command_line-2.8.12/src/zotero_cli/infra/inspire_hep_api.py +113 -0
  139. zotero_command_line-2.8.12/src/zotero_cli/infra/metadata_client_factory.py +223 -0
  140. zotero_command_line-2.8.12/src/zotero_cli/infra/openalex_api.py +156 -0
  141. zotero_command_line-2.8.12/src/zotero_cli/infra/opener.py +78 -0
  142. zotero_command_line-2.8.12/src/zotero_cli/infra/pubmed_api.py +195 -0
  143. zotero_command_line-2.8.12/src/zotero_cli/infra/repositories.py +149 -0
  144. zotero_command_line-2.8.12/src/zotero_cli/infra/repository_factory.py +110 -0
  145. zotero_command_line-2.8.12/src/zotero_cli/infra/resolver_factory.py +216 -0
  146. zotero_command_line-2.8.12/src/zotero_cli/infra/ris_lib.py +84 -0
  147. zotero_command_line-2.8.12/src/zotero_cli/infra/semantic_scholar_api.py +174 -0
  148. zotero_command_line-2.8.12/src/zotero_cli/infra/service_factory.py +462 -0
  149. zotero_command_line-2.8.12/src/zotero_cli/infra/springer_csv_lib.py +44 -0
  150. zotero_command_line-2.8.12/src/zotero_cli/infra/sqlite_repo.py +737 -0
  151. zotero_command_line-2.8.12/src/zotero_cli/infra/sqlite_vector_repo.py +184 -0
  152. zotero_command_line-2.8.12/src/zotero_cli/infra/unpaywall_api.py +88 -0
  153. zotero_command_line-2.8.12/src/zotero_cli/infra/zbmath_api.py +95 -0
  154. zotero_command_line-2.8.12/src/zotero_cli/infra/zotero_api.py +665 -0
  155. zotero_command_line-2.8.12/src/zotero_cli/templates/demo_sandbox.yaml +45 -0
  156. zotero_command_line-2.8.12/src/zotero_cli/templates/extraction_schema.yaml +31 -0
  157. zotero_command_line-2.8.12/src/zotero_command_line.egg-info/PKG-INFO +274 -0
  158. zotero_command_line-2.8.12/src/zotero_command_line.egg-info/SOURCES.txt +160 -0
  159. zotero_command_line-2.8.12/src/zotero_command_line.egg-info/dependency_links.txt +1 -0
  160. zotero_command_line-2.8.12/src/zotero_command_line.egg-info/entry_points.txt +2 -0
  161. zotero_command_line-2.8.12/src/zotero_command_line.egg-info/requires.txt +39 -0
  162. zotero_command_line-2.8.12/src/zotero_command_line.egg-info/top_level.txt +1 -0
@@ -0,0 +1,466 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project will be documented in this file.
4
+
5
+ ## [Unreleased]
6
+
7
+ ## [2.8.12] - 2026-09-24
8
+
9
+ This release fixes the findings of the pre-launch security audit (#328). The twelve fixed vulnerabilities are described in GitHub Security Advisories published with this release. **Upgrading is recommended for everyone.**
10
+
11
+ ### 🛡️ Security
12
+ - **Release pipeline and installers:**
13
+ - Release builds run with a read-only token; only the job that publishes the release can write.
14
+ - Every GitHub Action is pinned to a commit, and PyInstaller and fpm are pinned. No build cache is shared with CI.
15
+ - Releases are made only from tags on `main`, and a tag ruleset limits `v*` tags to maintainers.
16
+ - Every release now ships `SHA256SUMS` and signed build provenance (`gh attestation verify <file> -R fchicout/zotero-cli`).
17
+ - `install.sh` and `install.ps1` verify the download against `SHA256SUMS` before installing and accept `ZOTERO_CLI_VERSION` to pin a release. `install.sh` no longer installs the Linux binary on macOS.
18
+ - The release workflow also runs on pull requests that touch it, building and smoke-testing every artifact without publishing.
19
+ - **API keys kept out of logs and the terminal:** every log line, traceback included, is masked before it's written: configured secrets, credential URL parameters, and Zotero `/keys/<key>` paths. httpx request logging is quieted. `init` checks the key with `GET /keys/current` and sends it only in a header.
20
+ - **Local files private to your account:** the storage and log directories are `0700` and log files `0600`, including rotated ones. The offline copy of `zotero.sqlite` is now a `0600` file in a private temp directory and is removed at exit.
21
+ - **`serve` restricted to local use:**
22
+ - Requests whose `Host` header isn't `127.0.0.1`, `localhost` or `::1` are rejected, which stops web pages reaching the API through DNS rebinding.
23
+ - A non-loopback `--host` now requires `--allow-remote`, which prints a random bearer token that every request must send.
24
+ - `--allowed-host` adds extra host names.
25
+ - **Outbound requests (SSRF):**
26
+ - Only globally routable addresses are allowed, so CGNAT/Tailscale ranges and IPv4-mapped loopback are now refused.
27
+ - Connections are pinned to the address that was validated.
28
+ - BDTD PDF resolution goes through the guard and matches link hosts exactly.
29
+ - Every PDF resolver, and the upload step, requires the `%PDF` signature.
30
+ - A unit test fails on new direct HTTP calls outside the guard.
31
+ - **Terminal output:** every console strips control characters and bidi overrides from library data, so a title can no longer write to your clipboard or disguise a link. Library data interpolated into Rich markup is escaped, so a title containing `[/bold]` no longer crashes `item inspect`.
32
+ - **OS launcher:** `slr extraction` opens only `http(s)` item URLs, in the browser, and never passes item data to `os.startfile`/`xdg-open`.
33
+ - **Smaller fixes:**
34
+ - `slr list --xlsx` stores formula-like text as plain strings.
35
+ - Attachment downloads follow Zotero's storage redirect without the API key.
36
+ - `system restore`/`verify` read `.zaf` archives within size and entry limits.
37
+ - SDB and extraction notes are written as HTML-escaped JSON, so `<`/`>` in a reason can't alter the record.
38
+ - `storage checkout` refuses group libraries unless `--allow-group-library` is given, because the local path would sync to every member.
39
+ - **Security policy:** new `SECURITY.md` explaining how to report vulnerabilities privately (Issue #336).
40
+
41
+ ### 🐛 Bug Fixes
42
+ - **PubMed returned the wrong paper for a DOI (Issue #340):** NCBI read a DOI like `10.1145/...` as PMID 10, and that record's longer title then won the metadata merge. PubMed now resolves DOIs through its DOI search and rejects a record whose DOI differs. The aggregator also drops any candidate whose DOI contradicts the one looked up.
43
+ - **Requests carried the maintainer's email (Issue #337):** the User-Agent and NCBI parameters included a hardcoded personal address. The User-Agent now names the real version, and a `mailto` is added only when you set `unpaywall_email`. Unpaywall, which requires an email, is skipped without one.
44
+
45
+ ### 🛡️ Quality & Infrastructure
46
+ - **Dependency vulnerability scanning (Issue #335):** CI runs `pip-audit` on the locked runtime dependencies (blocking) and on extras/dev (informational). Dependabot covers uv, GitHub Actions and Docker. `soupsieve` is now 2.10 and `cryptography` 50.0.1, and the deprecated `safety` is gone.
47
+ - **Docker image hardened (Issue #338):** the base image is pinned by digest, the install comes from `uv.lock`, and the container runs as an unprivileged user with its config in `/config/zotero-cli`. The README shows `--env-file` and `--user "$(id -u):$(id -g)"`.
48
+ - **MSI installs per user (Issue #341):** `Scope="perUser"`, so installing no longer asks for elevation.
49
+
50
+ ### 📦 Distribution
51
+ - **Published on PyPI as `zotero-command-line`:** `uv tool install zotero-command-line` or `pipx install zotero-command-line` now installs the CLI. The command is still `zotero-cli` and the import package is still `zotero_cli`. The PyPI name differs because `zotero-cli` there belongs to an unrelated project inactive since 2016, and PyPI rejects look-alike names such as `zoterocli`. Pushing a `vX.Y.Z` tag now also runs a `publish-pypi` job after the binaries and GitHub release. It checks that the tag matches the package version, builds the sdist and wheel, smoke-tests the wheel in a clean virtualenv, and uploads with `uv publish`. The job has a read-only `GITHUB_TOKEN`, no build cache, and actions pinned by commit SHA; the PyPI token lives only in a `pypi` environment restricted to `v*` tags. The package metadata gains a description, the README as the long description, project URLs, keywords and classifiers. README links are now absolute so they work on the PyPI page.
52
+ - **`pytest` is no longer a runtime dependency (Issue #339):** nothing in `src/` imports it, but every install and release binary shipped it. It's now only in the `dev` extra.
53
+
54
+ ### 📚 Documentation
55
+ - **README repositioned around what zotero-cli is for:** managing a Zotero library from the command line, for scripting, automation and LLM-assisted workflows. It used to open as "The Systematic Review Forge". The SLR features are now an optional toolkit, semantic search (RAG) is no longer featured since it's moving to a separate tool, and a new cookbook shows JSON/CSV output, batch imports, feeding papers to an LLM, preview-then-`--execute`, offline mode and backups. Along the way it corrects claims that no longer matched the code: the local API is read-only, `item hydrate` only updates arXiv items, and `zotero-cli` isn't coming to PyPI. The version badge now tracks the latest release instead of a hardcoded 2.8.1. `zotero-cli --help`, the .deb/.rpm package descriptions and the getting-started tutorial use the same description. References to a specific downstream app were removed from the docs and docstrings; zotero-cli doesn't depend on it.
56
+
57
+ ## [2.8.11] - 2026-09-24
58
+
59
+ ### 🐛 Bug Fixes
60
+ - **Released binaries crashed on start with `ModuleNotFoundError: No module named 'httpx'` (Issue #333):** `httpx` is imported at module level by `core/utils/url_safety.py` and `core/services/network_gateway.py` but was only listed in the `dev` extra, so every install without dev dependencies (release binaries, `install.sh`/`install.ps1`, the Dockerfile, `pip install git+...`) lacked it. v2.8.9 and v2.8.10 crashed on every command, including `--help`; v2.8.4-v2.8.8 crashed on PDF fetching, snowball discovery and job workers (until v2.8.4, `openai` was a runtime dependency and pulled `httpx` in). `httpx` is now a runtime dependency. Fixing it exposed a second crash waiting behind it: the release workflow installed with an unlocked `uv pip install .`, which resolved `bibtexparser` 2.x, a breaking rewrite missing the `bibdatabase` module. Release builds now install exactly what `uv.lock` pins (`uv sync --locked --no-editable`), and `bibtexparser` is capped at `<2` for installs that don't use the lock.
61
+
62
+ ### 🛡️ Quality & Infrastructure
63
+ - **Smoke tests so a broken install can't ship again (Issue #333):** CI now installs runtime dependencies only (no extras) from the lock, runs `zotero-cli --help`, and imports every module via the new `scripts/smoke_imports.py`, before installing the dev extras for the rest of the job. The release workflow runs the built binary (`--help`) before publishing, on both Linux and Windows.
64
+
65
+ ## [2.8.10] - 2026-09-24
66
+
67
+ ### ✨ Features & Improvements
68
+ - **`item list` gains field selection, a `--wide` preset, and JSON/CSV/Markdown output (Issue #323):** `item list` only ever showed a fixed Key/Title/Type table, so gathering authors, year, venue or DOI for a collection meant running `item inspect` once per item. New `--fields` takes a comma-separated list of friendly field names (`key`, `title`, `type`, `creators`, `first_author`, `date`, `year`, `venue`, `doi`, `url`, `abstract`, `extra`, `tags`, `collections`, `parent`, `date_added`, `date_modified`) or any raw Zotero field name (`publicationTitle`, `volume`, `ISSN`, ...); `-w`/`--wide` is a preset for key, title, first author, year, venue and DOI; `-f`/`--format` switches between `table` (default, output unchanged), `json`, `csv` and `markdown`. `venue` picks whichever of `publicationTitle`/`proceedingsTitle`/`conferenceName`/`bookTitle` the item type uses. Non-table formats write only the data to stdout (no title or item-count footer), and a warning about a field no listed item has goes to stderr, so output can be piped into `jq` or redirected to a file. CSV cells go through the existing formula-injection guard (#237), and table cells render as literal text so titles like `[Retracted] ...` aren't eaten as Rich markup (#253). The offline SQLite gateway now also reads the four venue fields, so `venue` works in `--offline` mode.
69
+
70
+ ### 🐛 Bug Fixes
71
+ - **Snowball discovery failed on compressed API responses with "Error -3 while decompressing data: incorrect header check" (Issue #321):** `read_capped_async` (the size-capped reader behind `NetworkGateway`, used by both backward/CrossRef and forward/Semantic Scholar discovery) collects the body via `aiter_bytes()`, which already undoes `Content-Encoding` - but then rebuilt the response with the original headers, still declaring `Content-Encoding: gzip`. httpx then decoded the already-plain bytes a second time and raised a `DecodingError` on any gzip- or deflate-compressed response. The rebuilt response now drops `Content-Encoding` and the stale compressed `Content-Length`. The existing tests used mocked responses with empty headers, so they never exercised httpx's real decoding path; the new regression tests stream real gzip and deflate bodies through an `httpx.AsyncClient`.
72
+ - **`collection list --offline` crashed with `KeyError: 'meta'` (Issue #322):** `SqliteZoteroGateway.get_all_collections()`/`get_collection()` returned collections without the `meta` envelope the Zotero Web API includes, while `collection list` read `meta.numItems` unguarded - so the offline tree view crashed, and so did `--table` mode (the same unguarded access at a second call site, not mentioned in the report). The offline gateway now computes a real `meta.numItems` per collection (items in `collectionItems`, excluding trashed items, matching the Web API's count) instead of defaulting to 0, so offline and online output agree; both `collection list` render paths also fall back to 0 if any gateway omits `meta`.
73
+
74
+ ## [2.8.9] - 2026-09-10
75
+
76
+ ### ✨ Features & Improvements
77
+ - **Snowball discovery graph candidates now persist author/creator data (Issue #318):** `SnowballGraphService.add_candidate()` had no way to store authors at all - nodes only ever held `title`/`abstract`/`is_influential`/`generation`/`status`, making it structurally impossible to build any author-based relevance signal (e.g. "this candidate shares an author with your seed paper") on top of the graph, even though the underlying discovery APIs already return author data. `add_candidate()` now recognizes an `authors: List[str]` key in `paper_metadata` and persists it on the node (backfilling it on an existing stub the same way `title`/`abstract` already are, and never overwriting an already-populated list). `_discover_forward` (Semantic Scholar) now passes through the `authors` list its request already receives (`{"authorId": ..., "name": ...}` objects mapped to plain names) instead of discarding it. `_discover_backward` (CrossRef) now passes through the single `author` string CrossRef's per-reference metadata exposes (sparser than the full author array on the top-level work itself - one name is still strictly better than none for an author-overlap signal). Since `to_json()`/`save_graph()` serialize whatever attributes a node carries via `networkx.node_link_data`, `authors` is automatically included in `slr snowball export --format json` and any direct read of `SnowballGraphService.graph` with no further changes needed.
78
+
79
+ ### 🛡️ Quality & Infrastructure
80
+ - **Docs-only PRs were permanently blocked by an unsatisfiable required status check (Issue #310):** #274's `tests.yml` `paths-ignore: ['**.md', 'docs/**']` correctly skips the real pipeline for a PR whose diff is entirely docs - but branch protection marks that job's `test` check as required, and a required check that no workflow run ever posts a status for can never be satisfied through the normal merge flow. Confirmed hands-on: PR #309 (a pure documentation fix) showed "no checks reported" and stayed `BLOCKED` indefinitely, requiring an admin-override merge to unblock. New companion workflow `tests-docs-only.yml` uses the exact inverse path filter (`paths: ['**.md', 'docs/**']`) and the same job name (`test`) as `tests.yml` - a no-op that succeeds immediately. Between the two workflows, every PR now gets a `test` status posted by whichever one actually triggers (or, for a PR touching both code and docs, by both - a documented, harmless outcome of this pattern, since branch protection only needs one to succeed). This is GitHub's own recommended pattern for `paths-ignore` + required-status-check interactions.
81
+ - **Removed the Infisical CI dependency, and fixed 17 `tests/unit` tests that were silently depending on it being present:** `tests.yml`'s test step installed the Infisical CLI and wrapped `pytest tests/unit` in `infisical run --domain https://infisical.fchicout.dev --env dev -- ...` on every CI run. Removing it surfaced that 17 tests across 8 files (`test_collection_cmd.py`, `test_rag_cmd.py`, `test_reconcile_cmd.py`, `test_report_cmd_coverage.py`, `test_slr_cmd.py`, `test_system_cmd.py`, `test_v2_commands.py`, `test_verify_cmd.py`) weren't actually isolated: their command under test unconditionally calls `GatewayFactory.get_zotero_gateway` (or another factory method that itself calls `RepositoryFactory.get_zotero_gateway`/`get_slr_status_service`/`get_slr_orchestrator`/`get_screening_service` directly, bypassing the mocked facade) before reaching the specific service call the test actually mocks - without a real `ZOTERO_API_KEY`-shaped value present (previously supplied by Infisical in CI, or a developer's own `~/.config/zotero-cli/config.toml` locally), this raised `ConfigurationError` deep inside command dispatch. Each test now mocks the actual factory method(s) its command path calls, matching the pattern already used correctly elsewhere in the same files - consistent with CLAUDE.md's own safety boundary that `tests/unit` must stay fully offline/credential-free (only `tests/e2e` may touch live state). (An earlier attempt at this fix added a placeholder `ZOTERO_API_KEY` default to `tests/unit/conftest.py` instead - reverted, since letting these tests construct a real, unmocked gateway object turned out to let at least one of them reach a real network retry loop, intermittently inflating the suite's wall time to minutes; mocking the actual call site is the correct fix, not papering over the missing mock with a plausible-looking credential.) Replaced the two Infisical workflow steps with a plain `uv run pytest tests/unit --cov=... --cov-report=xml` step - one fewer external dependency, one fewer network call, and CI is no longer coupled to a third-party secrets-management service's uptime for a test suite that was never actually using it.
82
+ - **`RAGServiceBase.ingest()` mixed item-selection, tag/QA filtering, and concurrency orchestration in one method (Issue #256):** `ingest()` was a single ~130-line method with 9 parameters handling 5 mutually-exclusive item-selection modes, screening-tag filtering (`approved_only`/`rsl:include`), QA-score pre-filtering (`min_qa_score`), an inlined `ThreadPoolExecutor` concurrency pipeline, and the actual chunk/embed/store pipeline, all tangled together - hard to unit-test each selection mode in isolation. New `RAGIngestItemSelector` collaborator (same file) owns the 3 selection/filter concerns via a single `resolve(...) -> (items, skipped_low_qa_count)` method (plus the QA-score lookup helper, moved verbatim from `RAGServiceBase._get_item_max_qa_score`); the `ThreadPoolExecutor` pipeline was extracted into its own `_run_ingest_pipeline()` method. `ingest()` itself is now ~25 lines that just wires the two together - mirrors the composable-primitives refactor already done for `ScreeningService.record_decision` in #201/#202. Pure refactor, no behavior change (all existing RAG tests pass unmodified); new `test_rag_ingest_item_selector.py` exercises each selection/filter mode directly against the extracted class. Found via the adversarial two-agent code-quality audit tracked in #232.
83
+ - **4 files (1,212 lines) carried a blanket `# mypy: ignore-errors`, silently narrowing the CI mypy gate (Issue #255):** `core/services/rag_service.py`, `core/services/resolvers/bdtd.py`, `core/services/slr/orchestrator.py`, and `infra/bdtd_api.py` each carried a file-scope `# mypy: ignore-errors` directive, with no corresponding `[[tool.mypy.overrides]]` entry in `pyproject.toml` recording the exception (unlike the Sonar/coverage exclusions, which are explicit in `sonar-project.properties`) - easy to assume "mypy passes" covered more of the codebase than it actually did. Removing the directive from all 4 files surfaced only 3 real errors total, all trivial missing return-type annotations (`record_duplicate_resolution`/`BDTDAPIClient.__init__` → `-> None`, `RAGServiceBase.ingest()`'s inlined `process_item` closure → `-> Dict[str, Any]`) - `resolvers/bdtd.py` needed no changes at all. All 4 files now pass `mypy .` with no suppression whatsoever. Found via the adversarial two-agent code-quality audit tracked in #232.
84
+ - **`tests/unit` had no `conftest.py` - `ZoteroConfig` test fixtures hand-rolled and duplicated (Issue #254):** `tests/unit/` (143 files) had zero `conftest.py` files - only the suite-root `tests/conftest.py` (env-var setup and optional-dependency mocking, no fixtures) and `tests/e2e/conftest.py`. `tests/unit/infra/test_factory.py`, `tests/unit/core/test_rag_integration.py`, and `tests/unit/core/test_storage_service.py` each independently hand-constructed a minimal `ZoteroConfig(...)` rather than sharing one fixture - a maintenance trap, since any future required-field addition to `ZoteroConfig` would mean editing every ad hoc call site individually. New `tests/unit/conftest.py` provides a shared `mock_config` fixture; the three files above now build on it via `dataclasses.replace(mock_config, ...)` (it's a frozen dataclass) instead of reconstructing one from scratch. `test_config.py`'s own constructions were deliberately left alone - that file tests `ZoteroConfig`'s own field-resolution logic and needs distinct, deliberately-varied field combinations per case, not a shared default. Found via the adversarial two-agent code-quality audit tracked in #232.
85
+ - **Metadata-provider clients used `print()` instead of `logger.exception()` for error reporting (Issue #251):** 10 of 13 `infra/*_api.py` metadata-provider clients (plus `zotero_api.py`, ~40 call sites total) swallowed exceptions with `print(f"Error ...: {e}")` instead of `logger.exception(...)` - the pattern the three newest clients (`bdtd_api.py`/`core_api.py`/`doaj_api.py`, from #182/#190) already used correctly, but was never back-ported. `print()` can't be leveled/filtered/redirected via standard logging config and corrupts any command whose stdout must stay machine-parseable (JSON export, piping to `jq`). Five of the affected files (`hal_api.py`, `dblp_api.py`, `eric_api.py`, `zbmath_api.py`, `inspire_hep_api.py`) already had an unused, never-called `logger = logging.getLogger(__name__)` sitting next to the bug. Every `print(...)` error/warning call in these clients (and `zotero_api.py`, mostly funneled through its shared `_safe_execute`/`_paginate_items` helpers) now routes through `logger.exception()`/`logger.warning()` instead, with the remaining clients gaining the missing `logging.getLogger(__name__)`. Found via the adversarial two-agent code-quality audit tracked in #232.
86
+
87
+ - **Hardening: unvalidated DOI values interpolated into external API request URLs (Issue #242):** `SnowballDiscoveryWorker._discover_backward`/`_discover_forward` built the CrossRef/Semantic Scholar request URL by interpolating a `doi` value straight into an f-string - that `doi` can originate from a prior CrossRef/Semantic Scholar API response (`ref.get("DOI")`/`cite_doi`), untrusted third-party bibliographic metadata fed back into subsequent URL construction with no format validation. The hostname is a fixed literal at both sites (no SSRF path - confirmed by both audit passes), but an unescaped `?`/`#` in a DOI value could still inject extra query parameters or alter the intended path. `_process_job` now rejects (via a new `is_valid_doi()` in `core/utils/normalization.py`, checking the plausible `10.NNNN/suffix` shape) any DOI that doesn't look like one before it reaches either discovery method, failing that job outright rather than sending garbage to the external API; both URL-construction sites also now `urllib.parse.quote()` the DOI as defense-in-depth, since a technically DOI-shaped suffix can still legally contain characters that would otherwise inject into the URL. Found via the adversarial two-agent security audit tracked in #231.
88
+ - **Hardening: `x-api-key` header retained across cross-origin redirects in `NetworkGateway` (Issue #241):** `httpx`'s built-in redirect handling strips `Authorization` on a cross-origin hop, but `NetworkGateway` follows redirects manually (Issue #235's SSRF re-validation) and was reusing the original request's headers - including non-standard auth headers like Semantic Scholar's `x-api-key` - unconditionally on every hop, regardless of origin. Not currently exploitable purely from Zotero library or API response data (both audit passes confirmed this requires the fixed, hardcoded `api.semanticscholar.org` target to itself issue a cross-origin redirect - infrastructure compromise or a MITM, not attacker-triggerable input), but a real defense-in-depth gap. `_fetch_validated` now compares each redirect hop's origin (scheme/host/port) against the previous hop's, stripping `Authorization`/`x-api-key` (case-insensitively) from the headers carried forward whenever they differ - mirroring what `httpx` already does for `Authorization`, extended to this codebase's actual non-standard auth header. Found via the adversarial two-agent security audit tracked in #231.
89
+
90
+ ### 🐛 Bug Fixes
91
+ - **`get_config()`'s process-wide cache had no lock around `config_path` reload / `reset_config()` races (Issue #301):** `get_config()`'s read-modify-write of the module-global `_GLOBAL_CONFIG`/`_GLOBAL_CONFIG_PATH` had no lock - `serve`'s threadpool-offloaded routes (#269) already give this module real in-process concurrency, so a concurrent caller could observe the cache mid-reassignment, or `reset_config()` (fired by a concurrent `system` config-write) could race a reader between its None-check and its use of the module-global. `get_config()`/`reset_config()` now both acquire a `threading.Lock` around their read-modify-write. This is the last confirmed finding from the #234 audit tracked to a follow-on issue - closes that audit's full fix cycle. Found via the adversarial two-agent general/correctness audit tracked in #234 (critic pass).
92
+ - **`library_type` was never validated - typos silently coerced to `"groups"`, producing confusing 404s (Issue #300):** `infra/http_client.py` resolved the API URL prefix as `"users" if library_type == "user" else "groups"` - any value other than the exact literal `"user"` (a typo like `"personal"`, or a differently-cased `"Group"`) silently fell through to `"groups"`, with no validation anywhere in `ConfigLoader`/`ZoteroConfig`. A user with a typo'd config got no error at load time - the CLI proceeded to build a `/groups/{library_id}/...` URL against what's actually a user ID, and the Zotero API returned a generic 404/403 with no indication the real cause was three layers up in `config.toml`. `ConfigLoader.load()` now validates `library_type` against the literal `{"user", "group"}` set immediately after resolving it (env > file > default), raising `ConfigurationError` with a clear message at config-load time instead. Found via the adversarial two-agent general/correctness audit tracked in #234 (critic pass).
93
+ - **`SnowballGraphService.save_graph` had no locking or atomic write - concurrent discovery/review sessions could silently corrupt or lose updates to the discovery graph (Issue #299):** `save_graph` did a plain truncating write with no file lock and no atomicity - two concurrent `slr snowball discovery`/`review` sessions on the same graph could interleave writes (a crash mid-write could also leave a corrupt/partial JSON file), and even without interleaving, the second session's save fully overwrote the first's in-memory snapshot with no merge, silently discarding its updates. `save_graph` now writes to a temp file and atomically renames it into place via `os.replace` (crash-safe: a reader never observes a half-written file), and the write is serialized across processes via a new `filelock.FileLock` colocated with the storage path (no more interleaved/torn writes from two concurrent writers). This does not fully close the lost-update window for two sessions' concurrent in-memory edits (that would need a reload-and-merge strategy, or moving this store to sqlite the way #150 did for the job queue - noted as a still-open gap in the code comment) but eliminates the corruption/torn-write failure mode entirely. Found via the adversarial two-agent general/correctness audit tracked in #234 (critic pass).
94
+ - **`get_pdf_finder_service` didn't propagate `force_user` to its job queue, reopening the #150/#228 cross-tenant risk (Issue #298):** `ServiceFactory.get_pdf_finder_service` called `ServiceFactory.get_job_queue_service(config)` - note `force_user` wasn't passed, unlike every sibling factory method in the same file (`get_snowball_worker`, item/collection/attachment repo lookups all thread it through correctly). `get_job_queue_service`'s `resolve_scoping_id(force_user)` therefore always resolved against the default library/group scope for PDF-finder jobs, even when the user passed `--user` to target their personal library instead - `zotero-cli --user pdf-finder run`'s jobs got enqueued/read under the wrong scope, so `pdf-finder status`/retries could look into (or write into) the wrong tenant's job rows. One-line fix: `get_job_queue_service(config, force_user)`, matching every sibling factory method's pattern. Found via the adversarial two-agent general/correctness audit tracked in #234 (critic pass).
95
+
96
+ ### 📚 Documentation
97
+ - **`serve`'s single-library-per-process design was undocumented (Issue #297):** `JobQueueService`/`SnowballGraphService` are library-scoped (#150/#228), but `zotero-cli serve` binds a single `ZoteroGateway`/`JobQueueService` for its entire process lifetime via module-level singletons in `api/dependencies.py`, with no per-request library selection anywhere in `api/`. This is an intentional simplification (one process per library, run multiple instances on different ports for multi-library HTTP access), but nothing said so anywhere. Documented explicitly in `api/dependencies.py`'s own comment, `serve`'s `--help` epilog, and `docs/help_specs/serve.md`'s Cognitive Safeguards section. Found via the adversarial two-agent general/correctness audit tracked in #234.
98
+
99
+ ### 🐛 Bug Fixes
100
+ - **`resolve_scoping_id`'s silent `"default"` fallback could merge unrelated under-configured setups' job-queue/discovery-graph storage with zero warning (Issue #296):** `ZoteroConfig.resolve_scoping_id` (used to scope #150's job queue and #228's snowball discovery graph) deliberately never raises - it falls back to the literal string `"default"` when neither `library_id` nor `user_id` is set. But that fallback was completely silent: any user with an incomplete config (offline-mode users, who don't need Zotero API credentials at all, are exactly the population most likely to hit this) got their local storage silently merged into one shared bucket with any other similarly under-configured session on the same machine, with no way to know it was happening. `resolve_scoping_id` now logs a `logger.warning` whenever it falls through to `"default"`, so there's at least a log trail (once #292's logging configuration is in place) explaining why a discovery graph or job queue appears to mix with another session's data. Found via the adversarial two-agent general/correctness audit tracked in #234.
101
+
102
+ ### 📚 Documentation
103
+ - **`slr_snowball.md` implied the review TUI's "Accept" auto-imports into Zotero; a separate manual `import` command is actually required (Issue #295):** The doc's mermaid diagram (`Accept --> Import: Bulk Upload to Zotero Collection`) and step 4's prose ("Successfully screened papers are automatically added to your Zotero library") both implied Accept writes directly to Zotero. In reality, `review`'s Accept action only calls `graph_service.update_status(..., STATUS_ACCEPTED)` on the local discovery graph - confirmed no Zotero write calls anywhere in `snowball_tui.py`. A separate, manually-invoked `slr snowball import --target <collection>` is what actually writes ACCEPTED candidates to Zotero; nothing runs it automatically. A user following the old doc could reasonably believe their review was lost, or the tool was broken, when accepted candidates didn't appear in their library. Diagram and prose corrected to show the two steps as distinct. Also added a short `--offline` callout to `docs/SETUP_GUIDE.md` noting that most write-oriented commands (`slr decide`, `slr screen`, `item edit`, `slr snowball import`, etc.) aren't offline-compatible and will fail with "Offline mode is read-only" - only `item trash`/`item restore` are offline-mode write paths (verified against `SqliteZoteroGateway`'s write guards). Found via the adversarial two-agent general/correctness audit tracked in #234.
104
+
105
+ ### 🛡️ Quality & Infrastructure
106
+ - **`sonar-project.properties` coverage exclusions bundled real domain logic with presentation-layer exclusions (Issue #294):** `collection_service.py`, `sdb/sdb_service.py`, and `rag_service.py` were excluded from the SonarQube coverage gate alongside `cli/**`/`cli/tui/**` under a "legacy core services" rationale that CLAUDE.md's own architecture description contradicts for these files specifically - `sdb_service.py` in particular is the domain layer for the SLR audit trail ("Screening decisions are written back into Zotero as immutable JSON notes (SDB v1.2) via `SDBService`"), not legacy or presentational. `sdb_service.py`'s `filter_items_by_sdb` (used to build PRISMA/report filtering by decision/criteria/persona/phase) had zero test coverage of its match logic before this fix - new tests in `test_sdb_service.py` exercise every filter combination (tag fast-path short-circuiting, decision/criteria/persona/phase deep-filter matching and non-matching, multiple-entry first-match selection, combined filters), raising the file's coverage from 68% to 89%. All three files removed from `sonar.coverage.exclusions` (kept `cli/**`/`cli/tui/**`, genuinely presentation-layer, and `core/interfaces.py`, Protocols only). Found via the adversarial two-agent general/correctness audit tracked in #234.
107
+
108
+ ### 🐛 Bug Fixes
109
+ - **Repo-wide `except Exception: print(...)` pattern swallowed ~20 sites' worth of diagnosable errors (Issue #293):** `core/services/attachment_service.py`, `sync_service.py`, `storage_service.py`, `screening_state.py`, `audit_service.py`, `report_service.py`, `snapshot_service.py`, and `infra/bibtex_lib.py`/`ris_lib.py`/`ieee_csv_lib.py`/`springer_csv_lib.py`/`resolver_factory.py` all caught exceptions and reported them via bare `print()` instead of `logger.exception()`/`logger.warning()` - even after #292 added logging configuration, none of these sites would route through it, since they call `print()` directly rather than the logging system at all. If any of these run inside `zotero-cli serve` or a background `slr snowball discovery` process with no attached terminal, the output went to whatever stdout/stderr happened to be, with no level, timestamp, or way to filter/aggregate. Since these `print()` calls are each command's primary interactive feedback (unlike #251's internal API-client diagnostics, which that earlier fix correctly replaced outright), a matching `logger.exception(...)`/`logger.warning(...)` call was added *alongside* each existing `print()` rather than replacing it - preserving console UX while making every one of these failures diagnosable from the log file #292 introduced. Found via the adversarial two-agent general/correctness audit tracked in #234 (critic pass).
110
+ - **No logging configuration anywhere - `logger.info`/`.debug` calls vanished, warnings/exceptions weren't persisted (Issue #292):** Despite disciplined `logger.info`/`.warning`/`.exception` use across ~33 modules, nothing anywhere called `logging.basicConfig()` or attached a handler - Python's default root logger level is WARNING with no handler, so every `.info`/`.debug` call was silently discarded and `.warning`/`.exception` fell through to `logging.lastResort` (one unformatted stderr line, no timestamp, no persistence). For a tool whose own docs describe long-running background jobs (`slr snowball discovery`, `system jobs`) meant to be revisited later, this left no way to debug a failed job from logs alone. New `core/logging_config.py::setup_logging()` configures the root logger once per process: a stderr handler matching the pre-existing default visibility (WARNING+, or DEBUG+ with the new `-v`/`--verbose` flag) plus a rotating file handler under `get_storage_dir() / "logs" / "zotero-cli.log"` that always captures INFO+ regardless of verbosity. Called once at the top of `cli/main.py`'s `main()` (before command dispatch, so it covers every command including `serve`, which routes through the same entry point); the top-level unhandled-exception handler now also routes through `logger.exception(...)` in addition to the existing `traceback.print_exc()`. **`tests/unit` isolation note:** since `main()` unconditionally calls `setup_logging()`, any test driving `main()` (most of `test_cli.py`) would otherwise silently write a real `zotero-cli.log` to the developer's actual `~/.config/zotero-cli/` directory - a new autouse fixture in `tests/unit/conftest.py` patches `cli.main.setup_logging` to a no-op for the whole `tests/unit` tree, keeping the suite fully offline/file-write-free per CLAUDE.md's safety boundary; `setup_logging()`'s own behavior is exercised directly against a `tmp_path` in new `test_logging_config.py`. Found via the adversarial two-agent general/correctness audit tracked in #234.
111
+ - **Offline `SqliteZoteroGateway` has zero library/collection scoping - now warns loudly instead of silently spanning every locally-synced library (Issue #291):** Unlike online mode (which always scopes every request to the configured `library_id`/`user_id`), `--offline` mode's `SqliteZoteroGateway` runs every query against the *entire* local `zotero.sqlite`, ignoring `library_id`/`library_type`/`user_id` entirely - anyone syncing more than one library locally (personal + any groups, a completely standard Zotero Desktop setup) silently gets results spanning all of them, and a Zotero item-key collision across libraries can return the wrong item. Properly scoping every query by `libraryID` would require resolving Zotero's local `libraries`/`groups` table schema, which isn't safely verifiable without a real Zotero Desktop database to test against - so per this audit finding's own explicitly-permitted fallback, `SqliteZoteroGateway.__init__` now prints a loud warning (stderr + `logger.warning`) at construction time, `--offline`'s own `--help` text now says so directly, and `docs/SETUP_GUIDE.md` gained a callout explaining the scope and its practical impact (item counts/PRISMA totals that don't match online mode). Found via the adversarial two-agent general/correctness audit tracked in #234.
112
+ - **Malformed `config.toml` was silently swallowed to an empty config, with no `is_valid()` gate anywhere (Issue #290):** `ConfigLoader._load_from_file` caught any TOML parse failure (syntax error, bad encoding) with a blanket `except Exception: print("Warning..."); return {}`, so the loader proceeded as if the file didn't exist and the real failure surfaced several layers deeper as an unrelated, misleading error (e.g. a generic "No target library defined" with no hint the actual cause was a TOML syntax error) - and no code path anywhere called `config.is_valid()` before command dispatch to catch this earlier. A genuine parse failure now raises `ConfigurationError(f"Failed to parse config file {path}: {e}")`; an `OSError` (permissions, file genuinely unreadable) keeps the pre-existing degrade-to-empty-config behavior, since that's not a config *authoring* mistake. `cli/main.py`'s `get_config(args.config)` call - previously outside its command-dispatch try/except entirely, so a raised `ConfigurationError` there would have produced a raw traceback instead of the clean handling - is now inside the same try/except, and skipped specifically for the `init` command (which never reads the global config and is the dedicated recovery path for a missing/broken config file - eagerly parsing here would otherwise block a user from ever reaching `init` to fix a broken `config.toml`). Found via the adversarial two-agent general/correctness audit tracked in #234.
113
+ - **Legacy (`library_id IS NULL`) jobs were never exclusively claimed by one library's queue (Issue #289):** Unlike #228's discovery-graph migration (which permanently claims the legacy file for the first library that touches it via a destructive rename), #150's job-queue scoping left `library_id IS NULL` jobs permanently visible to every scoped queue forever - `get_next_pending`'s claim (`UPDATE jobs SET status = 'PROCESSING' WHERE id = ?`) never stamped `library_id` onto the row, so a legacy job stayed poachable by every library's worker on every retry cycle, not just the first claim (confirmed: `jobs.sqlite` is one shared file for all libraries by default). A legacy job whose `item_key` actually belongs to library A could be picked up by library B's worker, calling library B's gateway with library A's item key - silently burning through `max_attempts` under the wrong library in the common case, or mutating/fetching the wrong item on an accidental key collision. `get_next_pending` now atomically stamps `library_id` onto a claimed legacy job in the same `BEGIN IMMEDIATE` transaction that pops it, so the first library to touch a legacy job claims it exclusively from then on - matching the one-time-claim guarantee #228's migration already provides. Found via the adversarial two-agent general/correctness audit tracked in #234.
114
+ - **`discovery_graph.json` legacy-migration rename had an unguarded TOCTOU race (Issue #288):** `ResolverFactory.get_snowball_graph_service` migrates a legacy, pre-#228 unscoped `discovery_graph.json` into a per-library scoped path via `if not storage_path.exists() and legacy_path.exists(): legacy_path.rename(storage_path)`, with no lock and no error handling. Two processes touching the same library before either has scoped storage yet (e.g. a background `slr snowball discovery` job and a second terminal invoking any snowball-graph-backed command at the same moment) could both pass the check before either renamed; the loser's `rename()` then raised an uncaught `FileNotFoundError`, crashing that process's command entirely instead of gracefully proceeding with the winner's already-migrated file. The rename is now wrapped in `try/except (FileNotFoundError, FileExistsError): pass` - either exception means another process already completed the migration (POSIX rename is atomic, so by the time a loser's own `rename()` call can fail with `FileNotFoundError`, the winner's `storage_path` is guaranteed to already exist), so the loser now safely falls through to using whatever `storage_path` already holds. Found via the adversarial two-agent general/correctness audit tracked in #234.
115
+ - **`RestoreService` stacked two full-library-scan lookups per restored item, plus refetched all collections per collection (Issue #277):** `_restore_collections` called `self.gateway.get_all_collections()` inside its per-collection loop instead of once before it; separately, `_find_existing_item` (called once per top-level item) did `get_items_by_doi` (a full library scan per #267) and, on a miss, also fell through to `get_all_items()` (a second full scan) for title matching - for M items and C collections, up to `C × (collection-list size)` + `M × 2 × (library size)` calls, worse than #267 since it stacked two scan types per item. `_restore_collections` now fetches `get_all_collections()` once before its loop. New `_build_existing_item_indexes()` does one `get_all_items()` pass, indexed by normalized DOI (via the same `normalize_doi()` used by #252/#270's DOI-matching fixes) and lowercased title; `restore_archive` builds both indexes once before the item-restore loop and threads them into `_find_existing_item`, which now does dict lookups instead of two more full-library re-scans per item - matching behavior is unchanged (DOI match first, title fallback on a miss), only the per-item network cost. Found via the adversarial two-agent performance audit tracked in #233.
116
+ - **`PurgeService.purge_attachments`/`purge_notes` made one `get_item_children` call per item, same class as #189 (Issue #276):** Both methods looped `for parent_key in item_keys: ... self.gateway.get_item_children(parent_key)` - one network round-trip per item being purged, O(N) sequential calls for a bulk purge over N items instead of O(1-2) batched scans. `slr/source_cmd.py`'s `_fetch_pdf_and_note_parent_keys` already fixed the identical bug class (#189) via one `search_items(item_type=...)` scan indexed by `parentItem`. New `PurgeService._get_children_by_parent` mirrors that pattern: for more than 2 parent keys, does one `search_items(item_type="attachment"|"note")` scan and groups results by `parent_item`; at or below that threshold it keeps the original per-parent `get_item_children` lookup, since a full-library scan would cost more than 1-2 direct lookups (this keeps `item purge`'s single-item path, the common case, exactly as cheap as before). A failed batched scan is treated as one error per requested parent, preserving the pre-batching per-item error-isolation behavior. Found via the adversarial two-agent performance audit tracked in #233.
117
+ - **CI workflow performance: no concurrency cancellation, double-run on PRs, no path filtering, no mypy cache (Issue #274):** `tests.yml` triggered on `[push, pull_request]` with no `paths`/`paths-ignore` and no `concurrency:` block, so every push to a branch with an open PR ran the full pipeline (lint, mypy, `pytest tests/unit --cov`, SonarQube scan + gate polling) twice - once for `push`, once for `synchronize` on `pull_request` - and superseded runs from rapid successive pushes kept running to completion instead of being cancelled. `mypy .` also re-type-checked the whole tree from a cold cache every run, since only `setup-uv`'s dependency cache was enabled. Added `concurrency: { group: ${{ github.workflow }}-${{ github.ref }}, cancel-in-progress: true }`; changed the `push` trigger to `branches: [main]` only (avoiding the duplicate push+PR run on feature branches with an open PR, while still running on every PR push via the `pull_request` trigger); added `paths-ignore: ['**.md', 'docs/**']` to both triggers (confirmed safe: `tests/docs`'s doc/repo-structure consistency checks aren't part of this workflow - only `tests/unit` is - so a docs-only diff genuinely can't affect what this pipeline exercises); and added an `actions/cache` step for `.mypy_cache` keyed on `pyproject.toml`/`uv.lock`. `release.yml`'s `build-linux`/`build-windows` jobs also gained `enable-cache: true` on their `setup-uv` steps, matching the pattern already used in `tests.yml`. (The Infisical-install-caching item from the original audit finding is now moot - the Infisical CI step itself was removed entirely, see the "Removed the Infisical CI dependency" entry above.) Found via the adversarial two-agent performance audit tracked in #233.
118
+ - **CLI startup paid ~130ms for eager metadata-client imports most commands never use (Issue #273):** `infra/metadata_client_factory.py` imported all ~20 external client/format modules at module level, including `bdtd_api.py` (pulls in `requests` + `bs4`/`soupsieve`, ~34-37ms) and `bibtex_lib.py` (`bibtexparser`, ~17-18ms) - measured via `python -X importtime -c "import zotero_cli.cli.main"`, the module's own cumulative import cost was ~62-64ms, out of `infra/factory`'s ~130-140ms total. Since nearly every `cli/commands/*.py` file imports `GatewayFactory` at module top level, a command like `zotero-cli item list` (or even `--help`) paid the full BDTD/bibtex/CrossRef/PubMed/etc. import cost even though it never constructs those clients. `ai_provider_factory.py`/`resolver_factory.py` already used function-local, `TYPE_CHECKING`-gated imports correctly - `metadata_client_factory.py` now follows the same pattern: every per-client import moved inside its corresponding `get_<x>_client()`/`get_<x>_gateway()` method. Post-fix, the module's own cumulative import cost dropped to ~0.14ms. Found via the adversarial two-agent performance audit tracked in #233.
119
+ - **Offline SQLite gateway: N+1 query pattern in the core item-fetch path, missing jobs-table index (Issue #270):** `SqliteZoteroGateway._fetch_items_with_filter` (the shared implementation behind `get_all_items`/`search_items`/`get_items_in_collection`/`get_trash_items`/`get_orphan_items`/`get_item`) ran one aggregate query for the item rows, then issued three more queries per row (creators/collections/tags), each redundantly re-resolving `itemID` from `key` via a nested subquery instead of reusing the `itemID` the outer query didn't even select - for N items this was 1 + 3N queries instead of ~4, e.g. ~6,001 statements to list 2,000 offline items, on the primary read path for every `--offline` command. Now selects `i.itemID` in the outer query and batch-fetches creators/collections/tags for the whole result set in 3 queries total (`WHERE itemID IN (...)`), building per-`itemID` lookup dicts before the final per-item yield - query count is now flat regardless of result-set size. Separately, `SqliteJobRepository`'s `jobs` table had no index beyond the implicit PK, despite `get_next_pending` filtering on `(task_type, status)` on every poll of `process_jobs`'s tight loop - added `idx_jobs_task_status` to the schema-creation SQL. Found via the adversarial two-agent performance audit tracked in #233.
120
+ - **FastAPI `serve` routes were `async def` but called synchronous gateway methods, blocking the event loop (Issue #269):** `api/routes/items.py` (`list_items`, `get_item`), `api/routes/collections.py` (`list_collections`), and `api/routes/jobs.py` (`list_jobs`, `get_job`) were all declared `async def` but called synchronous `gateway.*`/`job_queue.*` methods (backed by blocking `requests`/`sqlite3`). Because these routes are `async def`, FastAPI/Starlette ran them directly on the event loop instead of offloading to its threadpool (which only happens automatically for plain `def` routes) - one slow/hanging upstream Zotero API call stalled every other concurrent client, including the `/health` check, for the call's full duration. All five now wrap their blocking call in `starlette.concurrency.run_in_threadpool(...)`. Found via the adversarial two-agent performance audit tracked in #233.
121
+ - **Rate-limit pacing gaps: unpaced metadata-enrichment fan-out + CrossRef backward discovery missing #223-style throttle (Issue #268):** Of the 13 `infra/*_api.py` metadata-provider clients, only `pubmed_api.py` and `semantic_scholar_api.py` self-throttled before every request; the other 11 (crossref, openalex, dblp, hal, eric, zbmath, inspire_hep, doaj, bdtd, unpaywall, core) had zero proactive pacing, only reactive retry/backoff (which does nothing to stop a burst from tripping a rate limit in the first place) - `core_api.py` even documented a "3 req/s" cap in its own docstring without enforcing it anywhere. Separately, `SnowballDiscoveryWorker._discover_backward` (CrossRef) had no `asyncio.sleep` pacing, unlike its sibling `_discover_forward` (Semantic Scholar), which added `await asyncio.sleep(1.1)` under #223 - the fix was simply never backfilled to the CrossRef path. `BaseAPIClient` (the shared base every metadata-provider client already inherits) now self-throttles at a conservative default of ~3 req/s in `_get()` via a new `_apply_rate_limit()`, covering all 11 previously-unpaced clients from one place; `pubmed_api.py`/`semantic_scholar_api.py` opt out via `min_request_interval=0` since they already self-throttle more precisely (API-key-aware). `_discover_backward` now sleeps `1.1s` before its CrossRef call, mirroring `_discover_forward` exactly. Found via the adversarial two-agent performance audit tracked in #233.
122
+ - **`SnowballIngestionService.ingest_candidates` did a full paginated library scan per accepted candidate (Issue #267):** `_is_duplicate` called `item_repo.get_items_by_doi(doi)` - a full scan/paginated pass over the entire library - once per candidate in `ingest_candidates`'s loop, an O(N × library_size) cost instead of O(library_size + N) for an N-candidate import batch. The service's own docstring justified the per-call scan as "acceptably slow... normally called once per candidate," which is exactly backwards once the caller invokes it once per candidate in a loop. `cli/tui/snowball_tui.py`'s review step already solved this identically under #224 (`_build_library_doi_index`: one batched `get_all_items()` pass, reused for every candidate). `ItemRepository` gained a `get_all_items()` method (mirroring `ZoteroGateway`'s own) so `SnowballIngestionService` - which only held the narrower `ItemRepository`, not the full gateway - could do the same: `ingest_candidates` now builds one `Dict[normalized_doi, bool]` index up front via `_build_library_doi_index()`, and `_is_duplicate` does an O(1) lookup against it instead of a per-candidate scan (the old per-call `get_items_by_doi` scan is kept as a fallback for any caller that only needs a single one-off check). Found via the adversarial two-agent performance audit tracked in #233.
123
+ - **`tests/unit` hit live network/config, violating the `tests/unit`/`tests/e2e` safety boundary - and cost ~76% of the suite's wall time (Issue #275):** `tests/unit/cli/test_status_cmd.py::test_status_command_execute` mocked `GatewayFactory.get_zotero_gateway` but not `get_slr_status_service`, which `_handle_status` (`cli/commands/slr/report_cmd.py`) tries first - unmocked, it built a real gateway from real `~/.config/zotero-cli/config.toml` and hit the live Zotero API, with `infra/http_client.py`'s `stop_after_attempt(10)`/`wait_exponential(max=60)` retry policy accounting for ~124s of accumulated backoff in a single test. `tests/unit/infra/test_bdtd_client.py`'s `client` fixture only patched `client._get`, but `BDTDAPIClient._map_to_research_paper` also calls `_resolve_pdf_url_sync`, which makes an unmocked `requests.get` against whatever real URL is embedded in the test's own record data - two tests each cost ~10s doing a real HTTP round-trip. Measured before the fix: full `tests/unit` runs took 162-210s wall-clock against only ~12s of actual CPU time (~6% utilization) - almost entirely idle-waiting on live network, not compute-bound. Fixed by mocking `get_slr_status_service` directly in the first test, and patching `requests.get` at the `client` fixture level in the second (so no test in that file can reach real network without an explicit override, matching the one test that already correctly does this for `_resolve_pdf_url_sync`). `tests/unit` now completes in ~18s. Found via the adversarial two-agent performance audit tracked in #233 (critic pass, investigating a "would pytest-xdist help?" question the principal pass had left unsized - this fix made that question moot).
124
+ - **`slr snowball` subcommands also silently ignored the global `--user` flag (Issue #265):** Same bug as #257, in the sibling command tree: `force_user` was read from `args.user` in `slr/snowball_cmd.py` but only ever threaded into `get_snowball_ingestion_service()` for the `import` verb - `discovery`/`review`/`status`/`export`/`seed --from-accepted`'s `GatewayFactory.get_snowball_worker()`/`get_snowball_graph_service()`/`get_job_queue_service()` calls all omitted it, even though both factory methods gained a `force_user` parameter in #257. `ResolverFactory.get_snowball_ingestion_service()` itself had the same gap internally - it already used its own `force_user` parameter for `item_repo`/`col_repo`, but never passed it to its internal `get_snowball_graph_service()` call, so even the one call site that did thread `force_user` through didn't get it applied to graph-storage scoping. All of the above now thread `force_user` consistently. Found while fixing #257 (adversarial two-agent code-quality audit tracked in #232).
125
+ - **`system jobs` subcommands silently ignored the global `--user` flag (Issue #257):** `--user` ("Force Personal Library mode") is a global CLI flag every subcommand is expected to honor - every other handler in `cli/commands/system_cmd.py` reads `force_user = getattr(args, "user", False)` and threads it into its `GatewayFactory.get_*` call, but `_handle_jobs`/`_watch_jobs` never read `args.user` at all, so `zotero-cli --user system jobs run --type discover` (or `list`/`retry`/`watch`) silently operated against whatever library the cached global config resolved to instead. Tracing the fix surfaced that `get_job_queue_service`/`get_snowball_worker`/`get_snowball_graph_service` couldn't have honored `--user` even if it had been passed - unlike the online gateway's `ZoteroConfig.resolve_library_target(force_user)`, their per-library storage scoping (job queue DB, snowball discovery graph, from #150/#228) read `library_id`/`user_id` directly with no `force_user` parameter at all. New `ZoteroConfig.resolve_scoping_id(force_user)` mirrors `resolve_library_target`'s force-user semantics for this purely-local storage-scoping use case (never raises, unlike `resolve_library_target` - falls back to `"default"`); `get_job_queue_service`/`get_snowball_graph_service`/`get_snowball_worker` now accept and use it, and `system_cmd.py`'s job handlers now thread `force_user` through like every other handler in the file. Found via the adversarial two-agent code-quality audit tracked in #232 (surfaced by the critic pass while investigating a related config-threading question).
126
+ - **`console.print(f"...")` swallowed bracketed item/collection/query text across `cli/` (Issue #253):** Issue #208 fixed exactly one instance of this bug class (`slr/snowball_cmd.py`, via `markup=False`) - a real paper title, collection name, or user-typed query containing a literal `[...]` (e.g. "[Retracted]", "[Review]") gets silently consumed/reinterpreted by Rich's console-markup parser as a style tag instead of rendered as text. The fix was never generalized: `markup=False`/`rich.markup.escape()` appeared nowhere else across roughly 126 `console.print(f"...")` call sites in `cli/`. Every interpolated title/collection-name/query/error-message site identified by the audit (`search_cmd.py`, `item_cmd.py`, `collection_cmd.py`, `report_cmd.py`, `system_cmd.py`, `slr/sdb_cmd.py`, `slr/report_cmd.py`, `slr/reconcile_cmd.py`, `slr/snowball_cmd.py`, `slr/load_cmd.py`, `rag_cmd.py`, `cli/tui/extraction_tui.py`, `cli/tui/screening_tui.py`) now wraps the interpolated data segment in `rich.markup.escape()`, keeping the surrounding literal Rich markup (colors/bold) intact. `rich.text.Text`-based displays (e.g. the screening TUI's abstract panel) were confirmed already safe - `Text()` doesn't parse markup, unlike an f-string passed to `console.print()`. Found via the adversarial two-agent code-quality audit tracked in #232.
127
+ - **Offline SQLite gateway's `get_items_by_doi` didn't normalize DOIs (Issue #252):** `SqliteZoteroGateway.get_items_by_doi` (`--offline` mode) did a raw exact-match SQL query (`WHERE f.fieldName = 'DOI' AND dv.value = ?`) using the literal, un-normalized `doi` argument, while the online `ZoteroAPIClient.get_items_by_doi` was rewritten in #221 (reopened #205) to abandon Zotero's `q=` search entirely and do a full client-side scan with `normalize_doi()` on both sides, since Zotero's search API doesn't reliably index the DOI field. The offline gateway was left running the architecturally-superseded approach: in `--offline` mode, duplicate-DOI detection (hit by `restore_service.py` and the snowball import dedup path) silently missed matches whenever the stored DOI differed in case or carried a `https://doi.org/` prefix vs. the bare queried DOI. `get_items_by_doi` now mirrors the online implementation exactly - scan `get_all_items()` and compare via `normalize_doi()` in Python rather than trusting an exact SQL string match. Found via the adversarial two-agent code-quality audit tracked in #232.
128
+ - **Predictable shared-tmp-dir filenames across the PDF-resolver family (Issue #240):** Every PDF resolver (`unpaywall.py`, `semantic_scholar.py`, `arxiv.py`, `generic_scraper.py`, `openalex.py`, `bdtd.py`), `zotero_api.py`'s thesis-import PDF fetch, and `sqlite_repo.py`'s offline-mode shadow-copy path all built their temp-file path by manually joining `tempfile.gettempdir()` with a fully deterministic filename (`f"unpaywall_{item.key}.pdf"`, `f"zotero_cli_shadow_{os.getpid()}.sqlite"`, etc.) instead of using `tempfile.mkstemp()` (already used correctly elsewhere, e.g. `restore_service.py`, `attachment_service.py`). On a shared multi-user host with a world-writable `/tmp`, a local attacker could pre-create a symlink at that guessable path - item keys are visible to any library collaborator, and PID space is bounded/reused - causing the write to follow the symlink and overwrite an arbitrary victim-owned file. New shared `core/utils/safe_tempfile.py` (`write_secure_temp_file()`) wraps `tempfile.mkstemp()` (which creates the file itself, `O_CREAT|O_EXCL`, mode `0600` - nothing for an attacker to have pre-planted) and is now used at all 6 resolver sites; `zotero_api.py` and `sqlite_repo.py` call `tempfile.mkstemp()` directly since they stream/copy into the file descriptor rather than writing a single `bytes` payload. Found via the adversarial two-agent security audit tracked in #231 (principal found 3 instances, the critic pass found 5 more identical ones).
129
+ - **Unbounded response body reads on every PDF-download code path (Issue #239):** Neither `requests` nor `httpx` caps response body size by default, and every PDF-fetch code path in the resolver chain (every `core/services/resolvers/*.py` PDF resolver via `NetworkGateway`, `attachment_service.py`'s existing-URL check, and `zotero_api.py`'s thesis-import PDF fetch) either buffered the whole body into memory or streamed it to disk with no cap - a malicious/compromised metadata source, or (combined with #235's SSRF finding) an attacker-controlled `item.url`, could serve an arbitrarily large or slow-drip response and exhaust memory or disk on the machine running `item pdf fetch`/`item attach-pdfs`. `core/utils/url_safety.py` now enforces a shared 50MB `MAX_RESPONSE_BYTES` cap via an upfront `Content-Length` pre-check plus a hard streaming cutoff that also catches a server lying about or omitting that header: `iter_capped_content()` wraps the sync (`requests`) streaming path used by `attachment_service.py`/`zotero_api.py`, and `read_capped_async()` wraps the async (`httpx`) path, now used internally by both `safe_async_get()` (switched from `client.get()` to `client.send(..., stream=True)` so the cap can actually abort mid-download rather than after the fact) and `NetworkGateway._fetch_validated` (same `build_request`/`send(stream=True)` switch), covering every resolver that routes through the gateway for free. A response that trips the cap raises `ResponseTooLargeError`; the two sync download sites also now clean up their partially-written temp file on any failure (previously only handled on a successful-but-non-PDF response), closing a related disk-leak this cap would otherwise still leave behind. Found via the adversarial two-agent security audit tracked in #231.
130
+ - **Zip-slip via attachment filename in `.zaf` backup archives (Issue #238):** `BackupService._download_attachment_file` built the zip entry's arcname as `f"attachments/{p_key}/{filename}"` using `filename = data.get("filename") or data.get("title")` — Zotero attachment metadata settable by any collaborator with write access to a shared library — with no sanitization. A crafted filename containing `../` path-traversal segments (e.g. `"../../../../etc/passwd"`) could place the written entry outside the intended `attachments/<parent>/` directory inside the archive. Now sanitized via `os.path.basename()` before use, with a fallback to the item's own key when that sanitizes down to an empty or dot-only result (e.g. a filename of just `..`). `RestoreService`'s corresponding read path was independently confirmed already safe by both audit agents (it uses `zf.read()` on named manifest entries, never `zipfile.extractall()`), so no read-side change was needed. Found via the adversarial two-agent security audit tracked in #231.
131
+ - **CSV export enables spreadsheet formula injection (Issue #237):** Every CSV-writing code path in the project (`report_cmd.py`'s duplicate-report export, `slr list qa-approved --csv`, `extraction_service.py`'s synthesis-matrix export, `sync_service.py`'s screening-state recovery, `merge_plan_io.py`'s merge-plan export, `screening_state.py`, and `canonical_csv_lib.py`) wrote Zotero item fields (title, abstract, etc. - settable by any collaborator with write access to a shared library) directly into cells with no sanitization, enabling CSV/spreadsheet formula injection (e.g. `=HYPERLINK(...)`) when a reviewer opens the export in Excel/LibreOffice/Google Sheets. New shared `core/utils/csv_safety.py` (`sanitize_csv_cell`/`sanitize_csv_row`/`sanitize_csv_rows`) prefixes any cell value starting with `=`, `+`, `-`, `@`, tab, or CR with a leading `'`, per OWASP's recommended mitigation, applied at all 7 confirmed writer sites. Since `canonical_csv_lib.py`'s format is both written *and re-read* by this project (`system normalize`), also added `unsanitize_csv_cell()` on its read path so re-importing a file this project itself exported doesn't treat the escape marker as literal data.
132
+ - **Credential-bearing config.toml written world-readable (Issue #236):** `config.toml` holds live API keys/tokens (Zotero, OpenAI, Gemini, Hugging Face, Semantic Scholar, CORE, NCBI) but was written with a plain `open(path, "w")`, inheriting the process's default umask - empirically confirmed to leave it `0644` (world-readable) on a real config file. Any other local user on a shared multi-user machine could read every credential this tool has stored. New shared `secure_config_open()` helper (`core/config.py`) opens the file via `os.open` with an explicit `0600` mode set at creation, creates the parent directory `0700`, and `chmod`s the file descriptor too (the `os.open` mode argument is silently ignored for a *pre-existing* file, so a config written by an older version of zotero-cli gets its permissions tightened the next time it's updated, not left insecure). Applied to both write sites: `ConfigManager.update_config` (used by `system switch`/other config-mutation flows) and `init_cmd.py`'s first-run config-creation wizard.
133
+ - **SSRF + exfiltration via unvalidated URL fetch across every PDF-resolver code path (Issue #235):** Every PDF-acquisition path reachable from `item pdf fetch`/`item attach-pdfs` (`attachment_service.py`'s existing-URL check, every `core/services/resolvers/*.py` PDF resolver, and `zotero_api.py`'s thesis-import PDF fetch) fetched an externally-influenceable URL — `item.url` (attacker-settable by any collaborator with write access to a shared library) or a third-party API's `pdf_url` — with no scheme/IP validation, then uploaded the response back onto the Zotero item as an attachment, readable by that same collaborator. New shared `core/utils/url_safety.py` (`validate_public_url`/`safe_get`/`safe_async_get`) rejects non-`http(s)` schemes and any hostname resolving to a loopback/private/link-local/multicast/reserved address, re-validated on *every* redirect hop rather than trusting `requests`/`httpx`'s default redirect-following (a validated public URL could otherwise still redirect into an internal address). `NetworkGateway` (used by `bdtd.py`/`generic_scraper.py`/`unpaywall.py`/`semantic_scholar.py`/`arxiv.py`'s resolvers) now routes every request through this guard centrally; `attachment_service.py`, `zotero_api.py`'s thesis path, and `openalex.py`'s standalone httpx client were updated individually since they don't share the gateway. Also added a `%PDF` magic-byte check before upload at every site that lacked one (`attachment_service.py`, `openalex.py`, `zotero_api.py`), checked on the first streamed chunk rather than re-reading the file afterward. Found via the adversarial two-agent security audit tracked in #231.
134
+
135
+ ## [2.8.8] - 2026-09-07
136
+
137
+ ### 🐛 Bug Fixes
138
+ - **`SnowballGraphService` had no per-library storage scoping (Issue #228):** Every other stateful snowballing service already supports it - `JobQueueService` filters by `library_id` (Issue #150), and `get_snowball_worker`/`get_snowball_ingestion_service` both accept an optional `ZoteroConfig` - but `get_snowball_graph_service()` took no config at all and always resolved to one process-global `discovery_graph.json`. Any two callers targeting different Zotero libraries (a multi-tenant caller running one discovery graph per research project, or even a single CLI user who `system switch`ed between groups mid-session) would silently read/write the identical file, corrupting each other's candidate graphs. `get_snowball_graph_service` now accepts an optional `config: Optional[ZoteroConfig]`, resolving to `discovery_graph_{library_id}.json` when one is supplied (falling back to `user_id`, then `"default"`) and the pre-#228 global path when it isn't - `get_snowball_worker`/`get_snowball_ingestion_service` now thread their own resolved config through, and `SnowballCommand` resolves one config per CLI invocation and passes it to every snowball-graph call site (seed/discovery/review/import/status/export), so the whole CLI is consistently scoped too, not just multi-tenant callers. A pre-existing global `discovery_graph.json` is migrated (renamed, not just pointed at) into the first library that touches it after upgrading, so existing local state isn't silently orphaned.
139
+
140
+ ## [2.8.7] - 2026-09-07
141
+
142
+ ### ✨ Features & Improvements
143
+ - **Snowball review flags candidates already present in the library (Issue #224):** Library membership was only ever checked at import time (`SnowballIngestionService._is_duplicate`) - a researcher reviewing candidates had no way to know some were papers they already had until after accept + import, when the item was silently skipped with just a lower "N imported" count than "N accepted". `SnowballReviewTUI` now builds a normalized-DOI -> Zotero-key index in one batched `get_all_items()` pass at the start of a review session (reusing the same `normalize_doi()`-based matching #205 introduced, so a bare-DOI candidate still matches a library item stored in URL-form), and flags each candidate's metrics panel with `⚠ Already in library (KEY)` instead of presenting it identically to a genuinely new candidate. Optional (`gateway` param defaults to `None`) so a caller without one still gets a working, just un-annotated, review session. Export (`--format json`/`mermaid`) intentionally left untouched per the issue's own scoping - not urgent, a natural follow-on.
144
+
145
+ ### 🐛 Bug Fixes
146
+ - **Forward snowballing 403s and fails outright whenever `semantic_scholar_api_key` is configured (Issue #223):** Confirmed live against the real Semantic Scholar API: a configured key can be rejected (403) across every S2 endpoint while the identical unauthenticated request succeeds - `NetworkGateway`'s existing 403 handling (rotate identity, retry once) can never fix a bad credential, so it just surfaced a raw `httpx.HTTPStatusError`/traceback for what's really a rejected-key condition. `NetworkGateway._execute_request` now recognizes this case (a 403 that survives identity rotation *with* an API-key/auth header present) and raises a clear, actionable `ValueError` instead. `SnowballDiscoveryWorker._discover_forward` catches that specifically and falls back to a single unauthenticated retry rather than failing the whole job - turning a hard failure (0 candidates, every time a key is configured) into degraded-but-working behavior, with a logged warning. Also added the proactive `time.sleep`-equivalent pacing (`asyncio.sleep(1.1)`, matching `infra/semantic_scholar_api.py`'s own existing 1-req/sec self-throttle) that forward discovery never had - the only backoff that existed before was reactive (per-job, only after a 429 already happened), which isn't enough to avoid tripping Semantic Scholar's low, globally-shared unauthenticated pool once the fallback path is exercised.
147
+
148
+ ## [2.8.6] - 2026-09-07
149
+
150
+ ### 🐛 Bug Fixes
151
+ - **Reopened Issue #205 - `get_items_by_doi` structurally cannot find matches via Zotero's search API:** The v2.8.5 fix normalized DOI *format* in the comparison, but Corbenic-SLR's re-test (verified directly against the live Zotero Web API, not just this codebase) found `_is_duplicate` still couldn't detect any duplicate: `get_items_by_doi` relies on `search_items(ZoteroQuery(q=doi))`, and Zotero's `q`/`qmode` search does not index the structured DOI field under either `qmode` value (`titleCreatorYear` or `everything`) - it always returns zero results for a DOI-only query, regardless of format. `ZoteroAPIClient.get_items_by_doi` now does a client-side scan instead: paginate `get_all_items()` and filter locally with `normalize_doi()` - the only approach that actually works given this Zotero API limitation. This also transitively fixes the same structural gap in `RestoreService` and `search --doi`, both of which call the same shared `get_items_by_doi`. Offline mode (`SqliteZoteroGateway`) was never affected - it already queries the local DOI column directly via SQL.
152
+
153
+ ## [2.8.5] - 2026-09-06
154
+
155
+ ### ✨ Features & Improvements
156
+ - **Snowball accept/reject decisions now capture a reason and evidence depth (Issue #211):** `SnowballGraphService.update_status` recorded a bare status transition with no audit trail of *why* - inconsistent with this project's own SDB screening-decision pattern (`ScreeningService.record_decision`, which captures a criteria code alongside every include/exclude). `update_status` now accepts optional `reason: str` and `depth: Literal["title", "abstract", "full_text"]` params, storing them on the graph node as `decision_reason`/`decision_depth`; `SnowballReviewTUI` prompts for both after every accept/reject (depth defaults to `abstract`, reason is optional free text). Surfacing these in `get_stats()`/`to_mermaid()`/export is left as a natural follow-on now that they're captured, not done here.
157
+ - **Snowball review hydrates un-titled stub candidates before display (Issue #210):** `slr snowball review`'s candidates came straight off the graph as discovered — for backward/CrossRef candidates this is frequently just a generic `"Reference from {parent-doi}"` stub with no abstract (CrossRef reference lists very often omit `article-title`/`unstructured`), making an informed accept/reject decision impossible without looking the paper up externally. `SnowballReviewTUI` now calls the existing `MetadataAggregatorService.get_enriched_metadata(doi)` (already used by `SnowballIngestionService._hydrate_paper` at import time, just never wired into review) to backfill title/abstract/authors/year per-candidate as the TUI advances, whenever a candidate looks like an un-hydrated stub - and persists the title/abstract back onto the graph node so re-reviewing (or the later import) doesn't re-fetch the same metadata. Falls back to the un-hydrated view (no crash) if no metadata service is configured.
158
+ - **`slr snowball seed --dois`/`--from-accepted`: seed the next generation without an import round trip (Issue #206):** Wohlin's snowballing method is iterative (seed → discover → review → re-seed the next generation from this generation's accepted candidates → repeat), but `seed` previously only accepted `--keys`/`--collection`, both resolving *existing Zotero items* — a candidate accepted in generation N is only a graph node with a DOI, not yet a Zotero item, so there was no way to pass it back into `seed` for generation N+1 without importing every generation before starting the next one. Added `--dois` (comma-separated bare DOIs, skips the Zotero-item lookup entirely) and `--from-accepted [--from-generation N]` (reads the graph's own `ACCEPTED` nodes directly via a new `SnowballGraphService.get_accepted_dois()`, optionally scoped to one prior generation) — the latter is the "one-click next generation" flow closest to Wohlin's own guideline. `docs/help_specs/slr_snowball.md`'s parameter matrix and examples updated (and its pre-existing broken `--doi` example, which referenced a flag that never existed, corrected to the real `--dois`/`--keys` flags along the way).
159
+
160
+ ### 🐛 Bug Fixes
161
+ - **`system info`'s Group URL could point at a stale, inactive group (Issue #209):** The "Group URL" line printed `config.target_group_url` verbatim, but that's a separately-stored config value only ever consulted as a fallback when `library_id` itself is unset (`ZoteroConfig.resolve_library_target`'s priority cascade) - once `library_id` is set, `target_group_url` is inert and unrelated to what's actually active. `system switch` updates `library_id`/`library_type` but never touches `target_group_url`, so after switching, `system info` could show a real, different group's URL right next to the correct numeric Library ID - exactly the mistake `system switch`'s own documented safety tip warns against (scan the more legible "Group URL" line, land on the wrong library). Now derives the Group URL directly from the same `library_id`/`library_type` already printed above it (`https://www.zotero.org/groups/{library_id}`), so drift between the two lines is no longer possible.
162
+ - **`slr snowball export --format json` produced no output at all (Issue #208):** `_handle_export`'s `json` branch was a literal `pass` - no output, no error, exit 0, on both an empty and a populated graph. Added `SnowballGraphService.to_json()` (reuses the same `nx.node_link_data` shape `save_graph()` already persists to disk, mirroring the existing `to_mermaid()`) and wired it into the `json` branch. While in there: switched the export print path to `markup=False`, since a real paper title or Mermaid node label routinely contains literal `[...]` that Rich's console markup parser would otherwise silently swallow as a (nonexistent) style tag, corrupting the printed JSON/Mermaid.
163
+ - **`ZoteroAPIClient.get_item` gave an opaque crash for a malformed/invalid item key (Issue #207):** `slr snowball seed --keys <bad-key>` (e.g. a bare DOI passed where a Zotero item key was expected) printed a raw `'list' object has no attribute 'get'` — a malformed key can route to an endpoint that returns a list instead of a single item object or a 404, and `ZoteroItem.from_raw_zotero_item` then crashed calling `.get()` on it. `get_item` now checks the response shape and fails with a clear `'<key>' is not a valid Zotero item key` message instead.
164
+ - **Snowball import created duplicate Zotero items for URL-form DOIs (Issue #205):** `SnowballIngestionService._is_duplicate` lowercased both sides of a DOI comparison but never normalized *format* — a bare DOI already in the library (`10.1109/tse.2026.3694876`) and a URL-form DOI a metadata provider hydrated for the same paper during a snowball import (`https://doi.org/10.1109/tse.2026.3694876`) compared as unequal, so the paper was silently re-imported as a brand-new duplicate item. Switched to the project's existing `normalize_doi` helper (`core/utils/normalization.py`, already used by `slr csv_inbound`'s DOI matching) on both sides of the comparison instead of a bespoke `.lower()`-only check.
165
+ - **Forward snowballing never discovered any candidates (Issue #204):** `SnowballDiscoveryWorker._discover_forward`'s Semantic Scholar `/citations` request omitted `externalIds` from its `fields` param, so every `citingPaper.externalIds.DOI` lookup came back empty and every single forward candidate was silently dropped — confirmed live: a seed with 20 real citations (17 with a DOI) added 0 nodes via `slr snowball discovery`, with the job still reporting `COMPLETED`. Added `externalIds` to the requested fields; no other Semantic Scholar call site in the codebase shares this gap (`infra/semantic_scholar_api.py` already requests it).
166
+
167
+ ## [2.8.4] - 2026-09-05
168
+
169
+ ### ✨ Features & Improvements
170
+ - **`ScreeningService.record_decision` split into composable primitives (Issue #201):** `record_decision` used to unconditionally write the persona's audit note *and* apply shared, item-level tags/collection-movement in one call — fine for solo screening, but wrong for double-blind screening: two independent raters calling it on the same item would each fire the shared tags/move, driven by whichever called last, even when they disagreed. Split into `record_decision_note` (note write/upsert only, safe to call per-persona with no shared side effects), `apply_decision_outcome` (tags + collection move only, meant to be called once a decision is *final* — immediately after a solo decision, or after a double-blind pair is reconciled), and `get_decisions_for_item` (returns every persona's parsed decision, so a caller can check who decided what without hand-rolling note scanning). `record_decision` itself stays as a thin, non-breaking wrapper (`record_decision_note` + `apply_decision_outcome` in sequence) — every existing solo-screening caller (CLI, TUI) is unaffected.
171
+ - **API clients for CORE, Semantic Scholar search/count, and DOAJ (Issue #190):** Three new/extended library-level metadata sources, following the exact `ArxivGateway`/`SearchableMetadataProvider` pattern already established, for a downstream consumer (Corbenic-SLR) building live "search + count + import" catalog pages: OpenAlex/PubMed/ERIC/INSPIRE-HEP already had clients, so no ask there.
172
+ - **`CoreAPIClient`** (new, `infra/core_api.py`): CORE (core.ac.uk)'s open-access aggregator search API v3. Requires a free API key (`config.core_api_key`/`CORE_API_KEY` env var, registration at core.ac.uk/services/api; free tier is 3,000 req/month, 3 req/s) — deliberately *not* registered with `MetadataAggregatorService`'s always-on lookup fan-out, to avoid burning that budget on every single DOI lookup regardless of whether CORE results are wanted; reachable only via the new standalone `GatewayFactory.get_core_client()`.
173
+ - **`SemanticScholarAPIClient.count(query)`** (new): the client already had `search()` (#179); adding `count()` closes the gap the issue called out — a result count without paginating full records, reading the same `total` field `search()` already uses.
174
+ - **`DOAJAPIClient`** (new, `infra/doaj_api.py`): the Directory of Open Access Journals' search API, no API key required.
175
+ - New shared `CountableMetadataProvider` interface (`core/interfaces.py`), mirroring `SearchableMetadataProvider`'s "optional capability mixin" shape — `count(query) -> int`, implemented by all three of the above.
176
+ - `core/strategies.py`'s `BdtdImportStrategy` (added in #182) renamed to `SearchableProviderImportStrategy` and generalized: it always only depended on the shared `SearchableMetadataProvider.search()` shape, not anything BDTD-specific, so the same class now serves CORE/Semantic Scholar/DOAJ too without duplicating three near-identical wrapper classes.
177
+ - Library-level only, per the issue's own framing (Corbenic builds its own "Import from CORE/Semantic Scholar/DOAJ" UI on top of these) — no new CLI verb in this change.
178
+ - **`import bdtd --query`: free-text bulk import from BDTD (Issue #182):** `import bdtd` previously only accepted a single BDTD record ID, repository handle URL, or DOI. `BDTDAPIClient` now additionally implements `SearchableMetadataProvider` (the same interface added for OpenAlex/Semantic Scholar in #179) via BDTD's VuFind search endpoint, and `import bdtd --query "<free text>" --collection ... --limit N` bulk-imports up to `N` matching theses/dissertations through a new `BdtdImportStrategy`. Bulk search results deliberately skip the synchronous per-record PDF-URL scraping `get_paper_metadata` normally does for a single identifier — scraping+HEAD-probing every landing page for up to `N` records would be far too slow for a search preview — deferring PDF resolution to the existing async `BDTDResolver`, run later via `item pdf fetch`/the normal job pipeline. Exactly one of `identifier`/`--query` must be given.
179
+ - **`ArxivGateway.count(query)` (Issue #181):** Returns the total result count for a query without fetching/discarding individual `ResearchPaper` objects — reads `<opensearch:totalResults>` straight off a single-item Atom feed page, the same total the `arxiv` package's own `Client._results()` already fetches internally as part of the first page (`feed.header.total_results`) but never surfaced through `ArxivGateway`'s public interface. Enables a read-only "About N results" exploratory-search UI (Corbenic-SLR's use case) that judges a candidate query's breadth before committing to a real import, without the bandwidth/rate-limit cost of a capped fetch-and-count stopgap.
180
+ - **RAG/AI dependencies split into an optional `[rag]` extra — "lite" install (Issue #180):** `torch`, `sentence-transformers`, `huggingface-hub`, `numpy`, `einops`, `accelerate`, `openai`, and `google-generativeai` moved out of the base `dependencies` into `[project.optional-dependencies].rag`. Measured impact for a consumer like Corbenic-SLR that only needs Zotero I/O/SLR/screening/extraction, not RAG: a Docker image pinned to `v2.8.3` was 11.5GB (CUDA `torch` wheel); this alone doesn't fix a wheel-index choice, but removing the whole RAG stack from a lite install's dependency tree is the bigger win the pinned-CPU-wheel workaround couldn't reach. Every import of these packages in `src/` was already function-local/lazy (verified by reading the code, not assumed) — so this was purely a `pyproject.toml` packaging change, not the dependency-injection rework flagged as a future concern when Issue #154 was decided; a lite-install consumer who does hit a RAG code path gets a clear `ImportError`, not a crash. `markitdown` (and its `onnxruntime` dependency) deliberately stayed in the base dependencies despite being large, since `item`/`collection export --format md` — a non-RAG feature — depends on it too; moving it would have broken that command for lite installs. `dev` now self-references `zotero-cli[rag]` so `uv sync --extra dev`/CI keep installing everything the full test suite needs, unchanged.
181
+ - **Free-text/topic search for OpenAlex and Semantic Scholar (Issue #179):** `OpenAlexAPIClient`/`SemanticScholarAPIClient` previously implemented only `get_paper_metadata(identifier)` — resolving an already-known DOI/arXiv-ID/S2-ID, with no way to search by topic the way `ArxivGateway.search` already allows. Both now additionally implement a new `SearchableMetadataProvider` interface (a separate, additionally-inherited mixin — not every one of this project's 11 metadata sources supports free-text search, so it isn't a `MetadataProvider` method) with a `search(query, max_results=100, sort_by="relevance", sort_order="descending") -> Iterator[ResearchPaper]` method mirroring `ArxivGateway.search`'s shape, transparently paginating each API's native page-size cap (200 for OpenAlex, 100 for Semantic Scholar) up to `max_results`. Neither requires an API key. Library-level only per the issue's own scoping — no new CLI verb in this change.
182
+ - **Library-independent API key identity resolution (Issue #178):** `ZoteroAPIClient.resolve_key_identity(api_key)` (new `@staticmethod`, no instance/`library_id` required) calls Zotero's `GET /keys/<api_key>` — the one REST endpoint that needs no library scope — and returns a typed `KeyIdentity(user_id, username, access)`. Closes a chicken-and-egg gap for any account-setup flow (Corbenic-SLR's included) that needs to validate a researcher's API key and discover their personal `userID` *before* any library/group is known. `zotero-cli init` now uses it too: the wizard resolves and confirms the key's identity immediately after it's entered, prefilling the Library ID prompt (user mode) and the personal User ID prompt (group mode) with the resolved `userID` instead of asking the user to go look it up separately; resolution failures degrade to the prior plain-prompt behavior rather than blocking setup.
183
+ - **`GET /jobs`/`GET /jobs/{id}` API routes (Issue #150):** `serve`'s FastAPI layer now exposes read-only background-job status (queued `fetch_pdf`/snowball-discovery jobs and their retry state), so an external consumer like Corbenic-SLR can poll job progress through the API instead of reading `jobs.sqlite` directly — closing the same "serve API only has collections/items" gap that blocked this and several other corbenic-facing features.
184
+ - **`item trash`/`item restore`, Zotero-Desktop-compatible (Issue #145, Phase 1):** New offline-only commands that move an item to/from the trash by writing directly to the local `zotero.sqlite`, replicating exactly what Zotero Desktop's own client writes (confirmed against Desktop's real source: `Zotero.Items.trash()`/`trashTx()` and `item.deleted = false; item.save()`) — bumps `dateModified`/`clientDateModified`, marks the row dirty (`synced=0`) so Desktop's next real sync pushes the change to the server, and adds/removes a `deletedItems` row. `version` is deliberately left untouched, matching Desktop (only the server bumps it on sync). This is the first write path ever added to the previously fully-read-only `SqliteZoteroGateway`; every other offline mutation still raises `Offline mode is read-only`. Preview-only by default (`--execute` required, confirmation prompt unless `--force`); rejected outright against an online/API gateway, which has no documented reversible trash write. Does not replicate Desktop's merge-relations cleanup on restore (stripping `dc:replaces` relations from a prior `item merge`) — narrow edge case, documented as a known limitation rather than guessed at. Phase 2 (online mode) is intentionally out of scope for this change.
185
+
186
+ ### 🐛 Bug Fixes
187
+ - **`slr source list` no longer makes one Zotero API call per item (Issue #189):** `_handle_list` called `gateway.get_item_children(item.key)` once per item, purely to detect a PDF attachment and an SDB-note child — for a source with a few hundred items, a few hundred sequential `GET /items/<key>/children` round trips, making the command take minutes rather than seconds against the Zotero Cloud API. Replaced with two library-wide, paginated `search_items(ZoteroQuery(item_type=...))` scans (`attachment`, `note`) done once up front regardless of source/item count, building parent-key sets checked with an O(1) membership test per item instead of a network call. Extends the fix corbenic-slr already shipped for its own mirrored PDF-detection logic (which only handled the PDF side) to also cover the SDB-note check the same way, since both use the identical per-item pattern.
188
+ - **`item pdf attach`/`upload_attachment` reliably 428s (Issue #191):** `ZoteroApiClient.upload_attachment` failed at one of two steps against the real Zotero Web API, root-caused via direct diagnostic calls against a live library. Step 2 (upload authorization) sent both `If-None-Match: *` and `If-Unmodified-Since-Version` — Zotero 428s that combination, since there's no prior version of a just-created attachment to be "unmodified since"; removed the stale header (the comment above it citing #79 was a mislabeled leftover — #79 is unrelated CLI-parameter-naming work). Step 4 (registering the upload) was missing `If-None-Match: *` entirely, which Zotero also rejects with 428 (`"If-Match/If-None-Match header not provided"`); added it. A failure at either step also now cleans up the orphaned, empty attachment placeholder item created in step 1, instead of leaving it behind for the user to delete by hand.
189
+ - **`tests/docs` was silently vacuous when run in isolation (Issue #147):** `test_doc_consistency.py` imported `CommandRegistry` but never `zotero_cli.cli.main` — every CLI command only self-registers with `CommandRegistry` when its own module is imported, so `CommandRegistry.get_commands()` was empty whenever `pytest tests/docs` ran standalone (exactly how the project's own documentation-consistency protocol instructs it to be run). 5 of the 8 tests in that file walk the registry and so passed vacuously, checking nothing; only the 2 `*_no_orphans` tests (which check the opposite direction — docs → registry) failed, because real files on disk made the empty registry visible instead of silent. Fixed by importing `zotero_cli.cli.main` directly; all 8 tests now genuinely pass. Running the now-real reachability check by hand (per the new protocol below) also found and fixed 4 documented example commands (`item_trash.md`, `item_restore.md`, `docs/commands/item.md` ×2, plus the CLI's own `--help` epilogs for `item trash`/`item restore`) that put the global `--offline` flag *after* the subcommand — argparse rejects that with `unrecognized arguments: --offline`, since global flags must precede the subcommand. None of those examples were actually runnable as written.
190
+ - **Super-linear regex backtracking in `MarkdownRecursiveSplitter` (SonarQube `python:S8786`):** `rag_service.py`'s markdown-header regex used `\s+` (which also matches `\n`) directly against a `[^\r\n]*$` tail under `re.MULTILINE`, giving the engine an ambiguous boundary across newlines. Narrowed to `[ \t]+` — markdown headers never have a literal newline between the `#`s and the header text, so this isn't a behavior change, just removes the backtracking hazard. Found while investigating an unrelated CI quality-gate failure on PR #183; fixed here rather than filing a separate issue since it was the one thing blocking that gate.
191
+ - **`SqliteZoteroGateway`'s read path now matches real Zotero Desktop databases (Issue #174):** Offline mode's entire read surface (`search_items`, `get_all_items`, `get_orphan_items`, `get_trash_items`, `get_items_in_collection`, `get_all_collections`, `get_item_children`, and creator/author resolution) was built and tested exclusively against a hand-typed mock schema that didn't match a real `zotero.sqlite` — verified against an actual Zotero Desktop database (extracted from a local flatpak install) that essentially all offline reads crashed (`no such column: i.parentItemID`, `no such table: collectionData`) or silently mis-resolved creators (`creatorData` table doesn't exist; `creators.firstName`/`lastName` live directly on the row). Fixed the SQL to match: attachment/note parent linkage now resolves via `itemAttachments`/`itemNotes` (real Zotero has no `items.parentItemID`), collection trees resolve `parentCollectionID` (an integer FK) to the parent's key string via a self-join (no `collectionData`/`parentCollection` column), and creator lookups read `firstName`/`lastName` straight off `creators`. Rebuilt both test fixtures (`tests/unit/test_sqlite_repo.py`, `tests/unit/infra/test_sqlite_repo_extended.py`) to match the real schema exactly instead of the previous self-consistent-but-wrong shape, and verified end-to-end against a disposable copy of a real ~15k-item `zotero.sqlite` (never the live file).
192
+ - **`collection purge` unreachable (Issue #146):** Registers the `purge` subparser that `_handle_purge` was missing (same dead-dispatch-branch bug class as #161) — `collection purge --name <NAME> --files --notes --tags` now actually works. Also corrected `docs/commands/collection.md`'s stale positional-argument example to the `--name` flag convention every other `collection` verb uses, and added `docs/help_specs/collection_purge.md` per the DOC-SPEC template.
193
+
194
+ ### 🛡️ Quality & Infrastructure
195
+ - **SonarQube `python:S5958` cleanup — specific exception type in a VerifyService test (Issue #197):** `test_verify_service_calculate_checksum_error` mocked `zipfile.ZipFile.open` to raise a bare `Exception` and asserted with `pytest.raises(Exception)` — a catch-all that would silently pass even if the wrong kind of failure occurred. `VerifyService._calculate_checksum` doesn't wrap or narrow whatever `zf.open()` raises, so the mock now raises `zipfile.BadZipFile` (what real `zipfile` actually raises for a corrupt/unreadable entry) and the assertion narrows to that same type.
196
+ - **SonarQube `python:S8519` cleanup — `next(iter(...))` instead of `list(...)[0]` (Issue #196):** `collection_service.py`'s ambiguous-source resolution and `zotero_api.py`'s `_parse_write_response` each materialized a full `list`/`.keys()` just to take the first element, in both cases already guarded to be non-empty by the surrounding code. Switched to `next(iter(...))`, which gets the same element without the intermediate full-list allocation. No behavior change.
197
+ - **SonarQube `python:S8572` cleanup — `logging.exception()` in 6 except blocks (Issue #195):** `resolvers/bdtd.py`, `resolvers/generic_scraper.py`, `snowball_graph.py`, `snowball_ingestion.py`, `snowball_worker.py`, and `infra/bdtd_api.py` each caught an exception and logged it via `logger.error(f"...: {e}")`, discarding the traceback. Switched to `logger.exception(...)`, which captures it automatically — no behavior change, purely a logging-fidelity improvement. Filed as part of a project-wide SonarQube audit of its then-9 open findings (#195/#196/#197).
198
+ - **Documentation-consistency protocol formalized (Issue #147):** New `docs/DOC_CONSISTENCY_PROTOCOL.md` — a repeatable 4-step process (ground-truth tree from source → per-leaf structural/reachability/prose checks → record findings → fix and re-verify) for the class of doc drift `pytest tests/docs`'s structural checks can't catch on their own (a documented example that's textually correct but unreachable, or prose describing old behavior). Referenced from `docs/PROCESS.md`'s Phase D and `CLAUDE.md`. See the Bug Fixes entry above for what the protocol's first real pass (once the automated check was actually working) found.
199
+ - **`JobQueueService`'s execution model settled; jobs scoped by library (Issue #150):** Documented in `docs/ARCHITECTURE.md` — `zotero-cli` does not grow a persistent daemon (`system jobs run --count N`/`--watch` remain the only ways to drain the queue); an external consumer like Corbenic-SLR's backend owns its own scheduling and observes status via the new `/jobs` API routes above. Separately, `Job`/`jobs.sqlite` had zero per-library scoping — two SLR projects sharing (or both omitting) `--config` would have silently pooled their jobs into one queue with no way to tell them apart. `JobQueueService` now takes a `library_id` (resolved from `config.library_id`/`user_id`, falling back to `"default"`), tags every job it enqueues, and filters every read by it; jobs enqueued before this migration (`library_id IS NULL`) are treated as legacy/unscoped and stay visible everywhere rather than becoming silently orphaned. `SqliteJobRepository` also now explicitly sets `PRAGMA journal_mode=WAL` — deliberate, not incidental, so a status-polling reader doesn't block behind an in-flight worker once this queue is driven by more than one local CLI invocation at a time.
200
+ - **`SnapshotService` naming collision resolved; deprecated `CollectionAuditor` shim deleted (Issue #148):** `core/services/snapshot_service.py` and `core/services/slr/snapshot.py` each defined an unrelated class both named `SnapshotService` — one writes JSON collection freezes (`slr report snapshot`), the other diffs two of them (`slr report shift`) — importing "the" `SnapshotService` depended silently on which module you happened to import from. Renamed to `SnapshotWriter` and `SnapshotDiffService` respectively. Separately, `slr report shift` was actually running through `CollectionAuditor`, a class explicitly docstringed `DEPRECATED: ... Maintained for backward compatibility during Phase B` that thinly wrapped `IntegrityService`/`SnapshotDiffService`/`CSVInboundService` — despite the deprecation notice it was the only live code path to the diff logic, not dead code. Deleted `CollectionAuditor` entirely (finishing "Phase B"); `report_cmd.py` now calls `GatewayFactory.get_snapshot_diff_service()`/`get_snapshot_writer_service()` directly instead of instantiating services inline, and the 4 test files that imported the wrapper now exercise `IntegrityService`, `CSVInboundService`, and `SnapshotDiffService` directly. `docs/ARCHITECTURE.md` now states explicitly that the lightweight JSON freeze format and `system backup`'s full ZAF archive format are intentionally separate mechanisms, not meant to converge.
201
+ - **`ZoteroGateway.get_trash_items` declared in the contract (Issue #140):** `item_cmd.py::_handle_list` called `gateway.get_trash_items()` on a gateway typed `Any` (a workaround from the #132 mypy-strictness pass, since the real `ZoteroGateway` ABC didn't declare the method the concrete `ZoteroAPIClient` already implemented). Added `get_trash_items` to the `ZoteroGateway` ABC, implemented it on the offline `SqliteZoteroGateway` too (previously missing entirely - `item list --trash --offline` would have crashed with `AttributeError`), and retyped `_handle_list`'s `gateway` parameter back to the honest `ZoteroGateway` type.
202
+
203
+ ### 🔥 Removed
204
+ - **Text-to-speech feature removed (Issue #149):** Deleted `item speech`, `core/services/speech_service.py`, `core/utils/speech_filter.py`, the `SpeechProvider` interface, `GatewayFactory.get_speech_service`, and the `tts_lang`/`tts_voice` config fields. Reading a paper aloud isn't part of Zotero library management or SLR tooling — this was flagged as a confirmed removal candidate in an external architecture review and never actioned. The `kokoro`/`soundfile` TTS engine was always a lazy runtime import, never a declared dependency, so no `pyproject.toml`/`uv.lock` change was needed.
205
+
206
+ ## [2.8.3] - 2026-07-26
207
+
208
+ Resolves both Known Issues disclosed in v2.8.2's release notes.
209
+
210
+ ### ⚠️ Breaking Changes
211
+ - **Minimum Python version raised to 3.11 (Issue #166):** Fixes 2 transitive `onnxruntime` path-traversal advisories (PVE-2026-88357/88358, pulled in via `markitdown` -> `magika`) disclosed as a Known Issue in v2.8.2. `onnxruntime` >= 1.24.2 (the first fixed release) ships no Python 3.10 wheels at all, so remediating this required raising the floor rather than a plain version bump. `requires-python`, `.python-version`, both CI workflows, the Docker/dev-container base images, and `sonar-project.properties` are all updated to 3.11; `onnxruntime` is now pinned directly (`>=1.24.2`, resolved to 1.28.0) instead of floating on whatever `magika` happens to pull in. `uv run safety check` now reports zero vulnerabilities.
212
+
213
+ ### 🐛 Bug Fixes
214
+ - **`item delete` unreachable (Issue #161):** Registers the `delete` subparser that `_handle_delete` was missing (same bug class as #146) — `item delete --key <KEY>` now actually works, matching the `--key` convention every other `item` verb uses. Also corrected `docs/commands/item.md`'s description, which wrongly called this a "move to trash": the Web API only exposes a hard, permanent `DELETE`, no soft-delete path exists.
215
+
216
+ ## [2.8.2] - 2026-07-25
217
+
218
+ ### ✨ Features & Improvements
219
+ - **Duplicate Detection Parity + Improvements (Issue #152):** `report duplicates` now matches by ISBN in addition to DOI/ArXiv/title, and can scan the whole library instead of specified collections (`--collections` is now optional). Items that don't exactly match anything else are additionally compared via a fuzzy fallback tier (title similarity + publication year within 1 year + at least one shared author last-name/first-initial), the same corroborating-signal approach Zotero Desktop uses — including a distinct `preprint-published-pair` label for a preprint matched against its later published version, a common SLR case Desktop's own algorithm doesn't call out explicitly.
220
+ - **`item merge` (Issue #155):** New generic, SLR-independent primitive for merging duplicate items — pick a master and one or more duplicates (found via `report duplicates`), and the command unions their tags/collections, moves notes/attachments onto the master, then permanently deletes the duplicates. Conflicting scalar fields (title, date, DOI, ISBN, URL, abstract) require an explicit per-field choice, no silent "first wins". Preview-only by default; `--execute` (plus confirmation, or `--force`) is required to actually write. This is necessarily a one-way, permanent operation — the Zotero Web API only exposes a hard delete, unlike Zotero Desktop's internal, reversible merge mechanism.
221
+ - **Bulk merge plans: `report duplicates --export-plan` + `item merge --from-plan` (Issue #156):** `report duplicates` can now export every found group as an editable plan file (`.csv` opens in a spreadsheet with blank role/reason columns; `.json` additionally embeds each occurrence's full SDB screening history for a richer review UI). Fill in which occurrence is the `MASTER`, which are `MERGE`/`KEEP`, and why, then run `item merge --from-plan <file> --execute` to commit every fully-resolved group in one pass. Completeness is all-or-nothing — a single group missing a decision blocks the *entire* plan, not just that group. The same `MergePlan`/`MergeDecision` dataclasses are meant to be built and consumed directly as Python objects (no file round-trip) by a future SLR-aware caller like Corbenic-SLR.
222
+ - **`slr dedupe`: SLR-specific duplicate reconciliation (Issue #157):** New command that reuses `report duplicates`'s detection scoped to the SLR source tree (every `raw_*` collection and its phase subfolders, or a given `--sources` set), and classifies each duplicate group by whether existing SDB screening decisions agree (`MATCHING`/`CONFLICTING`/`UNSCREENED`) — relocating that classification out of `report_cmd.py` into `SDBService.classify_decision_agreement`, now shared by both commands. `MATCHING`/`UNSCREENED` groups get an auto-filled merge decision and can be consolidated with `--execute`; `CONFLICTING` groups (independently-screened copies whose decisions genuinely disagree) are always left untouched — export the plan and resolve them via `item merge --from-plan`. Physical consolidation is delegated entirely to `MergeService`; a richer SDB reconciliation note (folded occurrences' own prior decisions and source collections, via an extended `SLROrchestrator.record_duplicate_resolution`) is written per merge, preserving audit history rather than collapsing it into a flat "merged" note. `slr report prisma --dedupe-source` can now also feed a read-only duplicate count into the PRISMA Identification-stage numbers.
223
+
224
+ ### 🛡️ Quality & Infrastructure
225
+ - **Distribution path for library consumers (Issue #154):** Documented the decision in `docs/ARCHITECTURE.md` — `zotero-cli` is not published to PyPI and won't be; a consumer like Corbenic-SLR that needs the `core/` services directly should depend on it via a git dependency pinned to a release tag (`pip install git+https://github.com/fchicout/zotero-cli@vX.Y.Z`, or `[tool.uv.sources]`), which needs zero new release infrastructure since every release already gets a git tag. Also fixed a pre-existing unclosed mermaid code fence in that same doc (the "Data Contracts" section, including this new one, was rendering as part of an unclosed code block).
226
+ - **`DuplicateFinder` as a Clean Library API (Issue #153):** `DuplicateFinder.find_duplicates`/`compare_collections` now return typed `DuplicateGroup`/`DuplicateOccurrence` dataclasses instead of ad hoc dicts, and no longer `print()` internally — non-fatal issues (e.g. a named collection that doesn't exist) are collected in `self.warnings` for the caller to surface instead. `MergeService`, the merge-plan dataclasses (#155/#156), and `SLRDedupeService` (#157) were all built to the same standard — typed dataclass returns, no stdout side effects, narrow constructor dependencies — completing the four services this issue covers.
227
+ - **README & Help-Text Accuracy Sweep:** Refreshed `README.md` to cover features that existed but weren't documented (BDTD import, `system check`, `system demo-sandbox`, Docker/devcontainer packaging, citation snowballing, RAG); fixed several dead command references left over from prior refactors (`slr validate` → `report audit`, `slr graph`/`slr shift`/`report status`/`report prisma` → their real `slr report <verb>` forms). Fixed two real bugs found in the process: `item list --help`'s description/example referenced SDB filtering flags (`--included`, `--criteria`, `--persona`) that were removed from that command in an earlier refactor (that filtering now lives in `slr list`), and `tag purge`'s runtime deprecation warning pointed to `collection purge --tags`, a command that was never implemented. Also fixed `scripts/generate_badges.py` silently dropping the CI status badges on every regeneration.
228
+
229
+ ### ⚠️ Known Issues
230
+ - **`item delete` unreachable (Issue #161):** `_handle_delete` is fully implemented and documented in `docs/commands/item.md`, but no `delete` subparser is registered in `item_cmd.py`'s `register_args()`, so the command cannot actually be invoked from the CLI today. Not fixed in this release; filed and tracked separately.
231
+ - **Transitive `onnxruntime` advisories (pre-existing, not a regression):** `safety check` reports 2 path-traversal advisories (PVE-2026-88357/88358) in `onnxruntime` 1.20.1, pulled in transitively via `sentence-transformers`. Not a direct dependency and not bumped in this release; tracked for a future dependency-upgrade pass.
232
+
233
+ ## [2.8.1] - 2026-07-19
234
+
235
+ ### ✨ Features & Improvements
236
+ - **Deeper Duplicate Analysis (Issue #107):** `report duplicates` now reports which collection each duplicate occurrence came from and whether their SDB screening decisions agree (`MATCHING`/`CONFLICTING`/`UNSCREENED`), plus a `--csv` export flag for audit records.
237
+ - **Diagnostic Health Checks (Issue #129):** New `system check` command probes Zotero, Semantic Scholar, Unpaywall, PubMed/NCBI, and the configured LLM/embedding providers, reporting CONNECTED/FAILED/NOT CONFIGURED for each.
238
+ - **Onboarding Demo Sandbox (Issue #130):** New `system demo-sandbox` command provisions a temporary collection with 6 mock papers (one pre-seeded with a mock SDB note) so new users can try screening/reporting/RAG commands without touching their real library; `--clean` removes it afterward.
239
+ - **Docker / Dev Container / Installer Scaffolding (Issue #131):** Added a root `Dockerfile` (lightweight runtime image packaging the same PyInstaller binary as the standalone releases), `.devcontainer/` for GitHub Codespaces/VS Code, and `install.sh`/`install.ps1` one-line installer scripts fetching the latest release binary.
240
+
241
+ ### 🛡️ Quality & Infrastructure
242
+ - **Strict Type Checking (Issue #132):** `mypy`'s `disallow_untyped_defs` is now `true` for `src/` (annotating the ~1050 pre-existing test-function signatures under `tests/` is a separate follow-up, kept lenient via an override for now); fixed all 231 real gaps this surfaced across 69 files.
243
+ - **`GatewayFactory` Decomposition:** Split the 941-line, 58-method `GatewayFactory` "God Object" into 5 focused sub-factories (`RepositoryFactory`, `MetadataClientFactory`, `ResolverFactory`, `AIProviderFactory`, `ServiceFactory`); `GatewayFactory` itself is now a thin, fully backward-compatible facade.
244
+ - **No More `sys.exit()` in Infra:** Removed all `sys.exit()` calls from gateway construction; invalid configuration (missing credentials, unparseable group URL, unresolved library) now raises a typed `ConfigurationError`, caught cleanly at the CLI boundary.
245
+ - **Interface Segregation:** Narrowed `TagService`, `AuditService`, `ExportService`, `CitationGraphService`, `DuplicateFinder`, `SnapshotService`, and `SyncService` off the full `ZoteroGateway` onto the specific narrow repositories (Item/Collection/Tag) they actually use.
246
+ - **AI-Ready SDLC Pivot:** Retired the stale Gemini-persona process-doc layer in favor of `CLAUDE.md` + GitHub Issues as the single source of truth; added mechanically-enforced quality gates (pre-commit hooks for ruff/mypy/bandit/pytest) and migrated tooling from pip to `uv`.
247
+ - **Documentation Consistency Sweep:** Removed 23 stale doc files describing commands renamed/removed in earlier refactors (`slr reset`/`migrate` → `slr sdb reset`/`upgrade`, `collection duplicates` → `report duplicates`, `find-pdf` → `item pdf`, and others), corrected the README command-reference table's key-verb listings, and fixed `tests/docs` to catch this class of drift going forward (header-row false positive in the Parameter Matrix parser, and new orphan-doc checks for both `docs/help_specs/` and `docs/commands/`).
248
+
249
+ ## [2.7.0] - 2026-05-07
250
+
251
+ ### ✨ Features & Improvements
252
+ - **Formal Project Documentation:** Added comprehensive `REQUIREMENTS.md`, `USE_CASES.md`, and `USER_STORIES.md` to the `docs/` directory, establishing a clear functional and user-centric baseline for the project.
253
+
254
+ ### 🛡️ Quality & Infrastructure (The Council Audit)
255
+ - **Root Directory Hygiene:** Performed a major cleanup of the project root, moving misplaced data artifacts (`.csv`, `.json`, `.txt`) to the `data/` directory and removing legacy coverage artifacts.
256
+ - **Documentation Consolidation:** Synchronized internal architectural notes with user-facing requirements to ensure cognitive clarity across the codebase.
257
+
258
+ ## [2.6.1] - 2026-04-26
259
+
260
+ ### ✨ Features & Improvements
261
+ - **SLR Status Dashboard:** Enhanced `slr status` with a new "Tree Total" column and global aggregate rows, providing a 360-degree view of the systematic review funnel across all sources.
262
+ - **Traceable Duplicate Auditing:** Implemented forensic duplicate logging; every resolution during system restore is now permanently recorded in SDB Audit Notes for 100% accountability.
263
+ - **Dependency Injection Refactor:** Major architectural cleanup of core services using constructor injection, centralized via the `GatewayFactory` for better testability and isolation.
264
+ - **Unified Purge Engine:** Consolidated all destructive operations (tag removal, PDF stripping) into a single, high-fidelity `PurgeService` to ensure consistent "Dry Run" and safety checks.
265
+
266
+ ### 🛡️ Quality & Infrastructure (Valerius Protocol)
267
+ - **Green State Certification:** Reached a landmark stability milestone with 100% test pass rate, 0 lint/type errors, and 0 warnings.
268
+ - **80% Coverage Gate:** Successfully cleared the global 80% code coverage threshold with new unit tests for high-impact SLR commands.
269
+ - **Zero-Leak E2E Sentinel:** Integrated a robust `ResourceTracker` in the E2E suite that guarantees remote Zotero resource cleanup even on test failures or crashes.
270
+ - **Python 3.14 Modernization:** Hardened the codebase for Python 3.14 compatibility and implemented targeted suppression of legacy library deprecation noise.
271
+
272
+ ## [2.6.0] - 2026-04-21
273
+
274
+ ### ✨ Features & Improvements
275
+ - **RAG Verification Engine (Spec v1.1):** Introduced automated integrity checks for semantic search results. Results can now be verified against mandatory academic identifiers (DOI/arXiv) and screening status.
276
+ - **Fidelity Integrity Guards:** Enforced high-fidelity JSON serialization in the RAG pipeline. Snippets are now preserved without truncation in `--json` output, ensuring 100% data reliability for citation verification.
277
+ - **Citation Key Traceability:** Enhanced the `ZoteroItem` model to automatically extract and verify Citation Keys from the Zotero 'extra' field.
278
+ - **Verification CLI:** Added the `--verify` flag to `rag query`, providing real-time feedback on the "verified" status of retrieved context.
279
+
280
+ ### 🛡️ Quality & Infrastructure
281
+ - **Restoration Gate:** Established a new safety protocol to verify the integrity of critical research database backups (`.bak_research`) during the test lifecycle.
282
+ - **Valerius Protocol Expansion:** Hardened the RAG test suite with exhaustive unit and fidelity tests.
283
+ - **Interface Consolidation:** Refactored `RAGService` to use a unified and more flexible ingestion strategy.
284
+
285
+ ## [2.5.0] - 2026-03-14
286
+
287
+ ### ✨ Features & Improvements
288
+ - **RAG Core (Issue #93):** Introduced Systematic Knowledge Retrieval. Allows building a local vector store from PDF full-texts and metadata for LLM context injection.
289
+ - **BibTeX Engine (Issue #94, #95):** Added direct collection export to `.bib` format. Includes phase-aware screening notes and criteria in the metadata.
290
+ - **Universal Item Transfer (Issue #91, #90):** Implemented high-fidelity cross-library move operations. Supports transferring items between personal and group libraries while preserving metadata and unfiled items.
291
+ - **Direct DOI Import (Issue #81):** Added `import doi <DOI>` command for instant bibliographic resolution and PDF discovery.
292
+ - **Full-Text Resilience:** Integrated `markitdown` for improved PDF-to-Markdown extraction, powering the RAG pipeline.
293
+
294
+ ## [2.4.1] - 2026-01-29
295
+
296
+ ### ✨ Features & Improvements
297
+ - **Safe Reset Engine (Issue #52):** Introduced \`slr reset\` command for phase-aware clearing of screening and extraction progress.
298
+ - **Granular Purging:** Enhanced \`PurgeService\` to support filtering by reviewer persona and screening phase, ensuring high-fidelity data management.
299
+ - **Tag Auto-Cleanup:** Automatic removal of phase-specific tags during reset operations.
300
+
301
+ ## [2.4.0] - 2026-01-29
302
+
303
+ ### ✨ Features & Improvements
304
+ - **SDB-Aware Listing (Issue #56):** Enhanced \`list items\` with support for screening database filters (\`--included\`, \`--excluded\`, \`--criteria\`, \`--persona\`, \`--phase\`).
305
+ - **Dynamic UX Rendering:** Active SDB filters trigger a specialized table schema showing Decisions, Criteria, and Persona metadata with color-coded status.
306
+ - **Auto-Move on Load (Issue #55):** The \`slr load\` command now supports automatic collection movement using \`--move-to-included\` and \`--move-to-excluded\` flags.
307
+ - **Improved CSV Matching:** Enhanced \`AuditService\` to handle case-insensitive CSV headers (\`status\`, \`decision\`) for better compatibility with external exports.
308
+
309
+ ## [2.3.0] - 2026-01-22
310
+
311
+ ### ✨ Features & Improvements
312
+ - **Semantic CLI Consolidation (Issue #38, #40):** Unified all systematic review commands under the `slr` namespace for improved ergonomics.
313
+ - **SLR Protocol Refinement (Issue #48):** Flattened the `slr` command tree (e.g., `slr load`, `slr validate`).
314
+ - **SDB v1.2 & Phase Isolation (Issue #49, #50):** Added support for `full_text` screening phase with evidence capture and phase-isolated notes.
315
+ - **Retroactive SDB Injection (Issue #32):** New `slr load` command with fuzzy matching for importing external decisions into the library.
316
+ - **Pre-flight Environment Checks (Issue #46):** Implemented "Boot Guard" pattern to enforce environment requirements at startup.
317
+
318
+ ### 🛡️ Quality & Infrastructure
319
+ - **MSI Installer Support (Issue #45):** Added official Windows MSI installer infrastructure using WiX v4.
320
+ - **Hard Coverage Gate (80%):** Successfully reached and enforced the 80% global test coverage threshold.
321
+ - **Recursive Deletion (Issue #37):** Refactored collection deletion to be truly recursive, preventing orphaned items.
322
+ - **ArXiv DOI Fallback (Issue #35):** Enhanced DOI extraction logic for ArXiv imports using regex on comments and references.
323
+ - **Test Hygiene (Issue #36):** Hardened E2E cleanup fixtures to ensure 100% resource reclamation.
324
+
325
+ ## [2.0.0] - 2026-01-17
326
+
327
+ ### 🚀 Major Architectural Shift (v2.0)
328
+ - **Service-Oriented Logic:** Completely decomposed the monolithic legacy `client.py` into specialized services (`ImportService`, `AttachmentService`, `CollectionService`).
329
+ - **Repository Pattern:** Solidified the persistence layer with a strict Repository Pattern, decoupling business logic from the Zotero API implementation.
330
+ - **Legacy Purge:** Successfully "liquidated" all remnants of the `paper2zotero` project name and associated garbage code.
331
+
332
+ ### ✨ Features & Improvements
333
+ - **Automated Quality Dashboard:** Implemented `scripts/generate_badges.py` providing real-time quality visualization (Coverage, Lint, Types) in `README.md`.
334
+ - **System Maintenance:** Added `system normalize` to convert external CSV formats (IEEE, Springer) into the CLI's Canonical Research Schema.
335
+ - **Advanced Operations:** Implemented `review prune` for enforcing mutual exclusivity between collections and `analyze shift` for tracking collection drift.
336
+ - **Robust Backup:** Introduced `.zaf` (LZMA-compressed ZIP) system-wide and collection-scoped backup/restore capabilities.
337
+
338
+ ### 🛡️ Quality & Testing
339
+ - **The Iron Gauntlet:** Established a comprehensive 221-test suite split into three deterministic categories:
340
+ - `unit`: Fast, isolated logic tests (80% qualitative coverage).
341
+ - `e2e`: Full-stack "Iron Gauntlet" tests against real Zotero API instances.
342
+ - `docs`: Automated consistency checks between CLI help and Markdown documentation.
343
+ - **Zero-Tolerance Quality:** Achieved 100% Green status on `ruff check` and `mypy` strict type checking.
344
+ - **Automated Test Runner:** Created `scripts/test_runner.sh` for unified, categorized test execution.
345
+
346
+ ### ⚠️ Breaking Changes
347
+ - Monolithic `PaperImporterClient` has been removed. Integration must now use `GatewayFactory` to obtain specific services.
348
+ - Version `2.0.0` is now the stable baseline for all future systematic review automation.
349
+
350
+ ## [2.0.0-rc1] - 2026-01-16 (Release Candidate)
351
+
352
+ ## [1.2.0] - 2026-01-15
353
+
354
+ ### Architecture
355
+ * **Command Pattern:** Refactored the entire CLI router into a registry-based Command Pattern. Logic is now modularized in `cli/commands/`.
356
+ * **Strategy Pattern:** Implemented the Strategy Pattern for paper importers, enabling easier extension for new bibliographic formats.
357
+ * **Dependency Injection:** Introduced `GatewayFactory` to centralize infrastructure creation and decouple commands from concrete implementations.
358
+ * **Centralized Configuration:** Moved all configuration and global state management to `core/config.py`.
359
+
360
+ ### Features
361
+ * **Decide Alias:** Added `d` alias for `decide` command to speed up manual screening.
362
+ * **Smart Move:** `manage move` and `decide` now support auto-inference of the source collection. If an item belongs to exactly one other collection, it is moved from there automatically. Fails safely on ambiguity.
363
+ * **Persistent State:** Added `--state <FILE.csv>` to `screen` command. Researchers can now resume sessions and track local screening decisions across restarts.
364
+ * **Extended Inspection:** Added `--full-notes` to `inspect` command to display untruncated note content (useful for auditing inclusion/exclusion rationale).
365
+
366
+ ### Fixes
367
+ * **Decide Command:** Fixed critical bug where `decide` failed to move items due to logic duplication. Now delegates to `CollectionService`.
368
+ * **Snapshot:** Fixed `ZeroDivisionError` in `report snapshot` when processing empty collections.
369
+ * **Inspect:** Resolved bug where `inspect --raw` failed due to missing `raw_data` attribute in `ZoteroItem`.
370
+
371
+ ### Quality
372
+ * **Mock Isolation:** Enhanced test suite to mock default configuration paths, preventing local developer configs from leaking into test environments.
373
+ * **Regressions:** Maintained 100% pass rate across 180 unit/integration tests.
374
+
375
+ ## [v1.1.0] - 2026-01-13 (Retrospective)
376
+ * **Configuration:** Added persistent configuration via `config.toml` (XDG Specification).
377
+ * **Precedence:** Established CLI Flags > Env > Config File hierarchy.
378
+
379
+ ## [v1.0.12] - 2026-01-13
380
+
381
+ ### Quality
382
+ * **Tests:** Fixed additional edge cases in `CollectionService` tests for move operations.
383
+
384
+ ## [v1.0.10] - 2026-01-13
385
+
386
+ ### Quality
387
+ * **Tests:** Fixed unit tests for `CollectionService` to correctly mock the new `get_item` optimization.
388
+
389
+ ## [v1.0.9] - 2026-01-13
390
+
391
+ ### Performance
392
+ * **Move Command:** Optimized `manage move` to use direct Item Key lookup (O(1)) instead of scanning the entire source collection (O(N)). Huge speedup for large libraries.
393
+
394
+ ## [v1.0.8] - 2026-01-13
395
+
396
+ ### Bug Fixes
397
+ * **Collection Movement:** Improved robustness of `screen` command item movement. Now correctly handles Collection Keys vs Names and avoids unnecessary API calls if collections haven't changed.
398
+
399
+ ## [v1.0.7] - 2026-01-13
400
+
401
+ ### Features
402
+ * **Bulk Screening:** New headless screening mode via CSV import.
403
+ * Command: `zotero-cli screen --file decisions.csv ...`
404
+ * Supports distributed team workflows.
405
+
406
+ ## [v1.0.6] - 2026-01-13
407
+
408
+ ### Bug Fixes
409
+ * **Inspect Command:** Fixed attribute mapping for `date` and `authors`.
410
+ * **Imports:** Fixed missing `Console` import in `info` command.
411
+
412
+ ## [v1.0.5] - 2026-01-13
413
+
414
+ ### Features
415
+ * **Global Flag:** Added `--user` flag to force the tool to use the Personal Library, bypassing any active `ZOTERO_TARGET_GROUP`.
416
+
417
+ ## [v1.0.4] - 2026-01-13
418
+
419
+ ### Features
420
+ * **Command:** Added `zotero-cli inspect` for viewing detailed item metadata and children.
421
+ * **UX:** `zotero-cli list items` now filters out nested items (attachments/notes) for a cleaner view.
422
+ * **UX:** `zotero-cli list items` supports case-insensitive partial collection names.
423
+
424
+ ## [v1.0.3] - 2026-01-13
425
+
426
+ ### Features
427
+ * **Info Command:** New `zotero-cli info` command to display diagnostic configuration.
428
+ * **Usability:** Improved collection name resolution with case-insensitive and partial match support.
429
+
430
+ ## [v1.0.2] - 2026-01-13
431
+
432
+ ### Quality
433
+ * **Test Coverage:** Increased to 82% (Green) by adding comprehensive failure scenarios for API wrappers.
434
+ * **Verification:** Verified full CLI command tree functionality.
435
+
436
+ ## [v1.0.1] - 2026-01-13
437
+
438
+ ### Architecture
439
+ * **SOLID Refactor:** Decoupled `ZoteroAPIClient` (Repository) from `ZoteroHttpClient` (Transport).
440
+ * **SRP Compliance:** Extracted HTTP logic, headers, and rate limiting to a dedicated transport layer.
441
+
442
+ ## [v1.0.0] - 2026-01-13
443
+
444
+ ### Major Changes
445
+ * **Command Tree Refactor:** Completely redesigned CLI structure for better usability.
446
+ * `import` (file, arxiv, manual)
447
+ * `screen` (TUI)
448
+ * `report` (prisma, snapshot)
449
+ * `manage` (tags, pdfs, duplicates, clean, move, migrate)
450
+ * `analyze` (audit, lookup, graph)
451
+ * `find` (arxiv)
452
+ * `list` (collections, groups, items)
453
+ * **Personal Library Support:** Added `ZOTERO_USER_ID` support. Tools now work with both Group and User libraries.
454
+ * **Universal Import:** Unified `import file` command auto-detects `.bib`, `.ris`, and `.csv`.
455
+
456
+ ### Features
457
+ * **List Groups:** New `list groups` command to discover User Group IDs.
458
+ * **List Items:** New `list items` command to inspect collections.
459
+ * **PRISMA Viz:** Integrated `mmdc` (Mermaid CLI) for high-quality flowchart generation.
460
+
461
+ ### Fixes
462
+ * **Concurrency:** Resolved `If-Unmodified-Since-Version` locking issues during batch migration.
463
+ * **TUI:** Fixed infinite loop in test mocks.
464
+
465
+ ### Breaking Changes
466
+ * Removed top-level commands: `bibtex`, `ris`, `springer-csv`, `ieee-csv`, `freeze`, `audit`, `duplicates`, `tag`, `attach-pdf`. These are now subcommands.