archiver-rag 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. archiver_rag-0.1.0/.claude/scheduled_tasks.lock +1 -0
  2. archiver_rag-0.1.0/.claude/settings.local.json +8 -0
  3. archiver_rag-0.1.0/.gitignore +11 -0
  4. archiver_rag-0.1.0/LICENSE +21 -0
  5. archiver_rag-0.1.0/PKG-INFO +247 -0
  6. archiver_rag-0.1.0/README.md +233 -0
  7. archiver_rag-0.1.0/archiver_rag/__init__.py +1 -0
  8. archiver_rag-0.1.0/archiver_rag/check_health.py +14 -0
  9. archiver_rag-0.1.0/archiver_rag/cli.py +243 -0
  10. archiver_rag-0.1.0/archiver_rag/const.py +3 -0
  11. archiver_rag-0.1.0/archiver_rag/core/__init__.py +7 -0
  12. archiver_rag-0.1.0/archiver_rag/core/chunker.py +12 -0
  13. archiver_rag-0.1.0/archiver_rag/core/db.py +15 -0
  14. archiver_rag-0.1.0/archiver_rag/core/embedder.py +26 -0
  15. archiver_rag-0.1.0/archiver_rag/core/ingest.py +139 -0
  16. archiver_rag-0.1.0/archiver_rag/core/search.py +34 -0
  17. archiver_rag-0.1.0/archiver_rag/graph/__init__.py +6 -0
  18. archiver_rag-0.1.0/archiver_rag/graph/clustering.py +150 -0
  19. archiver_rag-0.1.0/archiver_rag/graph/connections.py +54 -0
  20. archiver_rag-0.1.0/archiver_rag/graph/linker.py +104 -0
  21. archiver_rag-0.1.0/archiver_rag/graph/rerank.py +52 -0
  22. archiver_rag-0.1.0/archiver_rag/init_cmd.py +64 -0
  23. archiver_rag-0.1.0/archiver_rag/mcp/__init__.py +4 -0
  24. archiver_rag-0.1.0/archiver_rag/mcp/register.py +25 -0
  25. archiver_rag-0.1.0/archiver_rag/mcp/server.py +213 -0
  26. archiver_rag-0.1.0/archiver_rag/search_index.py +22 -0
  27. archiver_rag-0.1.0/archiver_rag/service.py +89 -0
  28. archiver_rag-0.1.0/archiver_rag/utils.py +27 -0
  29. archiver_rag-0.1.0/archiver_rag/vault/__init__.py +5 -0
  30. archiver_rag-0.1.0/archiver_rag/vault/health.py +105 -0
  31. archiver_rag-0.1.0/archiver_rag/vault/notes.py +79 -0
  32. archiver_rag-0.1.0/archiver_rag/vault/reorganize.py +104 -0
  33. archiver_rag-0.1.0/archiver_rag/watcher.py +114 -0
  34. archiver_rag-0.1.0/assets/archiver-rag-favicon.svg +7 -0
  35. archiver_rag-0.1.0/assets/archiver-rag-lockup-reversed.svg +16 -0
  36. archiver_rag-0.1.0/assets/archiver-rag-lockup.svg +16 -0
  37. archiver_rag-0.1.0/assets/archiver-rag-mark-mono.svg +9 -0
  38. archiver_rag-0.1.0/assets/archiver-rag-mark-reversed.svg +9 -0
  39. archiver_rag-0.1.0/assets/archiver-rag-mark.svg +9 -0
  40. archiver_rag-0.1.0/assets/logo.png +0 -0
  41. archiver_rag-0.1.0/assets/logo.svg +24 -0
  42. archiver_rag-0.1.0/docs/index.html +1067 -0
  43. archiver_rag-0.1.0/docs/logo.png +0 -0
  44. archiver_rag-0.1.0/pyproject.toml +24 -0
  45. archiver_rag-0.1.0/skill/claude-code/SKILL.md +207 -0
  46. archiver_rag-0.1.0/skill/codex/AGENTS.md +189 -0
  47. archiver_rag-0.1.0/skill/copilot/copilot-instructions.md +198 -0
  48. archiver_rag-0.1.0/skill/opencode/AGENTS.md +197 -0
@@ -0,0 +1 @@
1
+ {"sessionId":"64eb40c7-f89d-4919-ac0d-682d5bde0712","pid":19133,"procStart":"Wed Jun 10 21:13:44 2026","acquiredAt":1781283255570}
@@ -0,0 +1,8 @@
1
+ {
2
+ "permissions": {
3
+ "allow": [
4
+ "Bash(curl -s https://pypi.org/pypi/archiver-rag/json)",
5
+ "Bash(python3 -c \"import json,sys; d=json.load\\(sys.stdin\\); print\\('version:', d['info']['version']\\); print\\('name:', d['info']['name']\\)\")"
6
+ ]
7
+ }
8
+ }
@@ -0,0 +1,11 @@
1
+ persistence/
2
+ __pycache__/
3
+ *.pyc
4
+ .venv/
5
+ *.egg-info/
6
+ dist/
7
+ build/
8
+ .env
9
+ CLAUDE.md
10
+ .DS_Store
11
+ pyrightconfig.json
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 FernandoJRR
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,247 @@
1
+ Metadata-Version: 2.4
2
+ Name: archiver-rag
3
+ Version: 0.1.0
4
+ Summary: Semantic RAG for Obsidian vaults via MCP
5
+ License-File: LICENSE
6
+ Requires-Python: >=3.10
7
+ Requires-Dist: chromadb
8
+ Requires-Dist: mcp
9
+ Requires-Dist: rich
10
+ Requires-Dist: sentence-transformers
11
+ Requires-Dist: typer
12
+ Requires-Dist: watchdog
13
+ Description-Content-Type: text/markdown
14
+
15
+ <p align="center">
16
+ <img src="assets/archiver-rag-lockup.svg" alt="Archiver RAG" width="320" />
17
+ </p>
18
+
19
+ <p align="center">
20
+ <em>A finding aid for your knowledge graph</em>
21
+ </p>
22
+
23
+ <p align="center">
24
+ The agent-agnostic memory management system for your Obsidian vault
25
+ </p>
26
+
27
+ Archiver RAG turns your Obsidian vault into a live, queryable knowledge graph that any MCP-compatible AI agent can search, update, and reorganize — without ever leaving its native interface.
28
+
29
+ Connect it once. Every agent you use (Claude Code, Cursor, Gemini CLI, or your own) gets semantic search, automatic knowledge logging, wikilink-aware graph traversal, and vault health monitoring out of the box.
30
+
31
+ ---
32
+
33
+ ## How it works
34
+
35
+ ```
36
+ Your Obsidian vault (.md files)
37
+ ↓ file watcher + ingest pipeline
38
+ ChromaDB (persistent vector store)
39
+ ↓ MCP server
40
+ Any MCP-compatible agent
41
+ ```
42
+
43
+ Three layers make search smarter than plain embeddings:
44
+
45
+ 1. **Contextual prefix** — each chunk is embedded with its note's metadata (folder, tags, wikilinks), so vectors carry structural context
46
+ 2. **Rich metadata filtering** — ChromaDB stores folder, tags, incoming link count, and wikilinks for filtered retrieval
47
+ 3. **Graph reranking** — after vector search, results are re-scored by wikilink proximity to a context note and hub importance
48
+
49
+ The file watcher runs as a background service. Edit a note in Obsidian, save it, and it's indexed and auto-linked within seconds — no manual sync needed.
50
+
51
+ ---
52
+
53
+ ## Features
54
+
55
+ - **Semantic search** with graph reranking — finds notes by meaning, then boosts results connected via wikilinks
56
+ - **Auto-linking** — after every ingest, appends a `## Related` section with `[[wikilinks]]` to build the knowledge graph automatically
57
+ - **Knowledge logging** — create dated, categorized notes (`decision`, `lesson`, `gotcha`, `pattern`, …) from any agent
58
+ - **Vault health** — single call returns orphaned notes, broken links, missing frontmatter, tag stats, and recent activity
59
+ - **Wikilink-aware reorganization** — move files and every `[[link]]` across the vault is rewritten automatically
60
+ - **Smart clustering** — label-propagation algorithm groups notes by wikilink structure and suggests folder organization
61
+ - **Agent-agnostic** — exposes a standard MCP interface; works with any MCP-compatible client
62
+
63
+ ---
64
+
65
+ ## Requirements
66
+
67
+ - Python >= 3.10
68
+ - [pipx](https://pipx.pypa.io/) (recommended for installation)
69
+ - An Obsidian vault (local `.md` files)
70
+ - An MCP-compatible agent (Claude Code, Cursor, etc.)
71
+
72
+ ---
73
+
74
+ ## Installation
75
+
76
+ ```bash
77
+ pipx install archiver-rag
78
+ ```
79
+
80
+ > Use `pipx`, not `pip install` — pipx creates an isolated environment and exposes the CLI globally on `PATH`, which is required for MCP registration to find the correct executable.
81
+
82
+ For local development from a clone of this repo, use `pipx install --editable .` instead.
83
+
84
+ ---
85
+
86
+ ## Setup
87
+
88
+ Run the one-time setup wizard:
89
+
90
+ ```bash
91
+ archiver-rag init
92
+ ```
93
+
94
+ This will:
95
+ 1. Ask for your vault path
96
+ 2. Index your vault into ChromaDB
97
+ 3. Register the MCP server in `~/.claude.json` (or prompt you to do it manually for other clients)
98
+ 4. Install the background watcher as a launchd agent (Mac) or systemd service (Linux)
99
+
100
+ ---
101
+
102
+ ## MCP registration (manual)
103
+
104
+ If you prefer to register manually, add this to your MCP client config:
105
+
106
+ ```json
107
+ {
108
+ "mcpServers": {
109
+ "archiver-rag": {
110
+ "command": "/path/to/archiver-rag",
111
+ "args": ["serve"]
112
+ }
113
+ }
114
+ }
115
+ ```
116
+
117
+ Find the executable path with `which archiver-rag`.
118
+
119
+ For Claude Code specifically, use:
120
+
121
+ ```bash
122
+ claude mcp add --scope user archiver-rag $(which archiver-rag) serve
123
+ ```
124
+
125
+ ---
126
+
127
+ ## Agent instructions (skills)
128
+
129
+ Registering the MCP server gives an agent *access* to the tools — but agents tend to fall back on their own internal memory instead of reaching for the vault. The instruction files in [`skill/`](skill/) fix that: they enforce a **vault-first rule** so the agent searches and stores knowledge in your vault before anything else.
130
+
131
+ **What the skill enforces:**
132
+
133
+ - **Before answering or reading source files** — call `search_vault` first; only fall back to internal memory if the vault returns nothing relevant
134
+ - **When something important is missing from the vault** — proactively `log_note` it. If a fact, decision, or piece of context matters to the overall picture and a `search_vault` came back empty, record it so the knowledge graph grows instead of letting that context die in a single session
135
+ - **After solving a non-trivial problem** — call `log_note` to capture the decision/lesson/gotcha back into the vault
136
+ - **The vault is the authoritative memory system** — internal agent memory is a fallback only
137
+
138
+ A version is provided for each agent, since each loads instructions differently:
139
+
140
+ | Agent | File | Install to |
141
+ |---|---|---|
142
+ | Claude Code | [`skill/claude-code/SKILL.md`](skill/claude-code/SKILL.md) | `~/.claude/skills/archiver-rag/SKILL.md` (on-demand skill) |
143
+ | OpenCode | [`skill/opencode/AGENTS.md`](skill/opencode/AGENTS.md) | project root `AGENTS.md` or `~/.config/opencode/AGENTS.md` |
144
+ | Codex CLI | [`skill/codex/AGENTS.md`](skill/codex/AGENTS.md) | project root `AGENTS.md` or `~/.codex/AGENTS.md` |
145
+ | GitHub Copilot | [`skill/copilot/copilot-instructions.md`](skill/copilot/copilot-instructions.md) | `.github/copilot-instructions.md` |
146
+
147
+ Each file is self-contained — it includes the MCP registration snippet for that agent plus the full vault-first rules and tool reference. For Claude Code the file is an on-demand skill; for the others it's an always-on instruction file (loaded into every session), which makes the vault-first behavior unconditional.
148
+
149
+ ---
150
+
151
+ ## CLI reference
152
+
153
+ ```bash
154
+ archiver-rag init # one-time setup wizard
155
+ archiver-rag start # start the background watcher service
156
+ archiver-rag stop # stop the service
157
+ archiver-rag restart # restart the service
158
+ archiver-rag status # check if service is running
159
+ archiver-rag index # force re-index the entire vault
160
+ archiver-rag search "query" # test semantic search from the terminal
161
+ archiver-rag health # chunk count and index peek
162
+ archiver-rag logs # tail the service log
163
+
164
+ # Knowledge logging
165
+ archiver-rag log "Title" --type decision --tag arch --related NoteA
166
+
167
+ # Clustering
168
+ archiver-rag cluster # suggest folder groupings
169
+ archiver-rag cluster --apply # move files automatically
170
+ archiver-rag place <note> # suggest folder for a single note
171
+ archiver-rag place <note> --apply # move it immediately
172
+
173
+ # Config
174
+ archiver-rag config --auto-cluster # enable auto-clustering in the watcher
175
+ archiver-rag config --cluster-threshold 5 # notes before a full re-cluster
176
+
177
+ archiver-rag uninstall # remove all data, service, and MCP registration
178
+ ```
179
+
180
+ ---
181
+
182
+ ## MCP tools (for agents)
183
+
184
+ Once registered, agents have access to 7 tools:
185
+
186
+ | Tool | What it does |
187
+ |---|---|
188
+ | `search_vault` | Semantic search with graph reranking. Accepts a `context_note` to boost wikilink neighbors. |
189
+ | `vault_status` | Vault structure, health diagnostics, tag stats, and recent activity in one call. |
190
+ | `get_connections` | BFS wikilink traversal — outgoing and incoming links up to depth 3. |
191
+ | `move_notes` | Move files and auto-rewrite all `[[wikilinks]]` across the vault. |
192
+ | `log_note` | Create a dated knowledge note; watcher indexes and auto-links it immediately. |
193
+ | `cluster_note` | Suggest a folder for one note based on where its wikilink neighbors live. |
194
+ | `cluster_vault` | Label-propagation clustering of the entire vault with folder suggestions. |
195
+
196
+ ---
197
+
198
+ ## Configuration
199
+
200
+ All runtime config lives at `~/.archiver-rag/config.json`:
201
+
202
+ ```json
203
+ {
204
+ "vault_path": "/path/to/your/vault",
205
+ "install_path": "/Users/you/.archiver-rag",
206
+ "chroma_path": "/Users/you/.archiver-rag/chroma_db",
207
+ "auto_cluster": false,
208
+ "cluster_threshold": 5
209
+ }
210
+ ```
211
+
212
+ `auto_cluster` — automatically suggest and apply folder placement for new notes via the watcher.
213
+ `cluster_threshold` — number of new notes created before triggering a full `cluster_vault` run.
214
+
215
+ ---
216
+
217
+ ## The knowledge graph model
218
+
219
+ The vault is treated as a **knowledge graph**, not a file hierarchy. Notes are nodes; wikilinks are edges. Relationships range from tight (direct links) to loose (semantic proximity surfaced by search).
220
+
221
+ Note types are expressed through frontmatter, not folder structure:
222
+
223
+ ```yaml
224
+ ---
225
+ type: decision
226
+ tags: [architecture, async]
227
+ related: [[AsyncLocalStorage]], [[PrismaExtensions]]
228
+ date: 2026-04-27
229
+ ---
230
+ ```
231
+
232
+ The `## Related` section at the bottom of each note is managed automatically by the auto-linker after every ingest. Don't edit it manually — it will be overwritten.
233
+
234
+ ---
235
+
236
+ ## Roadmap
237
+
238
+ Features on the way:
239
+
240
+ - **RAG-Anything integration** — extend ingestion beyond Markdown to handle PDFs, Office documents, images, and other file types, so the vault can become a true multi-format knowledge base rather than `.md`-only.
241
+ - **Archiver subagents** — dedicated subagents that take over vault management (search, logging, reorganization, clustering) on the main agent's behalf, so the primary agent can delegate knowledge work instead of context-switching into it.
242
+
243
+ ---
244
+
245
+ ## License
246
+
247
+ MIT
@@ -0,0 +1,233 @@
1
+ <p align="center">
2
+ <img src="assets/archiver-rag-lockup.svg" alt="Archiver RAG" width="320" />
3
+ </p>
4
+
5
+ <p align="center">
6
+ <em>A finding aid for your knowledge graph</em>
7
+ </p>
8
+
9
+ <p align="center">
10
+ The agent-agnostic memory management system for your Obsidian vault
11
+ </p>
12
+
13
+ Archiver RAG turns your Obsidian vault into a live, queryable knowledge graph that any MCP-compatible AI agent can search, update, and reorganize — without ever leaving its native interface.
14
+
15
+ Connect it once. Every agent you use (Claude Code, Cursor, Gemini CLI, or your own) gets semantic search, automatic knowledge logging, wikilink-aware graph traversal, and vault health monitoring out of the box.
16
+
17
+ ---
18
+
19
+ ## How it works
20
+
21
+ ```
22
+ Your Obsidian vault (.md files)
23
+ ↓ file watcher + ingest pipeline
24
+ ChromaDB (persistent vector store)
25
+ ↓ MCP server
26
+ Any MCP-compatible agent
27
+ ```
28
+
29
+ Three layers make search smarter than plain embeddings:
30
+
31
+ 1. **Contextual prefix** — each chunk is embedded with its note's metadata (folder, tags, wikilinks), so vectors carry structural context
32
+ 2. **Rich metadata filtering** — ChromaDB stores folder, tags, incoming link count, and wikilinks for filtered retrieval
33
+ 3. **Graph reranking** — after vector search, results are re-scored by wikilink proximity to a context note and hub importance
34
+
35
+ The file watcher runs as a background service. Edit a note in Obsidian, save it, and it's indexed and auto-linked within seconds — no manual sync needed.
36
+
37
+ ---
38
+
39
+ ## Features
40
+
41
+ - **Semantic search** with graph reranking — finds notes by meaning, then boosts results connected via wikilinks
42
+ - **Auto-linking** — after every ingest, appends a `## Related` section with `[[wikilinks]]` to build the knowledge graph automatically
43
+ - **Knowledge logging** — create dated, categorized notes (`decision`, `lesson`, `gotcha`, `pattern`, …) from any agent
44
+ - **Vault health** — single call returns orphaned notes, broken links, missing frontmatter, tag stats, and recent activity
45
+ - **Wikilink-aware reorganization** — move files and every `[[link]]` across the vault is rewritten automatically
46
+ - **Smart clustering** — label-propagation algorithm groups notes by wikilink structure and suggests folder organization
47
+ - **Agent-agnostic** — exposes a standard MCP interface; works with any MCP-compatible client
48
+
49
+ ---
50
+
51
+ ## Requirements
52
+
53
+ - Python >= 3.10
54
+ - [pipx](https://pipx.pypa.io/) (recommended for installation)
55
+ - An Obsidian vault (local `.md` files)
56
+ - An MCP-compatible agent (Claude Code, Cursor, etc.)
57
+
58
+ ---
59
+
60
+ ## Installation
61
+
62
+ ```bash
63
+ pipx install archiver-rag
64
+ ```
65
+
66
+ > Use `pipx`, not `pip install` — pipx creates an isolated environment and exposes the CLI globally on `PATH`, which is required for MCP registration to find the correct executable.
67
+
68
+ For local development from a clone of this repo, use `pipx install --editable .` instead.
69
+
70
+ ---
71
+
72
+ ## Setup
73
+
74
+ Run the one-time setup wizard:
75
+
76
+ ```bash
77
+ archiver-rag init
78
+ ```
79
+
80
+ This will:
81
+ 1. Ask for your vault path
82
+ 2. Index your vault into ChromaDB
83
+ 3. Register the MCP server in `~/.claude.json` (or prompt you to do it manually for other clients)
84
+ 4. Install the background watcher as a launchd agent (Mac) or systemd service (Linux)
85
+
86
+ ---
87
+
88
+ ## MCP registration (manual)
89
+
90
+ If you prefer to register manually, add this to your MCP client config:
91
+
92
+ ```json
93
+ {
94
+ "mcpServers": {
95
+ "archiver-rag": {
96
+ "command": "/path/to/archiver-rag",
97
+ "args": ["serve"]
98
+ }
99
+ }
100
+ }
101
+ ```
102
+
103
+ Find the executable path with `which archiver-rag`.
104
+
105
+ For Claude Code specifically, use:
106
+
107
+ ```bash
108
+ claude mcp add --scope user archiver-rag $(which archiver-rag) serve
109
+ ```
110
+
111
+ ---
112
+
113
+ ## Agent instructions (skills)
114
+
115
+ Registering the MCP server gives an agent *access* to the tools — but agents tend to fall back on their own internal memory instead of reaching for the vault. The instruction files in [`skill/`](skill/) fix that: they enforce a **vault-first rule** so the agent searches and stores knowledge in your vault before anything else.
116
+
117
+ **What the skill enforces:**
118
+
119
+ - **Before answering or reading source files** — call `search_vault` first; only fall back to internal memory if the vault returns nothing relevant
120
+ - **When something important is missing from the vault** — proactively `log_note` it. If a fact, decision, or piece of context matters to the overall picture and a `search_vault` came back empty, record it so the knowledge graph grows instead of letting that context die in a single session
121
+ - **After solving a non-trivial problem** — call `log_note` to capture the decision/lesson/gotcha back into the vault
122
+ - **The vault is the authoritative memory system** — internal agent memory is a fallback only
123
+
124
+ A version is provided for each agent, since each loads instructions differently:
125
+
126
+ | Agent | File | Install to |
127
+ |---|---|---|
128
+ | Claude Code | [`skill/claude-code/SKILL.md`](skill/claude-code/SKILL.md) | `~/.claude/skills/archiver-rag/SKILL.md` (on-demand skill) |
129
+ | OpenCode | [`skill/opencode/AGENTS.md`](skill/opencode/AGENTS.md) | project root `AGENTS.md` or `~/.config/opencode/AGENTS.md` |
130
+ | Codex CLI | [`skill/codex/AGENTS.md`](skill/codex/AGENTS.md) | project root `AGENTS.md` or `~/.codex/AGENTS.md` |
131
+ | GitHub Copilot | [`skill/copilot/copilot-instructions.md`](skill/copilot/copilot-instructions.md) | `.github/copilot-instructions.md` |
132
+
133
+ Each file is self-contained — it includes the MCP registration snippet for that agent plus the full vault-first rules and tool reference. For Claude Code the file is an on-demand skill; for the others it's an always-on instruction file (loaded into every session), which makes the vault-first behavior unconditional.
134
+
135
+ ---
136
+
137
+ ## CLI reference
138
+
139
+ ```bash
140
+ archiver-rag init # one-time setup wizard
141
+ archiver-rag start # start the background watcher service
142
+ archiver-rag stop # stop the service
143
+ archiver-rag restart # restart the service
144
+ archiver-rag status # check if service is running
145
+ archiver-rag index # force re-index the entire vault
146
+ archiver-rag search "query" # test semantic search from the terminal
147
+ archiver-rag health # chunk count and index peek
148
+ archiver-rag logs # tail the service log
149
+
150
+ # Knowledge logging
151
+ archiver-rag log "Title" --type decision --tag arch --related NoteA
152
+
153
+ # Clustering
154
+ archiver-rag cluster # suggest folder groupings
155
+ archiver-rag cluster --apply # move files automatically
156
+ archiver-rag place <note> # suggest folder for a single note
157
+ archiver-rag place <note> --apply # move it immediately
158
+
159
+ # Config
160
+ archiver-rag config --auto-cluster # enable auto-clustering in the watcher
161
+ archiver-rag config --cluster-threshold 5 # notes before a full re-cluster
162
+
163
+ archiver-rag uninstall # remove all data, service, and MCP registration
164
+ ```
165
+
166
+ ---
167
+
168
+ ## MCP tools (for agents)
169
+
170
+ Once registered, agents have access to 7 tools:
171
+
172
+ | Tool | What it does |
173
+ |---|---|
174
+ | `search_vault` | Semantic search with graph reranking. Accepts a `context_note` to boost wikilink neighbors. |
175
+ | `vault_status` | Vault structure, health diagnostics, tag stats, and recent activity in one call. |
176
+ | `get_connections` | BFS wikilink traversal — outgoing and incoming links up to depth 3. |
177
+ | `move_notes` | Move files and auto-rewrite all `[[wikilinks]]` across the vault. |
178
+ | `log_note` | Create a dated knowledge note; watcher indexes and auto-links it immediately. |
179
+ | `cluster_note` | Suggest a folder for one note based on where its wikilink neighbors live. |
180
+ | `cluster_vault` | Label-propagation clustering of the entire vault with folder suggestions. |
181
+
182
+ ---
183
+
184
+ ## Configuration
185
+
186
+ All runtime config lives at `~/.archiver-rag/config.json`:
187
+
188
+ ```json
189
+ {
190
+ "vault_path": "/path/to/your/vault",
191
+ "install_path": "/Users/you/.archiver-rag",
192
+ "chroma_path": "/Users/you/.archiver-rag/chroma_db",
193
+ "auto_cluster": false,
194
+ "cluster_threshold": 5
195
+ }
196
+ ```
197
+
198
+ `auto_cluster` — automatically suggest and apply folder placement for new notes via the watcher.
199
+ `cluster_threshold` — number of new notes created before triggering a full `cluster_vault` run.
200
+
201
+ ---
202
+
203
+ ## The knowledge graph model
204
+
205
+ The vault is treated as a **knowledge graph**, not a file hierarchy. Notes are nodes; wikilinks are edges. Relationships range from tight (direct links) to loose (semantic proximity surfaced by search).
206
+
207
+ Note types are expressed through frontmatter, not folder structure:
208
+
209
+ ```yaml
210
+ ---
211
+ type: decision
212
+ tags: [architecture, async]
213
+ related: [[AsyncLocalStorage]], [[PrismaExtensions]]
214
+ date: 2026-04-27
215
+ ---
216
+ ```
217
+
218
+ The `## Related` section at the bottom of each note is managed automatically by the auto-linker after every ingest. Don't edit it manually — it will be overwritten.
219
+
220
+ ---
221
+
222
+ ## Roadmap
223
+
224
+ Features on the way:
225
+
226
+ - **RAG-Anything integration** — extend ingestion beyond Markdown to handle PDFs, Office documents, images, and other file types, so the vault can become a true multi-format knowledge base rather than `.md`-only.
227
+ - **Archiver subagents** — dedicated subagents that take over vault management (search, logging, reorganization, clustering) on the main agent's behalf, so the primary agent can delegate knowledge work instead of context-switching into it.
228
+
229
+ ---
230
+
231
+ ## License
232
+
233
+ MIT
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"
@@ -0,0 +1,14 @@
1
+ from archiver_rag.core.db import collection
2
+
3
+ count = collection.count()
4
+ print(f"Total chunks in index: {count}")
5
+
6
+ if count > 0:
7
+ # Peek at first few chunks
8
+ results = collection.peek(limit=3)
9
+ for i, (doc, meta) in enumerate(zip(results["documents"], results["metadatas"])):
10
+ print(f"\n--- Chunk {i+1} ---")
11
+ print(f"Source: {meta['source']}")
12
+ print(f"Preview: {doc[:200]}")
13
+ else:
14
+ print("Index is empty — ingest hasn't run or failed silently")