archiver-rag 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. archiver_rag-1.0.0/.claude/settings.local.json +7 -0
  2. archiver_rag-1.0.0/.gitignore +11 -0
  3. archiver_rag-1.0.0/PKG-INFO +239 -0
  4. archiver_rag-1.0.0/README.md +226 -0
  5. archiver_rag-1.0.0/archiver_rag/__init__.py +1 -0
  6. archiver_rag-1.0.0/archiver_rag/check_health.py +14 -0
  7. archiver_rag-1.0.0/archiver_rag/cli.py +243 -0
  8. archiver_rag-1.0.0/archiver_rag/const.py +3 -0
  9. archiver_rag-1.0.0/archiver_rag/core/__init__.py +7 -0
  10. archiver_rag-1.0.0/archiver_rag/core/chunker.py +12 -0
  11. archiver_rag-1.0.0/archiver_rag/core/db.py +15 -0
  12. archiver_rag-1.0.0/archiver_rag/core/embedder.py +26 -0
  13. archiver_rag-1.0.0/archiver_rag/core/ingest.py +139 -0
  14. archiver_rag-1.0.0/archiver_rag/core/search.py +34 -0
  15. archiver_rag-1.0.0/archiver_rag/graph/__init__.py +6 -0
  16. archiver_rag-1.0.0/archiver_rag/graph/clustering.py +150 -0
  17. archiver_rag-1.0.0/archiver_rag/graph/connections.py +54 -0
  18. archiver_rag-1.0.0/archiver_rag/graph/linker.py +104 -0
  19. archiver_rag-1.0.0/archiver_rag/graph/rerank.py +52 -0
  20. archiver_rag-1.0.0/archiver_rag/init_cmd.py +64 -0
  21. archiver_rag-1.0.0/archiver_rag/mcp/__init__.py +4 -0
  22. archiver_rag-1.0.0/archiver_rag/mcp/register.py +25 -0
  23. archiver_rag-1.0.0/archiver_rag/mcp/server.py +213 -0
  24. archiver_rag-1.0.0/archiver_rag/search_index.py +22 -0
  25. archiver_rag-1.0.0/archiver_rag/service.py +89 -0
  26. archiver_rag-1.0.0/archiver_rag/utils.py +27 -0
  27. archiver_rag-1.0.0/archiver_rag/vault/__init__.py +5 -0
  28. archiver_rag-1.0.0/archiver_rag/vault/health.py +105 -0
  29. archiver_rag-1.0.0/archiver_rag/vault/notes.py +79 -0
  30. archiver_rag-1.0.0/archiver_rag/vault/reorganize.py +104 -0
  31. archiver_rag-1.0.0/archiver_rag/watcher.py +114 -0
  32. archiver_rag-1.0.0/assets/logo.png +0 -0
  33. archiver_rag-1.0.0/pyproject.toml +24 -0
  34. archiver_rag-1.0.0/skill/claude-code/SKILL.md +212 -0
  35. archiver_rag-1.0.0/skill/codex/AGENTS.md +157 -0
  36. archiver_rag-1.0.0/skill/copilot/copilot-instructions.md +166 -0
  37. archiver_rag-1.0.0/skill/opencode/AGENTS.md +165 -0
@@ -0,0 +1,7 @@
1
+ {
2
+ "permissions": {
3
+ "allow": [
4
+ "Bash(ls /Users/fernanrod/Programacion/Proyectos/archiver_rag/skill/ 2>/dev/null || echo \"no skill dir\" *)"
5
+ ]
6
+ }
7
+ }
@@ -0,0 +1,11 @@
1
+ persistence/
2
+ __pycache__/
3
+ *.pyc
4
+ .venv/
5
+ *.egg-info/
6
+ dist/
7
+ build/
8
+ .env
9
+ CLAUDE.md
10
+ .DS_Store
11
+ pyrightconfig.json
@@ -0,0 +1,239 @@
1
+ Metadata-Version: 2.4
2
+ Name: archiver-rag
3
+ Version: 1.0.0
4
+ Summary: Semantic RAG for Obsidian vaults via MCP
5
+ Requires-Python: >=3.10
6
+ Requires-Dist: chromadb
7
+ Requires-Dist: mcp
8
+ Requires-Dist: rich
9
+ Requires-Dist: sentence-transformers
10
+ Requires-Dist: typer
11
+ Requires-Dist: watchdog
12
+ Description-Content-Type: text/markdown
13
+
14
+ <p align="center">
15
+ <img src="assets/logo.png" alt="Archiver RAG" width="180" />
16
+ </p>
17
+
18
+ # Archiver RAG
19
+ ### The agent-agnostic memory management system through Obsidian-esque techniques
20
+
21
+ Archiver RAG turns your Obsidian vault into a live, queryable knowledge graph that any MCP-compatible AI agent can search, update, and reorganize — without ever leaving its native interface.
22
+
23
+ Connect it once. Every agent you use (Claude Code, Cursor, Gemini CLI, or your own) gets semantic search, automatic knowledge logging, wikilink-aware graph traversal, and vault health monitoring out of the box.
24
+
25
+ ---
26
+
27
+ ## How it works
28
+
29
+ ```
30
+ Your Obsidian vault (.md files)
31
+ ↓ file watcher + ingest pipeline
32
+ ChromaDB (persistent vector store)
33
+ ↓ MCP server
34
+ Any MCP-compatible agent
35
+ ```
36
+
37
+ Three layers make search smarter than plain embeddings:
38
+
39
+ 1. **Contextual prefix** — each chunk is embedded with its note's metadata (folder, tags, wikilinks), so vectors carry structural context
40
+ 2. **Rich metadata filtering** — ChromaDB stores folder, tags, incoming link count, and wikilinks for filtered retrieval
41
+ 3. **Graph reranking** — after vector search, results are re-scored by wikilink proximity to a context note and hub importance
42
+
43
+ The file watcher runs as a background service. Edit a note in Obsidian, save it, and it's indexed and auto-linked within seconds — no manual sync needed.
44
+
45
+ ---
46
+
47
+ ## Features
48
+
49
+ - **Semantic search** with graph reranking — finds notes by meaning, then boosts results connected via wikilinks
50
+ - **Auto-linking** — after every ingest, appends a `## Related` section with `[[wikilinks]]` to build the knowledge graph automatically
51
+ - **Knowledge logging** — create dated, categorized notes (`decision`, `lesson`, `gotcha`, `pattern`, …) from any agent
52
+ - **Vault health** — single call returns orphaned notes, broken links, missing frontmatter, tag stats, and recent activity
53
+ - **Wikilink-aware reorganization** — move files and every `[[link]]` across the vault is rewritten automatically
54
+ - **Smart clustering** — label-propagation algorithm groups notes by wikilink structure and suggests folder organization
55
+ - **Agent-agnostic** — exposes a standard MCP interface; works with any MCP-compatible client
56
+
57
+ ---
58
+
59
+ ## Requirements
60
+
61
+ - Python >= 3.10
62
+ - [pipx](https://pipx.pypa.io/) (recommended for installation)
63
+ - An Obsidian vault (local `.md` files)
64
+ - An MCP-compatible agent (Claude Code, Cursor, etc.)
65
+
66
+ ---
67
+
68
+ ## Installation
69
+
70
+ ```bash
71
+ pipx install --editable .
72
+ ```
73
+
74
+ > Use `pipx`, not `pip install -e .` — pipx creates an isolated environment and exposes the CLI globally on `PATH`, which is required for MCP registration to find the correct executable.
75
+
76
+ ---
77
+
78
+ ## Setup
79
+
80
+ Run the one-time setup wizard:
81
+
82
+ ```bash
83
+ archiver-rag init
84
+ ```
85
+
86
+ This will:
87
+ 1. Ask for your vault path
88
+ 2. Index your vault into ChromaDB
89
+ 3. Register the MCP server in `~/.claude.json` (or prompt you to do it manually for other clients)
90
+ 4. Install the background watcher as a launchd agent (Mac) or systemd service (Linux)
91
+
92
+ ---
93
+
94
+ ## MCP registration (manual)
95
+
96
+ If you prefer to register manually, add this to your MCP client config:
97
+
98
+ ```json
99
+ {
100
+ "mcpServers": {
101
+ "archiver-rag": {
102
+ "command": "/path/to/archiver-rag",
103
+ "args": ["serve"]
104
+ }
105
+ }
106
+ }
107
+ ```
108
+
109
+ Find the executable path with `which archiver-rag`.
110
+
111
+ For Claude Code specifically, use:
112
+
113
+ ```bash
114
+ claude mcp add --scope user archiver-rag $(which archiver-rag) serve
115
+ ```
116
+
117
+ ---
118
+
119
+ ## Agent instructions (skills)
120
+
121
+ Registering the MCP server gives an agent *access* to the tools — but agents tend to fall back on their own internal memory instead of reaching for the vault. The instruction files in [`skill/`](skill/) fix that: they enforce a **vault-first rule** so the agent searches and stores knowledge in your vault before anything else.
122
+
123
+ **What the skill enforces:**
124
+
125
+ - **Before answering or reading source files** — call `search_vault` first; only fall back to internal memory if the vault returns nothing relevant
126
+ - **When something important is missing from the vault** — proactively `log_note` it. If a fact, decision, or piece of context matters to the overall picture and a `search_vault` came back empty, record it so the knowledge graph grows instead of letting that context die in a single session
127
+ - **After solving a non-trivial problem** — call `log_note` to capture the decision/lesson/gotcha back into the vault
128
+ - **The vault is the authoritative memory system** — internal agent memory is a fallback only
129
+
130
+ A version is provided for each agent, since each loads instructions differently:
131
+
132
+ | Agent | File | Install to |
133
+ |---|---|---|
134
+ | Claude Code | [`skill/claude-code/SKILL.md`](skill/claude-code/SKILL.md) | `~/.claude/skills/archiver-rag/SKILL.md` (on-demand skill) |
135
+ | OpenCode | [`skill/opencode/AGENTS.md`](skill/opencode/AGENTS.md) | project root `AGENTS.md` or `~/.config/opencode/AGENTS.md` |
136
+ | Codex CLI | [`skill/codex/AGENTS.md`](skill/codex/AGENTS.md) | project root `AGENTS.md` or `~/.codex/AGENTS.md` |
137
+ | GitHub Copilot | [`skill/copilot/copilot-instructions.md`](skill/copilot/copilot-instructions.md) | `.github/copilot-instructions.md` |
138
+
139
+ Each file is self-contained — it includes the MCP registration snippet for that agent plus the full vault-first rules and tool reference. For Claude Code the file is an on-demand skill; for the others it's an always-on instruction file (loaded into every session), which makes the vault-first behavior unconditional.
140
+
141
+ ---
142
+
143
+ ## CLI reference
144
+
145
+ ```bash
146
+ archiver-rag init # one-time setup wizard
147
+ archiver-rag start # start the background watcher service
148
+ archiver-rag stop # stop the service
149
+ archiver-rag restart # restart the service
150
+ archiver-rag status # check if service is running
151
+ archiver-rag index # force re-index the entire vault
152
+ archiver-rag search "query" # test semantic search from the terminal
153
+ archiver-rag health # chunk count and index peek
154
+ archiver-rag logs # tail the service log
155
+
156
+ # Knowledge logging
157
+ archiver-rag log "Title" --type decision --tag arch --related NoteA
158
+
159
+ # Clustering
160
+ archiver-rag cluster # suggest folder groupings
161
+ archiver-rag cluster --apply # move files automatically
162
+ archiver-rag place <note> # suggest folder for a single note
163
+ archiver-rag place <note> --apply # move it immediately
164
+
165
+ # Config
166
+ archiver-rag config --auto-cluster # enable auto-clustering in the watcher
167
+ archiver-rag config --cluster-threshold 5 # notes before a full re-cluster
168
+
169
+ archiver-rag uninstall # remove all data, service, and MCP registration
170
+ ```
171
+
172
+ ---
173
+
174
+ ## MCP tools (for agents)
175
+
176
+ Once registered, agents have access to 7 tools:
177
+
178
+ | Tool | What it does |
179
+ |---|---|
180
+ | `search_vault` | Semantic search with graph reranking. Accepts a `context_note` to boost wikilink neighbors. |
181
+ | `vault_status` | Vault structure, health diagnostics, tag stats, and recent activity in one call. |
182
+ | `get_connections` | BFS wikilink traversal — outgoing and incoming links up to depth 3. |
183
+ | `move_notes` | Move files and auto-rewrite all `[[wikilinks]]` across the vault. |
184
+ | `log_note` | Create a dated knowledge note; watcher indexes and auto-links it immediately. |
185
+ | `cluster_note` | Suggest a folder for one note based on where its wikilink neighbors live. |
186
+ | `cluster_vault` | Label-propagation clustering of the entire vault with folder suggestions. |
187
+
188
+ ---
189
+
190
+ ## Configuration
191
+
192
+ All runtime config lives at `~/.archiver-rag/config.json`:
193
+
194
+ ```json
195
+ {
196
+ "vault_path": "/path/to/your/vault",
197
+ "install_path": "/Users/you/.archiver-rag",
198
+ "chroma_path": "/Users/you/.archiver-rag/chroma_db",
199
+ "auto_cluster": false,
200
+ "cluster_threshold": 5
201
+ }
202
+ ```
203
+
204
+ `auto_cluster` — automatically suggest and apply folder placement for new notes via the watcher.
205
+ `cluster_threshold` — number of new notes created before triggering a full `cluster_vault` run.
206
+
207
+ ---
208
+
209
+ ## The knowledge graph model
210
+
211
+ The vault is treated as a **knowledge graph**, not a file hierarchy. Notes are nodes; wikilinks are edges. Relationships range from tight (direct links) to loose (semantic proximity surfaced by search).
212
+
213
+ Note types are expressed through frontmatter, not folder structure:
214
+
215
+ ```yaml
216
+ ---
217
+ type: decision
218
+ tags: [architecture, async]
219
+ related: [[AsyncLocalStorage]], [[PrismaExtensions]]
220
+ date: 2026-04-27
221
+ ---
222
+ ```
223
+
224
+ The `## Related` section at the bottom of each note is managed automatically by the auto-linker after every ingest. Don't edit it manually — it will be overwritten.
225
+
226
+ ---
227
+
228
+ ## Roadmap
229
+
230
+ Features on the way:
231
+
232
+ - **RAG-Anything integration** — extend ingestion beyond Markdown to handle PDFs, Office documents, images, and other file types, so the vault can become a true multi-format knowledge base rather than `.md`-only.
233
+ - **Archiver subagents** — dedicated subagents that take over vault management (search, logging, reorganization, clustering) on the main agent's behalf, so the primary agent can delegate knowledge work instead of context-switching into it.
234
+
235
+ ---
236
+
237
+ ## License
238
+
239
+ MIT
@@ -0,0 +1,226 @@
1
+ <p align="center">
2
+ <img src="assets/logo.png" alt="Archiver RAG" width="180" />
3
+ </p>
4
+
5
+ # Archiver RAG
6
+ ### The agent-agnostic memory management system through Obsidian-esque techniques
7
+
8
+ Archiver RAG turns your Obsidian vault into a live, queryable knowledge graph that any MCP-compatible AI agent can search, update, and reorganize — without ever leaving its native interface.
9
+
10
+ Connect it once. Every agent you use (Claude Code, Cursor, Gemini CLI, or your own) gets semantic search, automatic knowledge logging, wikilink-aware graph traversal, and vault health monitoring out of the box.
11
+
12
+ ---
13
+
14
+ ## How it works
15
+
16
+ ```
17
+ Your Obsidian vault (.md files)
18
+ ↓ file watcher + ingest pipeline
19
+ ChromaDB (persistent vector store)
20
+ ↓ MCP server
21
+ Any MCP-compatible agent
22
+ ```
23
+
24
+ Three layers make search smarter than plain embeddings:
25
+
26
+ 1. **Contextual prefix** — each chunk is embedded with its note's metadata (folder, tags, wikilinks), so vectors carry structural context
27
+ 2. **Rich metadata filtering** — ChromaDB stores folder, tags, incoming link count, and wikilinks for filtered retrieval
28
+ 3. **Graph reranking** — after vector search, results are re-scored by wikilink proximity to a context note and hub importance
29
+
30
+ The file watcher runs as a background service. Edit a note in Obsidian, save it, and it's indexed and auto-linked within seconds — no manual sync needed.
31
+
32
+ ---
33
+
34
+ ## Features
35
+
36
+ - **Semantic search** with graph reranking — finds notes by meaning, then boosts results connected via wikilinks
37
+ - **Auto-linking** — after every ingest, appends a `## Related` section with `[[wikilinks]]` to build the knowledge graph automatically
38
+ - **Knowledge logging** — create dated, categorized notes (`decision`, `lesson`, `gotcha`, `pattern`, …) from any agent
39
+ - **Vault health** — single call returns orphaned notes, broken links, missing frontmatter, tag stats, and recent activity
40
+ - **Wikilink-aware reorganization** — move files and every `[[link]]` across the vault is rewritten automatically
41
+ - **Smart clustering** — label-propagation algorithm groups notes by wikilink structure and suggests folder organization
42
+ - **Agent-agnostic** — exposes a standard MCP interface; works with any MCP-compatible client
43
+
44
+ ---
45
+
46
+ ## Requirements
47
+
48
+ - Python >= 3.10
49
+ - [pipx](https://pipx.pypa.io/) (recommended for installation)
50
+ - An Obsidian vault (local `.md` files)
51
+ - An MCP-compatible agent (Claude Code, Cursor, etc.)
52
+
53
+ ---
54
+
55
+ ## Installation
56
+
57
+ ```bash
58
+ pipx install --editable .
59
+ ```
60
+
61
+ > Use `pipx`, not `pip install -e .` — pipx creates an isolated environment and exposes the CLI globally on `PATH`, which is required for MCP registration to find the correct executable.
62
+
63
+ ---
64
+
65
+ ## Setup
66
+
67
+ Run the one-time setup wizard:
68
+
69
+ ```bash
70
+ archiver-rag init
71
+ ```
72
+
73
+ This will:
74
+ 1. Ask for your vault path
75
+ 2. Index your vault into ChromaDB
76
+ 3. Register the MCP server in `~/.claude.json` (or prompt you to do it manually for other clients)
77
+ 4. Install the background watcher as a launchd agent (Mac) or systemd service (Linux)
78
+
79
+ ---
80
+
81
+ ## MCP registration (manual)
82
+
83
+ If you prefer to register manually, add this to your MCP client config:
84
+
85
+ ```json
86
+ {
87
+ "mcpServers": {
88
+ "archiver-rag": {
89
+ "command": "/path/to/archiver-rag",
90
+ "args": ["serve"]
91
+ }
92
+ }
93
+ }
94
+ ```
95
+
96
+ Find the executable path with `which archiver-rag`.
97
+
98
+ For Claude Code specifically, use:
99
+
100
+ ```bash
101
+ claude mcp add --scope user archiver-rag $(which archiver-rag) serve
102
+ ```
103
+
104
+ ---
105
+
106
+ ## Agent instructions (skills)
107
+
108
+ Registering the MCP server gives an agent *access* to the tools — but agents tend to fall back on their own internal memory instead of reaching for the vault. The instruction files in [`skill/`](skill/) fix that: they enforce a **vault-first rule** so the agent searches and stores knowledge in your vault before anything else.
109
+
110
+ **What the skill enforces:**
111
+
112
+ - **Before answering or reading source files** — call `search_vault` first; only fall back to internal memory if the vault returns nothing relevant
113
+ - **When something important is missing from the vault** — proactively `log_note` it. If a fact, decision, or piece of context matters to the overall picture and a `search_vault` came back empty, record it so the knowledge graph grows instead of letting that context die in a single session
114
+ - **After solving a non-trivial problem** — call `log_note` to capture the decision/lesson/gotcha back into the vault
115
+ - **The vault is the authoritative memory system** — internal agent memory is a fallback only
116
+
117
+ A version is provided for each agent, since each loads instructions differently:
118
+
119
+ | Agent | File | Install to |
120
+ |---|---|---|
121
+ | Claude Code | [`skill/claude-code/SKILL.md`](skill/claude-code/SKILL.md) | `~/.claude/skills/archiver-rag/SKILL.md` (on-demand skill) |
122
+ | OpenCode | [`skill/opencode/AGENTS.md`](skill/opencode/AGENTS.md) | project root `AGENTS.md` or `~/.config/opencode/AGENTS.md` |
123
+ | Codex CLI | [`skill/codex/AGENTS.md`](skill/codex/AGENTS.md) | project root `AGENTS.md` or `~/.codex/AGENTS.md` |
124
+ | GitHub Copilot | [`skill/copilot/copilot-instructions.md`](skill/copilot/copilot-instructions.md) | `.github/copilot-instructions.md` |
125
+
126
+ Each file is self-contained — it includes the MCP registration snippet for that agent plus the full vault-first rules and tool reference. For Claude Code the file is an on-demand skill; for the others it's an always-on instruction file (loaded into every session), which makes the vault-first behavior unconditional.
127
+
128
+ ---
129
+
130
+ ## CLI reference
131
+
132
+ ```bash
133
+ archiver-rag init # one-time setup wizard
134
+ archiver-rag start # start the background watcher service
135
+ archiver-rag stop # stop the service
136
+ archiver-rag restart # restart the service
137
+ archiver-rag status # check if service is running
138
+ archiver-rag index # force re-index the entire vault
139
+ archiver-rag search "query" # test semantic search from the terminal
140
+ archiver-rag health # chunk count and index peek
141
+ archiver-rag logs # tail the service log
142
+
143
+ # Knowledge logging
144
+ archiver-rag log "Title" --type decision --tag arch --related NoteA
145
+
146
+ # Clustering
147
+ archiver-rag cluster # suggest folder groupings
148
+ archiver-rag cluster --apply # move files automatically
149
+ archiver-rag place <note> # suggest folder for a single note
150
+ archiver-rag place <note> --apply # move it immediately
151
+
152
+ # Config
153
+ archiver-rag config --auto-cluster # enable auto-clustering in the watcher
154
+ archiver-rag config --cluster-threshold 5 # notes before a full re-cluster
155
+
156
+ archiver-rag uninstall # remove all data, service, and MCP registration
157
+ ```
158
+
159
+ ---
160
+
161
+ ## MCP tools (for agents)
162
+
163
+ Once registered, agents have access to 7 tools:
164
+
165
+ | Tool | What it does |
166
+ |---|---|
167
+ | `search_vault` | Semantic search with graph reranking. Accepts a `context_note` to boost wikilink neighbors. |
168
+ | `vault_status` | Vault structure, health diagnostics, tag stats, and recent activity in one call. |
169
+ | `get_connections` | BFS wikilink traversal — outgoing and incoming links up to depth 3. |
170
+ | `move_notes` | Move files and auto-rewrite all `[[wikilinks]]` across the vault. |
171
+ | `log_note` | Create a dated knowledge note; watcher indexes and auto-links it immediately. |
172
+ | `cluster_note` | Suggest a folder for one note based on where its wikilink neighbors live. |
173
+ | `cluster_vault` | Label-propagation clustering of the entire vault with folder suggestions. |
174
+
175
+ ---
176
+
177
+ ## Configuration
178
+
179
+ All runtime config lives at `~/.archiver-rag/config.json`:
180
+
181
+ ```json
182
+ {
183
+ "vault_path": "/path/to/your/vault",
184
+ "install_path": "/Users/you/.archiver-rag",
185
+ "chroma_path": "/Users/you/.archiver-rag/chroma_db",
186
+ "auto_cluster": false,
187
+ "cluster_threshold": 5
188
+ }
189
+ ```
190
+
191
+ `auto_cluster` — automatically suggest and apply folder placement for new notes via the watcher.
192
+ `cluster_threshold` — number of new notes created before triggering a full `cluster_vault` run.
193
+
194
+ ---
195
+
196
+ ## The knowledge graph model
197
+
198
+ The vault is treated as a **knowledge graph**, not a file hierarchy. Notes are nodes; wikilinks are edges. Relationships range from tight (direct links) to loose (semantic proximity surfaced by search).
199
+
200
+ Note types are expressed through frontmatter, not folder structure:
201
+
202
+ ```yaml
203
+ ---
204
+ type: decision
205
+ tags: [architecture, async]
206
+ related: [[AsyncLocalStorage]], [[PrismaExtensions]]
207
+ date: 2026-04-27
208
+ ---
209
+ ```
210
+
211
+ The `## Related` section at the bottom of each note is managed automatically by the auto-linker after every ingest. Don't edit it manually — it will be overwritten.
212
+
213
+ ---
214
+
215
+ ## Roadmap
216
+
217
+ Features on the way:
218
+
219
+ - **RAG-Anything integration** — extend ingestion beyond Markdown to handle PDFs, Office documents, images, and other file types, so the vault can become a true multi-format knowledge base rather than `.md`-only.
220
+ - **Archiver subagents** — dedicated subagents that take over vault management (search, logging, reorganization, clustering) on the main agent's behalf, so the primary agent can delegate knowledge work instead of context-switching into it.
221
+
222
+ ---
223
+
224
+ ## License
225
+
226
+ MIT
@@ -0,0 +1 @@
1
+ __version__ = "1.0.0"
@@ -0,0 +1,14 @@
1
+ from archiver_rag.core.db import collection
2
+
3
+ count = collection.count()
4
+ print(f"Total chunks in index: {count}")
5
+
6
+ if count > 0:
7
+ # Peek at first few chunks
8
+ results = collection.peek(limit=3)
9
+ for i, (doc, meta) in enumerate(zip(results["documents"], results["metadatas"])):
10
+ print(f"\n--- Chunk {i+1} ---")
11
+ print(f"Source: {meta['source']}")
12
+ print(f"Preview: {doc[:200]}")
13
+ else:
14
+ print("Index is empty — ingest hasn't run or failed silently")