fastref 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fastref-0.1.0/PKG-INFO +230 -0
- fastref-0.1.0/README.md +218 -0
- fastref-0.1.0/pyproject.toml +33 -0
- fastref-0.1.0/pyproject.toml.orig +28 -0
- fastref-0.1.0/src/fastref/__init__.py +6 -0
- fastref-0.1.0/src/fastref/cli.py +887 -0
- fastref-0.1.0/src/fastref/config.py +87 -0
- fastref-0.1.0/src/fastref/embed.py +104 -0
- fastref-0.1.0/src/fastref/harvester.py +715 -0
- fastref-0.1.0/src/fastref/indexer.py +231 -0
- fastref-0.1.0/src/fastref/models.py +31 -0
- fastref-0.1.0/src/fastref/normalizer.py +95 -0
- fastref-0.1.0/src/fastref/searcher.py +455 -0
- fastref-0.1.0/src/fastref/security.py +152 -0
- fastref-0.1.0/src/fastref/skill.py +162 -0
- fastref-0.1.0/src/fastref/toc.py +95 -0
fastref-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fastref
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Fast offline documentation reference desk and skill generator for AI agents
|
|
5
|
+
Keywords: ai-agents,documentation,offline-docs,sqlite-fts5,skills,bm25
|
|
6
|
+
Author: Rafael Darder
|
|
7
|
+
Author-email: Rafael Darder <darder@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Requires-Dist: litert-lm-api>=0.18.0
|
|
10
|
+
Requires-Python: >=3.12
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# fastref 📚⚡
|
|
14
|
+
|
|
15
|
+
> Fast offline documentation reference desk and skill generator for AI coding agents.
|
|
16
|
+
|
|
17
|
+
`fastref` gives AI coding agents an instant, local offline reference desk. It indexes documentation from any tool, framework, or site into a high-performance **SQLite FTS5 database with BM25 ranking**, and automatically generates self-contained **Agent Skills** (under `~/.agents/skills/` or `./.agents/skills/`).
|
|
18
|
+
|
|
19
|
+
Coding agents can query authoritative documentation and code examples locally in sub-milliseconds without hallucinating APIs, wasting web search latency, or blowing context windows.
|
|
20
|
+
|
|
21
|
+
---
|
|
22
|
+
|
|
23
|
+
## The Ingestion Pattern
|
|
24
|
+
|
|
25
|
+
Instead of fragile ad-hoc scrapers, `fastref` unifies documentation ingestion around two core patterns:
|
|
26
|
+
|
|
27
|
+
1. **Harvest & Assemble (`fastref harvest`)**: For tools and websites that publish an `llms.txt` file (or website root). `fastref` spiders and crawls linked markdown documentation, resolves relative references, checks external domain boundaries, and bundles pages into clean, structured documentation chunks.
|
|
28
|
+
2. **Fetch & Index Structured JSON (`fastref add`)**: For tools and documentation systems that generate machine-readable search indexes (such as Fuse.js, Algolia, or MkDocs-style search JSON endpoints). `fastref` ingests both flat arrays and nested section trees directly.
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Features
|
|
33
|
+
|
|
34
|
+
- **Dual Ingestion Engine**:
|
|
35
|
+
- **`llms.txt` Spider & Harvester**: Discovers and indexes documentation hierarchies from `llms.txt` files or site roots, featuring domain boundary controls, crawl depth limits, and link preview (`--dry-run`).
|
|
36
|
+
- **Structured JSON Ingestor**: Automatically normalizes both flat lists and nested hierarchical search indexes into a unified schema.
|
|
37
|
+
- **Hybrid Search (BM25 + Dense Vectors)**:
|
|
38
|
+
- **Lexical BM25**: Native SQLite FTS5 for exact keywords, function names, and prefix matching.
|
|
39
|
+
- **On-Device Semantic Vectors**: Google's **LiteRT-LM** with **EmbeddingGemma 2 Text-270M** (quantized to 157 MB, 256d normalized Matryoshka embeddings) for conceptual retrieval without external APIs or background daemons.
|
|
40
|
+
- **Reciprocal Rank Fusion (RRF)**: Merges lexical and semantic rankings for optimal recall and precision.
|
|
41
|
+
- **Zero-Daemon Architecture**: In-process inference starts in ~180ms, eliminating background sockets, daemons, and orphaned processes.
|
|
42
|
+
- **Prompt Injection & SSRF Hardened**: Defends against indirect prompt injection by stripping invisible unicode/HTML comments, sanitizing skill frontmatter, blocking private/metadata network IPs, and framing retrieved docs in untrusted reference envelopes.
|
|
43
|
+
- **Automatic Agent Skill Generation**: Creates clean `SKILL.md` definitions with security boundary notices and automatically bridges them into active agent harnesses (`~/.agents/skills/`, `~/.gemini/config/skills/`, `~/.claude/skills/`, `~/.config/goose/skills/`).
|
|
44
|
+
- **Harness-Agnostic**: Compatible with Antigravity, Claude Code, Cursor, Codex, Goose, Aider, and any agent that supports shell commands or standard skill directories.
|
|
45
|
+
- **XDG-Compliant Caching**: Keeps database artifacts neatly organized in `$XDG_CACHE_HOME/fastref` (`~/.cache/fastref`).
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## Installation
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
# Using pip
|
|
53
|
+
pip install fastref
|
|
54
|
+
|
|
55
|
+
# Or with uv
|
|
56
|
+
uv tool install fastref
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
---
|
|
60
|
+
|
|
61
|
+
## Quick Start
|
|
62
|
+
|
|
63
|
+
### 1. Ingest Documentation & Generate a Skill
|
|
64
|
+
|
|
65
|
+
#### Pattern A: Harvest from `llms.txt` or Site Root
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
# Harvest from an llms.txt index
|
|
69
|
+
fastref harvest mytool https://example.com/llms.txt
|
|
70
|
+
|
|
71
|
+
# Or point directly to the website root (auto-detects /llms.txt)
|
|
72
|
+
fastref harvest mytool https://example.com
|
|
73
|
+
|
|
74
|
+
# Preview discovered links and external domains without fetching
|
|
75
|
+
fastref harvest mytool https://example.com/llms.txt --dry-run
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
#### Pattern B: Ingest Pre-Built Structured Search JSON
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
# Index structured search index JSON directly from a URL or local file
|
|
82
|
+
fastref add mytool https://example.com/search_index.json
|
|
83
|
+
fastref add mytool ./path/to/search_index.json
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
By default, both commands:
|
|
87
|
+
1. Compile an SQLite FTS5 database to `~/.cache/fastref/<name>.db`.
|
|
88
|
+
2. Archive the raw manifest/index to `~/.agents/skills/<name>/resources/raw_index.json`.
|
|
89
|
+
3. Write `~/.agents/skills/<name>/SKILL.md` with trigger rules for AI agents.
|
|
90
|
+
4. Auto-bridge the skill into installed agent environments (`~/.gemini/config/skills/`, `~/.claude/skills/`, `~/.config/goose/skills/`).
|
|
91
|
+
|
|
92
|
+
### 2. Search Documentation
|
|
93
|
+
|
|
94
|
+
`fastref search` provides three adaptive profiles:
|
|
95
|
+
- **`find`** (default when vectors are enabled): Returns the top focused code & API snippets (default limit: 3) without catalog clutter.
|
|
96
|
+
- **`directory`** (default when vectors are disabled): Returns a catalog of matching topic titles (capped at 20) plus top snippets.
|
|
97
|
+
- **`lucky`**: Prints the top matching document in full, followed by 5 related topic links.
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
# Focused search (find profile: top snippets, hybrid RRF when vectors enabled)
|
|
101
|
+
fastref search mytool "authentication"
|
|
102
|
+
fastref search mytool "middleware" --limit 5
|
|
103
|
+
|
|
104
|
+
# Directory profile: view matching topic catalog + top snippets
|
|
105
|
+
fastref search mytool "jwt_validation" --directory
|
|
106
|
+
|
|
107
|
+
# Lucky profile: print full content of top match + related topic links
|
|
108
|
+
fastref search mytool "installation" --lucky
|
|
109
|
+
|
|
110
|
+
# Show full content of all matches directly
|
|
111
|
+
fastref search mytool "config" --full
|
|
112
|
+
|
|
113
|
+
# Retrieve full documentation by exact URL or topic ID
|
|
114
|
+
fastref get mytool "/docs/auth#jwt"
|
|
115
|
+
fastref get mytool "/docs/users" "/docs/roles"
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### 3. Optional: Enable On-Device Vector & Hybrid Search
|
|
119
|
+
|
|
120
|
+
Vector features are disabled by default to keep the CLI lightweight and zero-noise. When enabled, `fastref` uses Google's **LiteRT-LM** and downloads the **EmbeddingGemma 2 Text-270M** model (157 MB).
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
# Enable vector features (downloads the model artifact on first run)
|
|
124
|
+
fastref vector enable
|
|
125
|
+
|
|
126
|
+
# Backfill vector embeddings for an already indexed library
|
|
127
|
+
fastref vector index mytool
|
|
128
|
+
|
|
129
|
+
# Check vector status and indexed libraries
|
|
130
|
+
fastref vector status
|
|
131
|
+
|
|
132
|
+
# When vectors are enabled, fastref search automatically defaults to hybrid RRF:
|
|
133
|
+
fastref search mytool "how to handle webhooks asynchronously"
|
|
134
|
+
|
|
135
|
+
# Or query explicitly via semantic or lexical mode:
|
|
136
|
+
fastref search mytool "reactive state management" --semantic
|
|
137
|
+
fastref search mytool "on_mount_hook" --lexical
|
|
138
|
+
|
|
139
|
+
# Disable vector features at any time
|
|
140
|
+
fastref vector disable
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
### 4. Explore Documentation Structure
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
# View table of contents chapters and topic counts
|
|
147
|
+
fastref toc mytool
|
|
148
|
+
|
|
149
|
+
# Filter chapters matching a keyword
|
|
150
|
+
fastref toc mytool auth
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
### 5. Manage Indexed Documentation
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
# List all indexed tools, chunk counts, vector status, and database sizes
|
|
157
|
+
fastref list
|
|
158
|
+
|
|
159
|
+
# Remove an indexed tool (both cache database and skill folder)
|
|
160
|
+
fastref remove mytool
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
## How Agents Use `fastref`
|
|
167
|
+
|
|
168
|
+
When `fastref` generates a skill, it writes a clean `SKILL.md` into your agent's skill directory. When you ask your agent to implement a feature or debug code using an indexed tool:
|
|
169
|
+
|
|
170
|
+
1. **Discovery**: The agent runs `fastref search <name> "<query>"` to locate exact components, function signatures, or concepts.
|
|
171
|
+
2. **Context-Efficient Retrieval**: The agent inspects targeted snippets or pulls complete topic documentation via `--lucky` or `fastref get <name> <url>`, without loading megabytes of irrelevant HTML into context.
|
|
172
|
+
3. **Accurate Implementation**: The agent codes against authoritative local documentation and tested code examples.
|
|
173
|
+
|
|
174
|
+
---
|
|
175
|
+
|
|
176
|
+
## CLI Reference
|
|
177
|
+
|
|
178
|
+
### `fastref harvest <name> <source>`
|
|
179
|
+
Harvest documentation from an `llms.txt` file or website root:
|
|
180
|
+
- `<name>`: Identifier for the tool or library.
|
|
181
|
+
- `<source>`: URL or local path to `llms.txt` (or website root URL).
|
|
182
|
+
- `--max-docs <N>`: Maximum documents to spider and index (default: 60).
|
|
183
|
+
- `--max-depth <D>`: Maximum spider crawl depth (default: 2).
|
|
184
|
+
- `--dry-run`: Preview discovered links and domain categorization without fetching or indexing.
|
|
185
|
+
- `--allow-urls <prefixes>`: Comma-separated URL or domain prefixes to allow (e.g. `https://raw.githubusercontent.com/...`).
|
|
186
|
+
- `--allow-all-urls`: Allow fetching documentation from any external URL or domain.
|
|
187
|
+
- `--allow-remote`: Allow network fetching when source is a local file.
|
|
188
|
+
- `--allow-private-ips`: Allow fetching from private, loopback, or local network IPs (e.g. localhost during dev).
|
|
189
|
+
- `-p`, `--project`: Install the skill into the local project directory (`./.agents/skills/`) instead of global user skills.
|
|
190
|
+
- `--skills-dir <dir>`: Custom destination directory for the generated skill.
|
|
191
|
+
- `--cache-dir <dir>`: Custom destination directory for the SQLite database.
|
|
192
|
+
- `--description <desc>`: Custom description for the YAML frontmatter in `SKILL.md`.
|
|
193
|
+
- `--no-skill`: Only build the SQLite index without generating an agent skill.
|
|
194
|
+
|
|
195
|
+
### `fastref add <name> <source>`
|
|
196
|
+
Index documentation from a pre-built machine-readable search index JSON:
|
|
197
|
+
- `<name>`: Identifier for the tool or library.
|
|
198
|
+
- `<source>`: HTTP/HTTPS URL or local path to `search_index.json`.
|
|
199
|
+
- `--allow-private-ips`: Allow fetching from private, loopback, or local network IPs.
|
|
200
|
+
- `-p`, `--project`: Install the skill into the local project directory (`./.agents/skills/`) instead of global user skills.
|
|
201
|
+
- `--skills-dir <dir>`: Custom destination directory for the generated skill.
|
|
202
|
+
- `--cache-dir <dir>`: Custom destination directory for the SQLite database.
|
|
203
|
+
- `--description <desc>`: Custom description for the YAML frontmatter in `SKILL.md`.
|
|
204
|
+
- `--no-skill`: Only build the SQLite index without generating an agent skill.
|
|
205
|
+
|
|
206
|
+
### `fastref search <name> <query>`
|
|
207
|
+
Search documentation with BM25 relevance ranking:
|
|
208
|
+
- `<name>`: Identifier for the indexed tool.
|
|
209
|
+
- `<query>`: Search query (supports SQLite FTS5 match expressions).
|
|
210
|
+
- `--lucky`: Print the entire top match immediately and list remaining candidate snippets.
|
|
211
|
+
- `--full`: Include full topic content for all matches directly.
|
|
212
|
+
- `--limit <N>`: Maximum results to return (default: 8).
|
|
213
|
+
|
|
214
|
+
### `fastref get <name> <url1> [url2 ...]`
|
|
215
|
+
Fetch and display the full documentation text for one or more documents by URL or topic ID.
|
|
216
|
+
|
|
217
|
+
### `fastref toc <name> [filter]`
|
|
218
|
+
Display Table of Contents chapters, topic counts, and sample titles, optionally filtered by keyword.
|
|
219
|
+
|
|
220
|
+
### `fastref list`
|
|
221
|
+
List all indexed tools, their chunk counts, and database sizes.
|
|
222
|
+
|
|
223
|
+
### `fastref remove <name>`
|
|
224
|
+
Remove an indexed tool's SQLite database cache and associated skill directories.
|
|
225
|
+
|
|
226
|
+
---
|
|
227
|
+
|
|
228
|
+
## License
|
|
229
|
+
|
|
230
|
+
MIT License.
|
fastref-0.1.0/README.md
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
# fastref 📚⚡
|
|
2
|
+
|
|
3
|
+
> Fast offline documentation reference desk and skill generator for AI coding agents.
|
|
4
|
+
|
|
5
|
+
`fastref` gives AI coding agents an instant, local offline reference desk. It indexes documentation from any tool, framework, or site into a high-performance **SQLite FTS5 database with BM25 ranking**, and automatically generates self-contained **Agent Skills** (under `~/.agents/skills/` or `./.agents/skills/`).
|
|
6
|
+
|
|
7
|
+
Coding agents can query authoritative documentation and code examples locally in sub-milliseconds without hallucinating APIs, wasting web search latency, or blowing context windows.
|
|
8
|
+
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
## The Ingestion Pattern
|
|
12
|
+
|
|
13
|
+
Instead of fragile ad-hoc scrapers, `fastref` unifies documentation ingestion around two core patterns:
|
|
14
|
+
|
|
15
|
+
1. **Harvest & Assemble (`fastref harvest`)**: For tools and websites that publish an `llms.txt` file (or website root). `fastref` spiders and crawls linked markdown documentation, resolves relative references, checks external domain boundaries, and bundles pages into clean, structured documentation chunks.
|
|
16
|
+
2. **Fetch & Index Structured JSON (`fastref add`)**: For tools and documentation systems that generate machine-readable search indexes (such as Fuse.js, Algolia, or MkDocs-style search JSON endpoints). `fastref` ingests both flat arrays and nested section trees directly.
|
|
17
|
+
|
|
18
|
+
---
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
|
|
22
|
+
- **Dual Ingestion Engine**:
|
|
23
|
+
- **`llms.txt` Spider & Harvester**: Discovers and indexes documentation hierarchies from `llms.txt` files or site roots, featuring domain boundary controls, crawl depth limits, and link preview (`--dry-run`).
|
|
24
|
+
- **Structured JSON Ingestor**: Automatically normalizes both flat lists and nested hierarchical search indexes into a unified schema.
|
|
25
|
+
- **Hybrid Search (BM25 + Dense Vectors)**:
|
|
26
|
+
- **Lexical BM25**: Native SQLite FTS5 for exact keywords, function names, and prefix matching.
|
|
27
|
+
- **On-Device Semantic Vectors**: Google's **LiteRT-LM** with **EmbeddingGemma 2 Text-270M** (quantized to 157 MB, 256d normalized Matryoshka embeddings) for conceptual retrieval without external APIs or background daemons.
|
|
28
|
+
- **Reciprocal Rank Fusion (RRF)**: Merges lexical and semantic rankings for optimal recall and precision.
|
|
29
|
+
- **Zero-Daemon Architecture**: In-process inference starts in ~180ms, eliminating background sockets, daemons, and orphaned processes.
|
|
30
|
+
- **Prompt Injection & SSRF Hardened**: Defends against indirect prompt injection by stripping invisible unicode/HTML comments, sanitizing skill frontmatter, blocking private/metadata network IPs, and framing retrieved docs in untrusted reference envelopes.
|
|
31
|
+
- **Automatic Agent Skill Generation**: Creates clean `SKILL.md` definitions with security boundary notices and automatically bridges them into active agent harnesses (`~/.agents/skills/`, `~/.gemini/config/skills/`, `~/.claude/skills/`, `~/.config/goose/skills/`).
|
|
32
|
+
- **Harness-Agnostic**: Compatible with Antigravity, Claude Code, Cursor, Codex, Goose, Aider, and any agent that supports shell commands or standard skill directories.
|
|
33
|
+
- **XDG-Compliant Caching**: Keeps database artifacts neatly organized in `$XDG_CACHE_HOME/fastref` (`~/.cache/fastref`).
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## Installation
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
# Using pip
|
|
41
|
+
pip install fastref
|
|
42
|
+
|
|
43
|
+
# Or with uv
|
|
44
|
+
uv tool install fastref
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## Quick Start
|
|
50
|
+
|
|
51
|
+
### 1. Ingest Documentation & Generate a Skill
|
|
52
|
+
|
|
53
|
+
#### Pattern A: Harvest from `llms.txt` or Site Root
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
# Harvest from an llms.txt index
|
|
57
|
+
fastref harvest mytool https://example.com/llms.txt
|
|
58
|
+
|
|
59
|
+
# Or point directly to the website root (auto-detects /llms.txt)
|
|
60
|
+
fastref harvest mytool https://example.com
|
|
61
|
+
|
|
62
|
+
# Preview discovered links and external domains without fetching
|
|
63
|
+
fastref harvest mytool https://example.com/llms.txt --dry-run
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
#### Pattern B: Ingest Pre-Built Structured Search JSON
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
# Index structured search index JSON directly from a URL or local file
|
|
70
|
+
fastref add mytool https://example.com/search_index.json
|
|
71
|
+
fastref add mytool ./path/to/search_index.json
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
By default, both commands:
|
|
75
|
+
1. Compile an SQLite FTS5 database to `~/.cache/fastref/<name>.db`.
|
|
76
|
+
2. Archive the raw manifest/index to `~/.agents/skills/<name>/resources/raw_index.json`.
|
|
77
|
+
3. Write `~/.agents/skills/<name>/SKILL.md` with trigger rules for AI agents.
|
|
78
|
+
4. Auto-bridge the skill into installed agent environments (`~/.gemini/config/skills/`, `~/.claude/skills/`, `~/.config/goose/skills/`).
|
|
79
|
+
|
|
80
|
+
### 2. Search Documentation
|
|
81
|
+
|
|
82
|
+
`fastref search` provides three adaptive profiles:
|
|
83
|
+
- **`find`** (default when vectors are enabled): Returns the top focused code & API snippets (default limit: 3) without catalog clutter.
|
|
84
|
+
- **`directory`** (default when vectors are disabled): Returns a catalog of matching topic titles (capped at 20) plus top snippets.
|
|
85
|
+
- **`lucky`**: Prints the top matching document in full, followed by 5 related topic links.
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
# Focused search (find profile: top snippets, hybrid RRF when vectors enabled)
|
|
89
|
+
fastref search mytool "authentication"
|
|
90
|
+
fastref search mytool "middleware" --limit 5
|
|
91
|
+
|
|
92
|
+
# Directory profile: view matching topic catalog + top snippets
|
|
93
|
+
fastref search mytool "jwt_validation" --directory
|
|
94
|
+
|
|
95
|
+
# Lucky profile: print full content of top match + related topic links
|
|
96
|
+
fastref search mytool "installation" --lucky
|
|
97
|
+
|
|
98
|
+
# Show full content of all matches directly
|
|
99
|
+
fastref search mytool "config" --full
|
|
100
|
+
|
|
101
|
+
# Retrieve full documentation by exact URL or topic ID
|
|
102
|
+
fastref get mytool "/docs/auth#jwt"
|
|
103
|
+
fastref get mytool "/docs/users" "/docs/roles"
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### 3. Optional: Enable On-Device Vector & Hybrid Search
|
|
107
|
+
|
|
108
|
+
Vector features are disabled by default to keep the CLI lightweight and zero-noise. When enabled, `fastref` uses Google's **LiteRT-LM** and downloads the **EmbeddingGemma 2 Text-270M** model (157 MB).
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
# Enable vector features (downloads the model artifact on first run)
|
|
112
|
+
fastref vector enable
|
|
113
|
+
|
|
114
|
+
# Backfill vector embeddings for an already indexed library
|
|
115
|
+
fastref vector index mytool
|
|
116
|
+
|
|
117
|
+
# Check vector status and indexed libraries
|
|
118
|
+
fastref vector status
|
|
119
|
+
|
|
120
|
+
# When vectors are enabled, fastref search automatically defaults to hybrid RRF:
|
|
121
|
+
fastref search mytool "how to handle webhooks asynchronously"
|
|
122
|
+
|
|
123
|
+
# Or query explicitly via semantic or lexical mode:
|
|
124
|
+
fastref search mytool "reactive state management" --semantic
|
|
125
|
+
fastref search mytool "on_mount_hook" --lexical
|
|
126
|
+
|
|
127
|
+
# Disable vector features at any time
|
|
128
|
+
fastref vector disable
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### 4. Explore Documentation Structure
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
# View table of contents chapters and topic counts
|
|
135
|
+
fastref toc mytool
|
|
136
|
+
|
|
137
|
+
# Filter chapters matching a keyword
|
|
138
|
+
fastref toc mytool auth
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### 5. Manage Indexed Documentation
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
# List all indexed tools, chunk counts, vector status, and database sizes
|
|
145
|
+
fastref list
|
|
146
|
+
|
|
147
|
+
# Remove an indexed tool (both cache database and skill folder)
|
|
148
|
+
fastref remove mytool
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
---
|
|
153
|
+
|
|
154
|
+
## How Agents Use `fastref`
|
|
155
|
+
|
|
156
|
+
When `fastref` generates a skill, it writes a clean `SKILL.md` into your agent's skill directory. When you ask your agent to implement a feature or debug code using an indexed tool:
|
|
157
|
+
|
|
158
|
+
1. **Discovery**: The agent runs `fastref search <name> "<query>"` to locate exact components, function signatures, or concepts.
|
|
159
|
+
2. **Context-Efficient Retrieval**: The agent inspects targeted snippets or pulls complete topic documentation via `--lucky` or `fastref get <name> <url>`, without loading megabytes of irrelevant HTML into context.
|
|
160
|
+
3. **Accurate Implementation**: The agent codes against authoritative local documentation and tested code examples.
|
|
161
|
+
|
|
162
|
+
---
|
|
163
|
+
|
|
164
|
+
## CLI Reference
|
|
165
|
+
|
|
166
|
+
### `fastref harvest <name> <source>`
|
|
167
|
+
Harvest documentation from an `llms.txt` file or website root:
|
|
168
|
+
- `<name>`: Identifier for the tool or library.
|
|
169
|
+
- `<source>`: URL or local path to `llms.txt` (or website root URL).
|
|
170
|
+
- `--max-docs <N>`: Maximum documents to spider and index (default: 60).
|
|
171
|
+
- `--max-depth <D>`: Maximum spider crawl depth (default: 2).
|
|
172
|
+
- `--dry-run`: Preview discovered links and domain categorization without fetching or indexing.
|
|
173
|
+
- `--allow-urls <prefixes>`: Comma-separated URL or domain prefixes to allow (e.g. `https://raw.githubusercontent.com/...`).
|
|
174
|
+
- `--allow-all-urls`: Allow fetching documentation from any external URL or domain.
|
|
175
|
+
- `--allow-remote`: Allow network fetching when source is a local file.
|
|
176
|
+
- `--allow-private-ips`: Allow fetching from private, loopback, or local network IPs (e.g. localhost during dev).
|
|
177
|
+
- `-p`, `--project`: Install the skill into the local project directory (`./.agents/skills/`) instead of global user skills.
|
|
178
|
+
- `--skills-dir <dir>`: Custom destination directory for the generated skill.
|
|
179
|
+
- `--cache-dir <dir>`: Custom destination directory for the SQLite database.
|
|
180
|
+
- `--description <desc>`: Custom description for the YAML frontmatter in `SKILL.md`.
|
|
181
|
+
- `--no-skill`: Only build the SQLite index without generating an agent skill.
|
|
182
|
+
|
|
183
|
+
### `fastref add <name> <source>`
|
|
184
|
+
Index documentation from a pre-built machine-readable search index JSON:
|
|
185
|
+
- `<name>`: Identifier for the tool or library.
|
|
186
|
+
- `<source>`: HTTP/HTTPS URL or local path to `search_index.json`.
|
|
187
|
+
- `--allow-private-ips`: Allow fetching from private, loopback, or local network IPs.
|
|
188
|
+
- `-p`, `--project`: Install the skill into the local project directory (`./.agents/skills/`) instead of global user skills.
|
|
189
|
+
- `--skills-dir <dir>`: Custom destination directory for the generated skill.
|
|
190
|
+
- `--cache-dir <dir>`: Custom destination directory for the SQLite database.
|
|
191
|
+
- `--description <desc>`: Custom description for the YAML frontmatter in `SKILL.md`.
|
|
192
|
+
- `--no-skill`: Only build the SQLite index without generating an agent skill.
|
|
193
|
+
|
|
194
|
+
### `fastref search <name> <query>`
|
|
195
|
+
Search documentation with BM25 relevance ranking:
|
|
196
|
+
- `<name>`: Identifier for the indexed tool.
|
|
197
|
+
- `<query>`: Search query (supports SQLite FTS5 match expressions).
|
|
198
|
+
- `--lucky`: Print the entire top match immediately and list remaining candidate snippets.
|
|
199
|
+
- `--full`: Include full topic content for all matches directly.
|
|
200
|
+
- `--limit <N>`: Maximum results to return (default: 8).
|
|
201
|
+
|
|
202
|
+
### `fastref get <name> <url1> [url2 ...]`
|
|
203
|
+
Fetch and display the full documentation text for one or more documents by URL or topic ID.
|
|
204
|
+
|
|
205
|
+
### `fastref toc <name> [filter]`
|
|
206
|
+
Display Table of Contents chapters, topic counts, and sample titles, optionally filtered by keyword.
|
|
207
|
+
|
|
208
|
+
### `fastref list`
|
|
209
|
+
List all indexed tools, their chunk counts, and database sizes.
|
|
210
|
+
|
|
211
|
+
### `fastref remove <name>`
|
|
212
|
+
Remove an indexed tool's SQLite database cache and associated skill directories.
|
|
213
|
+
|
|
214
|
+
---
|
|
215
|
+
|
|
216
|
+
## License
|
|
217
|
+
|
|
218
|
+
MIT License.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "fastref"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Fast offline documentation reference desk and skill generator for AI agents"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
keywords = [
|
|
8
|
+
"ai-agents",
|
|
9
|
+
"documentation",
|
|
10
|
+
"offline-docs",
|
|
11
|
+
"sqlite-fts5",
|
|
12
|
+
"skills",
|
|
13
|
+
"bm25",
|
|
14
|
+
]
|
|
15
|
+
requires-python = ">=3.12"
|
|
16
|
+
dependencies = ["litert-lm-api>=0.18.0"]
|
|
17
|
+
|
|
18
|
+
[[project.authors]]
|
|
19
|
+
name = "Rafael Darder"
|
|
20
|
+
email = "darder@gmail.com"
|
|
21
|
+
|
|
22
|
+
[project.scripts]
|
|
23
|
+
fastref = "fastref:main"
|
|
24
|
+
|
|
25
|
+
[build-system]
|
|
26
|
+
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
27
|
+
build-backend = "uv_build"
|
|
28
|
+
|
|
29
|
+
[dependency-groups]
|
|
30
|
+
dev = [
|
|
31
|
+
"pytest>=9.1.1",
|
|
32
|
+
"ruff>=0.16.9",
|
|
33
|
+
]
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "fastref"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Fast offline documentation reference desk and skill generator for AI agents"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
keywords = ["ai-agents", "documentation", "offline-docs", "sqlite-fts5", "skills", "bm25"]
|
|
8
|
+
|
|
9
|
+
authors = [
|
|
10
|
+
{ name = "Rafael Darder", email = "darder@gmail.com" }
|
|
11
|
+
]
|
|
12
|
+
requires-python = ">=3.12"
|
|
13
|
+
dependencies = [
|
|
14
|
+
"litert-lm-api>=0.18.0",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.scripts]
|
|
18
|
+
fastref = "fastref:main"
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["uv_build>=0.12.5,<0.13.0"]
|
|
22
|
+
build-backend = "uv_build"
|
|
23
|
+
|
|
24
|
+
[dependency-groups]
|
|
25
|
+
dev = [
|
|
26
|
+
"pytest>=9.1.1",
|
|
27
|
+
"ruff>=0.16.9",
|
|
28
|
+
]
|