haskie 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- haskie-0.13.0/PKG-INFO +359 -0
- haskie-0.13.0/README.md +326 -0
- haskie-0.13.0/pyproject.toml +84 -0
- haskie-0.13.0/pyproject.toml.orig +87 -0
- haskie-0.13.0/src/haskie/__init__.py +7 -0
- haskie-0.13.0/src/haskie/__main__.py +6 -0
- haskie-0.13.0/src/haskie/api/__init__.py +21 -0
- haskie-0.13.0/src/haskie/api/collections.py +200 -0
- haskie-0.13.0/src/haskie/api/common.py +33 -0
- haskie-0.13.0/src/haskie/api/documents.py +285 -0
- haskie-0.13.0/src/haskie/api/operations.py +57 -0
- haskie-0.13.0/src/haskie/api/search.py +250 -0
- haskie-0.13.0/src/haskie/api/settings.py +179 -0
- haskie-0.13.0/src/haskie/app.py +186 -0
- haskie-0.13.0/src/haskie/audit.py +226 -0
- haskie-0.13.0/src/haskie/catalogue/__init__.py +1 -0
- haskie-0.13.0/src/haskie/catalogue/calibrate.py +192 -0
- haskie-0.13.0/src/haskie/catalogue/catalogue.py +273 -0
- haskie-0.13.0/src/haskie/catalogue/seed.sql +118 -0
- haskie-0.13.0/src/haskie/claude.py +240 -0
- haskie-0.13.0/src/haskie/claude_code/rules/haskie.md +26 -0
- haskie-0.13.0/src/haskie/claude_code/skills/haskie/SKILL.md +165 -0
- haskie-0.13.0/src/haskie/cli.py +461 -0
- haskie-0.13.0/src/haskie/collection/__init__.py +1 -0
- haskie-0.13.0/src/haskie/collection/collection.py +516 -0
- haskie-0.13.0/src/haskie/collection/index.py +817 -0
- haskie-0.13.0/src/haskie/collection/maintenance.py +134 -0
- haskie-0.13.0/src/haskie/cpu.py +224 -0
- haskie-0.13.0/src/haskie/db.py +216 -0
- haskie-0.13.0/src/haskie/document/__init__.py +1 -0
- haskie-0.13.0/src/haskie/document/convert.py +192 -0
- haskie-0.13.0/src/haskie/document/document.py +589 -0
- haskie-0.13.0/src/haskie/document/render.py +124 -0
- haskie-0.13.0/src/haskie/errors.py +46 -0
- haskie-0.13.0/src/haskie/home.py +250 -0
- haskie-0.13.0/src/haskie/indexing/__init__.py +1 -0
- haskie-0.13.0/src/haskie/indexing/chunk.py +432 -0
- haskie-0.13.0/src/haskie/indexing/dbos_names.py +70 -0
- haskie-0.13.0/src/haskie/indexing/embed.py +304 -0
- haskie-0.13.0/src/haskie/indexing/embed_cache.py +281 -0
- haskie-0.13.0/src/haskie/indexing/gguf_models.py +118 -0
- haskie-0.13.0/src/haskie/indexing/hardware.py +71 -0
- haskie-0.13.0/src/haskie/indexing/mlx_models.py +254 -0
- haskie-0.13.0/src/haskie/indexing/models.py +311 -0
- haskie-0.13.0/src/haskie/indexing/onnx_rerank.py +107 -0
- haskie-0.13.0/src/haskie/indexing/operations.py +788 -0
- haskie-0.13.0/src/haskie/indexing/pipeline.py +271 -0
- haskie-0.13.0/src/haskie/indexing/segment.py +584 -0
- haskie-0.13.0/src/haskie/indexing/serializer.py +55 -0
- haskie-0.13.0/src/haskie/indexing/workflows.py +1518 -0
- haskie-0.13.0/src/haskie/logs.py +123 -0
- haskie-0.13.0/src/haskie/paging.py +212 -0
- haskie-0.13.0/src/haskie/search/__init__.py +41 -0
- haskie-0.13.0/src/haskie/search/aspects.py +277 -0
- haskie-0.13.0/src/haskie/search/collapse.py +486 -0
- haskie-0.13.0/src/haskie/search/fill.py +236 -0
- haskie-0.13.0/src/haskie/search/flow.py +497 -0
- haskie-0.13.0/src/haskie/search/passage.py +537 -0
- haskie-0.13.0/src/haskie/search/probe.py +115 -0
- haskie-0.13.0/src/haskie/search/retrieval.py +940 -0
- haskie-0.13.0/src/haskie/search/scoring.py +219 -0
- haskie-0.13.0/src/haskie/search/section.py +224 -0
- haskie-0.13.0/src/haskie/search/session.py +267 -0
- haskie-0.13.0/src/haskie/search/text.py +168 -0
- haskie-0.13.0/src/haskie/search/thin.py +180 -0
- haskie-0.13.0/src/haskie/settings.py +815 -0
- haskie-0.13.0/src/haskie/shutdown.py +134 -0
- haskie-0.13.0/src/haskie/sysdb.py +109 -0
- haskie-0.13.0/src/haskie/tables.py +206 -0
- haskie-0.13.0/src/haskie/web/assets/index-Cer6nXy-.css +1 -0
- haskie-0.13.0/src/haskie/web/assets/index-LWtpDfaD.js +14 -0
- haskie-0.13.0/src/haskie/web/favicon.svg +1 -0
- haskie-0.13.0/src/haskie/web/index.html +14 -0
haskie-0.13.0/PKG-INFO
ADDED
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: haskie
|
|
3
|
+
Version: 0.13.0
|
|
4
|
+
Summary: Personal document collections: markdown conversion, LanceDB search, web UI and MCP server
|
|
5
|
+
Requires-Dist: litestar[standard]>=2.23.0
|
|
6
|
+
Requires-Dist: litestar-mcp>=0.13.2
|
|
7
|
+
Requires-Dist: lancedb>=0.38.0
|
|
8
|
+
Requires-Dist: msgspec>=0.19.0
|
|
9
|
+
Requires-Dist: pydantic-graph>=2.48.0
|
|
10
|
+
Requires-Dist: firecrawl-anydoc>=0.2.4
|
|
11
|
+
Requires-Dist: pdf-inspector>=1.20.0
|
|
12
|
+
Requires-Dist: pyromark>=0.9.13
|
|
13
|
+
Requires-Dist: pypdf>=5.0.0
|
|
14
|
+
Requires-Dist: fastembed>=0.8.0
|
|
15
|
+
Requires-Dist: semantic-text-splitter>=0.32.0
|
|
16
|
+
Requires-Dist: aiosqlite>=0.21.0
|
|
17
|
+
Requires-Dist: sqlalchemy[asyncio]>=2.0.43
|
|
18
|
+
Requires-Dist: dbos>=3.0.0
|
|
19
|
+
Requires-Dist: structlog>=25.1.0
|
|
20
|
+
Requires-Dist: typer>=0.15.0
|
|
21
|
+
Requires-Dist: unicode-segmentation-rs>=0.3.3
|
|
22
|
+
Requires-Dist: numpy>=2.0
|
|
23
|
+
Requires-Dist: pebble>=5.2.2
|
|
24
|
+
Requires-Dist: snowballstemmer>=3.1.1
|
|
25
|
+
Requires-Dist: llama-cpp-python>=0.3.35 ; platform_machine == 'arm64' and sys_platform == 'darwin' and extra == 'gguf'
|
|
26
|
+
Requires-Dist: onnxruntime-gpu>=1.17.0 ; sys_platform != 'darwin' and extra == 'gpu'
|
|
27
|
+
Requires-Dist: mlx-embeddings>=0.1 ; platform_machine == 'arm64' and sys_platform == 'darwin' and extra == 'mlx'
|
|
28
|
+
Requires-Python: >=3.13
|
|
29
|
+
Provides-Extra: gguf
|
|
30
|
+
Provides-Extra: gpu
|
|
31
|
+
Provides-Extra: mlx
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
<p align="center">
|
|
35
|
+
<img src="web/public/favicon.svg" alt="" width="96">
|
|
36
|
+
</p>
|
|
37
|
+
|
|
38
|
+
<h1 align="center">haskie</h1>
|
|
39
|
+
|
|
40
|
+
<p align="center">
|
|
41
|
+
<strong>Haskie "has a key" to your private bookshelf, giving your AI agents your exact taste.</strong><br>
|
|
42
|
+
Your trusted sources, searchable by your agents, cited to the page, kept on your machine.
|
|
43
|
+
</p>
|
|
44
|
+
|
|
45
|
+
Your AI agent knows what everyone wrote. It does not know what you trust.
|
|
46
|
+
|
|
47
|
+
You picked the one book that settles the question, the standard that applies to your hardware,
|
|
48
|
+
the paper your team builds on. Your agent still answers from the average of the internet. haskie
|
|
49
|
+
hands it the key to your own shelf. Import your documents once, group them into collections, and
|
|
50
|
+
any Model Context Protocol (MCP) client can search them. Claude Code gets a rule that makes it
|
|
51
|
+
search them first. Every answer comes back as a short passage to quote, with the heading it sits
|
|
52
|
+
under and, for PDFs, the page.
|
|
53
|
+
|
|
54
|
+
It runs on your laptop. A few commands set it up, and a web UI handles the curating.
|
|
55
|
+
|
|
56
|
+
## Why this exists
|
|
57
|
+
|
|
58
|
+
**Your taste is what AI averages away.** Writers who use AI for ideas write stories rated more
|
|
59
|
+
creative, and "more similar to each other than stories by humans alone" [1]. Models trained on
|
|
60
|
+
model output lose "the tails of the original content distribution" [2]. The tails are where taste
|
|
61
|
+
lives: the niche book, the unpopular opinion that turned out right.
|
|
62
|
+
|
|
63
|
+
**Half the new web is written by AI.** About 50% of new English articles online are now mostly
|
|
64
|
+
AI-generated, a share that has held since early 2025 [3]. 74% of new web pages contain some
|
|
65
|
+
AI-written text [4]. NewsGuard tracks 3,749 AI content-farm news sites across 16 languages [5].
|
|
66
|
+
Your agent reads this text today, and tomorrow's models train on it [2].
|
|
67
|
+
|
|
68
|
+
**Web search hands your agent unverified claims.** It reads whatever ranks, sourced or not.
|
|
69
|
+
|
|
70
|
+
- Leading chatbots repeated false news claims 35% of the time in 2025, up from 18% a year before,
|
|
71
|
+
after they switched to live web search [6].
|
|
72
|
+
- AI search tools got more than 60% of source-citation queries wrong, and rarely signalled doubt
|
|
73
|
+
[7].
|
|
74
|
+
- A web page can hide instructions that take over the agent reading it, the top risk for large
|
|
75
|
+
language model (LLM) applications [8]. Five planted texts among millions steer a retrieval
|
|
76
|
+
system's answer 90% of the time [9].
|
|
77
|
+
|
|
78
|
+
**Developers do not trust the answers.** 46% distrust the accuracy of AI tools, and 66% name
|
|
79
|
+
answers that are "almost right, but not quite" as their top complaint [10].
|
|
80
|
+
|
|
81
|
+
**Pasting whole books does not work.** Models lose what sits in the middle of a long context [11],
|
|
82
|
+
and accuracy "consistently degrades with increasing input length" [12]. Good agent context is "the
|
|
83
|
+
smallest set of high-signal tokens" [13].
|
|
84
|
+
|
|
85
|
+
**Building it yourself is a project.** You need a parser, a chunker, embeddings, a vector store,
|
|
86
|
+
keyword search, a reranker, and a job queue that survives a closed laptop. Then you maintain all
|
|
87
|
+
of it.
|
|
88
|
+
|
|
89
|
+
haskie does not fact-check your documents. It makes sure your agent reads the ones you chose, and
|
|
90
|
+
shows where every answer came from.
|
|
91
|
+
|
|
92
|
+
## Mission
|
|
93
|
+
|
|
94
|
+
haskie is a fast, transparent, easy-to-manage library of the sources you trust. It steers your AI
|
|
95
|
+
agents with your taste instead of the internet's average. Over MCP it aims to give the agent
|
|
96
|
+
relevant evidence from several sources, with no repeats, and every piece says where it came from
|
|
97
|
+
so the agent can dig deeper. The agent keeps the reasoning. haskie makes the small retrieval
|
|
98
|
+
decisions, so the agent needs fewer round trips and fewer tokens.
|
|
99
|
+
|
|
100
|
+
## What that means in practice
|
|
101
|
+
|
|
102
|
+
- **Your taste, your sources.** Only what you import can answer. Group documents into collections
|
|
103
|
+
per topic: *coffee roasting*, *our architecture decisions*, *the standards for this board*. A
|
|
104
|
+
document imported once can sit in any number of them.
|
|
105
|
+
- **Transparent.** Every result carries its document, heading path and lines, and pages for PDFs.
|
|
106
|
+
Explore shows exactly what the agent receives, Operations every job, Sessions every search.
|
|
107
|
+
- **Relevant, without repeats.** Hybrid search matches meaning and exact terms, and an optional
|
|
108
|
+
reranker sharpens the order. A point several sources make comes back once, with the others
|
|
109
|
+
under `also_in`.
|
|
110
|
+
- **The agent decides, haskie does the legwork.** One `search_excerpts` call searches every
|
|
111
|
+
collection in scope, merges neighbouring hits and folds repeats. A question with several parts
|
|
112
|
+
goes in one call: each part gets its share of the slots, and each excerpt names the parts it
|
|
113
|
+
answers. `search_sources` names the
|
|
114
|
+
documents and collections that cover a topic. Each excerpt links to its full markdown file.
|
|
115
|
+
- **Local and polite to your machine.** Your documents never leave it. Only the models download,
|
|
116
|
+
once, from Hugging Face. Indexing runs in parallel within a CPU budget you set, and after a crash
|
|
117
|
+
the run resumes at the step it was on.
|
|
118
|
+
- **Sensible defaults, open to tuning.** The defaults are a small English embedding model, hybrid
|
|
119
|
+
search and 1,200-character chunks. Each collection can override the chunk and search settings.
|
|
120
|
+
|
|
121
|
+
## Status: early, and already useful
|
|
122
|
+
|
|
123
|
+
haskie is young, with much still to add, but it already covers the whole path from import to
|
|
124
|
+
cited answers in Claude Code. Not there yet:
|
|
125
|
+
|
|
126
|
+
- **More retrieval decisions made for the agent.** Today haskie merges neighbouring hits, grows
|
|
127
|
+
or drops passages too short to stand alone, folds repeats, groups passages by section, fills in
|
|
128
|
+
the text around and between them that answers too, and searches again for the words of a
|
|
129
|
+
question no excerpt holds. Next on the list, each one a round trip the agent would otherwise
|
|
130
|
+
spend:
|
|
131
|
+
- **Trimming** the sentences of a passage that do not answer. Today an excerpt keeps
|
|
132
|
+
every passage whole.
|
|
133
|
+
- **Cross-document merging**, so complementary passages from several documents arrive as one
|
|
134
|
+
answer with every source cited. Today only repeats are folded.
|
|
135
|
+
- **Distillation** of the results into a short, cited brief, for questions where the agent
|
|
136
|
+
needs the gist more than the quotes.
|
|
137
|
+
- **OCR.** Scanned pages and images are stored but not searchable.
|
|
138
|
+
- **Other MCP clients.** Any MCP client can use the tools over HTTP. Only Claude Code has a
|
|
139
|
+
one-command setup.
|
|
140
|
+
|
|
141
|
+
## Install
|
|
142
|
+
|
|
143
|
+
haskie needs [uv](https://docs.astral.sh/uv/), which fetches Python 3.13 or newer if you have none.
|
|
144
|
+
|
|
145
|
+
```sh
|
|
146
|
+
uv tool install haskie
|
|
147
|
+
haskie install claude # MCP server, skill, rule and SessionStart hook for Claude Code
|
|
148
|
+
haskie run # web UI, REST API and MCP on http://127.0.0.1:8451
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Open http://127.0.0.1:8451. The first screen asks for an embedding model and the search defaults.
|
|
152
|
+
The default model is bge-small (English, about 130 MB). Pick a multilingual one for other
|
|
153
|
+
languages, or none for keyword search only. The model applies to every collection, and changing
|
|
154
|
+
it later means running *Index all* in each one.
|
|
155
|
+
|
|
156
|
+
- **Extras:** install `"haskie[gpu]"` to run embeddings and
|
|
157
|
+
rerankers on CUDA. On Apple Silicon, `haskie[mlx]` adds the MLX rerankers and embedding models, and
|
|
158
|
+
`haskie[gguf]` the `-gguf` embedding profiles, which run on the GPU through llama.cpp (installing
|
|
159
|
+
it compiles llama.cpp, which needs the Xcode command-line tools and cmake). Other ONNX embeddings
|
|
160
|
+
run on the CPU there; the `coreml` hardware setting runs them through CoreML instead, which today
|
|
161
|
+
is slower. To install both Apple Silicon extras, drop the one you do not need:
|
|
162
|
+
|
|
163
|
+
```sh
|
|
164
|
+
uv tool install "haskie[mlx,gguf]"
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
- **Port:** 8451 by default, clear of the usual 8000 and 8080. For another, run
|
|
168
|
+
`haskie run --port <n>` and `haskie install claude --url http://127.0.0.1:<n>/mcp`, or set
|
|
169
|
+
`HASKIE_PORT`, which moves the default of `run`, `ensure` and `install claude` at once.
|
|
170
|
+
- **Other commands:** `haskie stop` stops the server. `haskie destroy` deletes `~/.haskie` after
|
|
171
|
+
showing what would be lost. `--home` or `HASKIE_HOME` keeps the data elsewhere.
|
|
172
|
+
|
|
173
|
+
## From files to answers
|
|
174
|
+
|
|
175
|
+
1. **Documents.** Drop files onto the page. haskie converts them to markdown (text formats are
|
|
176
|
+
read as they are) and embeds them in the background, with a side-by-side preview.
|
|
177
|
+
2. **Collections.** Create one per topic and add its documents. Give it a one-line description.
|
|
178
|
+
The agent reads it to choose where to look.
|
|
179
|
+
3. **Explore.** Search and see exactly what your agent gets: *Excerpts* and *Sources*. Switch to
|
|
180
|
+
*Chunks* or *Passages* to see how haskie cut the documents and built each answer.
|
|
181
|
+
|
|
182
|
+
**Operations** shows background jobs with their progress, and cancels running ones. **Sessions**
|
|
183
|
+
replays each agent conversation. **Insights** charts searches and indexed chunks over time.
|
|
184
|
+
**Settings** describes every default.
|
|
185
|
+
|
|
186
|
+
## How it works with Claude Code
|
|
187
|
+
|
|
188
|
+
`haskie install claude` adds four things:
|
|
189
|
+
|
|
190
|
+
| what | where | why |
|
|
191
|
+
| --- | --- | --- |
|
|
192
|
+
| MCP server entry | `claude mcp add --transport http` | gives Claude the tools |
|
|
193
|
+
| skill | `~/.claude/skills/haskie/SKILL.md` | how to use the tools and what to cite. Its trigger names your collections, so it fires on *coffee roasting*, not on the word "documents" |
|
|
194
|
+
| rule | `~/.claude/rules/haskie.md` | loads into every session, so Claude searches your collections first, even for a plain "what is X?" that never triggers a skill |
|
|
195
|
+
| SessionStart hook | `~/.claude/settings.json` | runs `haskie ensure`: starts the server if it is down, and passes the session id so Sessions can record it |
|
|
196
|
+
|
|
197
|
+
Run it again after adding a collection, to refresh the names. `--scope project` installs into
|
|
198
|
+
`./.claude` of the directory you run it from. The hook does not wait for the server, so a session
|
|
199
|
+
that starts while nothing is serving, such as the first after a reboot, has no haskie tools. Keep
|
|
200
|
+
`haskie run` open if that session matters.
|
|
201
|
+
|
|
202
|
+
A typical exchange: you ask *"How should a background job retry a failed HTTP call without
|
|
203
|
+
charging twice?"* The rule sends Claude to `search_excerpts` before the web. haskie returns
|
|
204
|
+
passages from your books, each with a `header` and `location`, such as
|
|
205
|
+
`Stream Processing > Idempotence` at `ddia.pdf p.478 L21904-21931` (lines count through the
|
|
206
|
+
whole markdown file). Claude answers and cites them. If nothing matches, it says so and goes to
|
|
207
|
+
the web.
|
|
208
|
+
|
|
209
|
+
## MCP tools
|
|
210
|
+
|
|
211
|
+
The endpoint is `http://127.0.0.1:8451/mcp`, over HTTP. Tools that search or change something take
|
|
212
|
+
a `session_id`, so Sessions can replay the conversation.
|
|
213
|
+
|
|
214
|
+
| tool | what it does |
|
|
215
|
+
| --- | --- |
|
|
216
|
+
| `search_excerpts` | **The main search.** Passages ready to quote, best first (in turns for several parts), each with `header` and `location`. Repeats fold into `also_in`. Takes up to 5 parts of one question, and tags each excerpt with the parts it answers |
|
|
217
|
+
| `search_sources` | Which documents and collections cover a topic. One row per document, with its best sections |
|
|
218
|
+
| `set_session_collections` | Limits the rest of the conversation to the collections `search_sources` suggested |
|
|
219
|
+
| `list_collections`, `get_collection`, `list_collection_documents` | Browse collections and their descriptions |
|
|
220
|
+
| `list_documents`, `get_document` | Browse documents |
|
|
221
|
+
| `add_document` | Import a local file by path |
|
|
222
|
+
| `add_document_to_collection`, `remove_document_from_collection` | Attach or detach a document |
|
|
223
|
+
| `describe_document` | Set what a document is about. `search_sources` shows it |
|
|
224
|
+
|
|
225
|
+
Every search looks in the `collections` argument, else the session's collections, else all of
|
|
226
|
+
them. Creating and deleting collections, and deleting documents, stay in the web UI.
|
|
227
|
+
[docs/mcp.md](docs/mcp.md) walks through a session, and the
|
|
228
|
+
[skill](src/haskie/claude_code/skills/haskie/SKILL.md) lists every argument and field.
|
|
229
|
+
|
|
230
|
+
## Under the hood
|
|
231
|
+
|
|
232
|
+
[docs/](docs/README.md) has a page with diagrams for each stage.
|
|
233
|
+
|
|
234
|
+
```
|
|
235
|
+
indexing: file ──► markdown ──► chunks ──────────► embeddings ──► LanceDB table
|
|
236
|
+
converted structure-aware cached once one per collection
|
|
237
|
+
|
|
238
|
+
search: query ──► hybrid search ──► rerank ────► passages ─────────► fold repeats ──► excerpts
|
|
239
|
+
vector + BM25 optional neighbours merged near-duplicates cited by heading,
|
|
240
|
+
fused by rank become also_in page and line
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
**Structure-Aware Chunking.** Chunks follow the author's structure. A chunk never spans two
|
|
244
|
+
sections. It cuts at a blank line before it cuts inside a paragraph, and between sentences before
|
|
245
|
+
it cuts inside one. A table or code block stays whole unless it is longer than a chunk. By default each
|
|
246
|
+
chunk is embedded and indexed with its heading path in front, such as
|
|
247
|
+
`Part II > Replication > Leaders`. Context added to chunks cuts failed retrievals by 35%, and by
|
|
248
|
+
67% with BM25 and a reranker on top [14]. There an LLM writes the context. haskie takes it from
|
|
249
|
+
the headings at no model call, and has not measured its own gain yet. Chunking by document
|
|
250
|
+
structure "largely improve[s]" retrieval-augmented generation (RAG) results [15]. Chunking by
|
|
251
|
+
embedding similarity does not justify its compute cost [16].
|
|
252
|
+
|
|
253
|
+
**LanceDB.** Each collection is one table on local disk. LanceDB is an embedded library with
|
|
254
|
+
vector and full-text (BM25) search in one table, on a columnar format built for fast random reads
|
|
255
|
+
[17]. So hybrid search needs no server.
|
|
256
|
+
|
|
257
|
+
**Hybrid search and reranking.** Vectors find meaning. BM25 finds exact terms, such as an error
|
|
258
|
+
code. haskie fuses both by rank (reciprocal rank fusion, RRF). An optional cross-encoder reads the
|
|
259
|
+
query and passage together and rescores the top candidates. Adding one takes the cut in failed
|
|
260
|
+
retrievals from 49% to 67% [14]. It is off by default. Settings offers models from 23 million
|
|
261
|
+
parameters up to multilingual ones.
|
|
262
|
+
|
|
263
|
+
**Repeats folded, passages whole.** Five books that make the same point would fill five of your
|
|
264
|
+
agent's slots. Most rerankers score one passage at a time, so they cannot see repeats [18]. haskie
|
|
265
|
+
merges hits on neighbouring chunks, then folds repeats with leader clustering. It walks the results
|
|
266
|
+
best first and compares each one only with the results already kept, by wording and, for models with
|
|
267
|
+
duplicate thresholds, by vector. The best result of each group keeps its place, so the ranking stays
|
|
268
|
+
intact, where diversity rerankers such as maximal marginal relevance (MMR) reorder it. Comparing
|
|
269
|
+
only with kept results stops chains, so A close to B and B close to C never merges A with C. The
|
|
270
|
+
same input always gives the same output. A repeat stays citable as an `also_in` entry (`duplicate`,
|
|
271
|
+
`contained` or `equivalent`), and its slot goes to the next distinct result. Repeated passages do not
|
|
272
|
+
significantly improve answer correctness, while different documents improve it by 17–47% [19].
|
|
273
|
+
|
|
274
|
+
**Async-first, with durable jobs.** Every IO is awaited, and CPU work runs in worker threads, so
|
|
275
|
+
search and the UI stay responsive while the machine indexes. Imports, indexing, deletes,
|
|
276
|
+
maintenance and model downloads run as [DBOS](https://docs.dbos.dev) workflows. DBOS records every
|
|
277
|
+
step in the same SQLite file, so after a crash a workflow will "resume from the last completed
|
|
278
|
+
step" [20]. It also deduplicates runs, cancels them and bounds the queues. Two collections with
|
|
279
|
+
the same chunk settings share one embedding run.
|
|
280
|
+
|
|
281
|
+
**Parallel within a budget.** The CPU budget (default: half your cores) caps concurrent tasks
|
|
282
|
+
across converting, embedding and indexing. Large PDFs split into batches of pages, so one big book
|
|
283
|
+
does not block the rest.
|
|
284
|
+
|
|
285
|
+
## Supported formats
|
|
286
|
+
|
|
287
|
+
| kind | extensions |
|
|
288
|
+
| --- | --- |
|
|
289
|
+
| PDF | `.pdf` (page by page, with page numbers kept for citations) |
|
|
290
|
+
| Office | `.doc` `.docx` `.docm` `.ppt` `.pptx` `.pptm` `.pps` `.ppsx` `.ppsm` `.pot` `.xls` `.xlsx` `.xlsm` `.xlsb` |
|
|
291
|
+
| OpenDocument | `.odt` `.ods` `.odp` |
|
|
292
|
+
| Other documents | `.epub` `.rtf` |
|
|
293
|
+
| Text | `.md` `.markdown` `.txt` `.csv` `.json` `.html` `.htm` |
|
|
294
|
+
| Images | `.png` `.jpg` `.jpeg` `.gif` `.webp` `.svg` (stored and previewed, but not searchable) |
|
|
295
|
+
|
|
296
|
+
haskie does no optical character recognition (OCR). It skips scanned PDF pages and indexes the
|
|
297
|
+
rest. A PDF with only scanned pages fails, with a clear message.
|
|
298
|
+
|
|
299
|
+
## Good to know
|
|
300
|
+
|
|
301
|
+
- **One user, one machine.** haskie has no login. It listens on `127.0.0.1` by default. Do not
|
|
302
|
+
expose it on a network.
|
|
303
|
+
- **Pre-1.0 storage.** A release that changes the storage format refuses to start on an older
|
|
304
|
+
home and says so. Run `haskie destroy`, import your documents again, and run
|
|
305
|
+
`haskie install claude` again.
|
|
306
|
+
- **One server per home.** A second `haskie run` on the same home refuses to start and names the
|
|
307
|
+
process that holds it.
|
|
308
|
+
|
|
309
|
+
## Contributing
|
|
310
|
+
|
|
311
|
+
[CONTRIBUTING.md](CONTRIBUTING.md) covers setup, the development commands and the technical
|
|
312
|
+
decisions. [docs/](docs/README.md) explains how each part works.
|
|
313
|
+
|
|
314
|
+
## References
|
|
315
|
+
|
|
316
|
+
1. Doshi, A. R. and Hauser, O. P. "Generative AI enhances individual creativity but reduces the
|
|
317
|
+
collective diversity of novel content." *Science Advances*, 2024.
|
|
318
|
+
https://doi.org/10.1126/sciadv.adn5290
|
|
319
|
+
2. Shumailov, I. et al. "AI models collapse when trained on recursively generated data."
|
|
320
|
+
*Nature*, 2024. https://doi.org/10.1038/s41586-024-07566-y
|
|
321
|
+
3. Paredes, J. L. et al. "AI Now Writes as Many Online Articles as Humans." Graphite, May 2026.
|
|
322
|
+
https://graphite.io/five-percent/ai-now-writes-as-many-online-articles-as-humans-do
|
|
323
|
+
4. Law, R. "74% of New Webpages Include AI Content (Study of 900k Pages)." Ahrefs, May 2025.
|
|
324
|
+
https://ahrefs.com/blog/what-percentage-of-new-content-is-ai-generated/
|
|
325
|
+
5. NewsGuard. "Tracking AI-enabled Misinformation." Updated June 2026.
|
|
326
|
+
https://www.newsguardtech.com/special-reports/ai-tracking-center/
|
|
327
|
+
6. NewsGuard. "AI False Information Rate Nearly Doubles in One Year." September 2025.
|
|
328
|
+
https://www.newsguardtech.com/ai-monitor/august-2025-ai-false-claim-monitor/
|
|
329
|
+
7. Jaźwińska, K. and Chandrasekar, A. "AI Search Has a Citation Problem." *Columbia Journalism
|
|
330
|
+
Review*, Tow Center, March 2025.
|
|
331
|
+
https://www.cjr.org/tow_center/we-compared-eight-ai-search-engines-theyre-all-bad-at-citing-news.php
|
|
332
|
+
8. OWASP. "LLM01:2025 Prompt Injection." *OWASP Top 10 for LLM Applications*, 2025.
|
|
333
|
+
https://genai.owasp.org/llmrisk/llm01-prompt-injection/
|
|
334
|
+
9. Zou, W. et al. "PoisonedRAG: Knowledge Corruption Attacks to Retrieval-Augmented Generation of
|
|
335
|
+
Large Language Models." *USENIX Security*, 2025. https://arxiv.org/abs/2402.07867
|
|
336
|
+
10. Stack Overflow. "2025 Developer Survey: AI." https://survey.stackoverflow.co/2025/ai
|
|
337
|
+
11. Liu, N. F. et al. "Lost in the Middle: How Language Models Use Long Contexts." *TACL*, 2024.
|
|
338
|
+
https://doi.org/10.1162/tacl_a_00638
|
|
339
|
+
12. Hong, K., Troynikov, A. and Huber, J. "Context Rot: How Increasing Input Tokens Impacts LLM
|
|
340
|
+
Performance." Chroma, July 2025. https://www.trychroma.com/research/context-rot
|
|
341
|
+
13. Anthropic. "Effective context engineering for AI agents." September 2025.
|
|
342
|
+
https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents
|
|
343
|
+
14. Anthropic. "Introducing Contextual Retrieval." September 2024.
|
|
344
|
+
https://www.anthropic.com/news/contextual-retrieval
|
|
345
|
+
15. Jimeno Yepes, A. et al. "Financial Report Chunking for Effective Retrieval Augmented
|
|
346
|
+
Generation." 2024. https://arxiv.org/abs/2402.05131
|
|
347
|
+
16. Qu, R., Tu, R. and Bao, F. "Is Semantic Chunking Worth the Computational Cost?" 2024.
|
|
348
|
+
https://arxiv.org/abs/2410.13070
|
|
349
|
+
17. Pace, W. et al. "Lance: Efficient Random Access in Columnar Storage through Adaptive
|
|
350
|
+
Structural Encodings." 2025. https://arxiv.org/abs/2504.15247
|
|
351
|
+
18. Schlatt, F. et al. "Set-Encoder: Permutation-Invariant Inter-Passage Attention for Listwise
|
|
352
|
+
Passage Re-Ranking with Cross-Encoders." *ECIR*, 2025. https://arxiv.org/abs/2404.06912
|
|
353
|
+
19. Ross, J. J. et al. "How retriever redundancy and diversity impact RAG effectiveness." 2026,
|
|
354
|
+
preprint. https://arxiv.org/abs/2608.13956
|
|
355
|
+
20. DBOS. "dbos-transact-py." https://github.com/dbos-inc/dbos-transact-py
|
|
356
|
+
|
|
357
|
+
## License
|
|
358
|
+
|
|
359
|
+
[MIT](LICENSE)
|