interactive-semantic-hunt 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interactive_semantic_hunt-0.1.0.dist-info/METADATA +316 -0
- interactive_semantic_hunt-0.1.0.dist-info/RECORD +54 -0
- interactive_semantic_hunt-0.1.0.dist-info/WHEEL +5 -0
- interactive_semantic_hunt-0.1.0.dist-info/entry_points.txt +4 -0
- interactive_semantic_hunt-0.1.0.dist-info/licenses/LICENSE +21 -0
- interactive_semantic_hunt-0.1.0.dist-info/top_level.txt +1 -0
- ish/__init__.py +6 -0
- ish/adapters/__init__.py +0 -0
- ish/adapters/embedder/__init__.py +0 -0
- ish/adapters/embedder/llama_cpp.py +50 -0
- ish/adapters/embedder/ollama.py +112 -0
- ish/adapters/embedder/prefixes.py +87 -0
- ish/adapters/embedder/sentence_transformer.py +47 -0
- ish/adapters/parser/__init__.py +0 -0
- ish/adapters/parser/limits.py +90 -0
- ish/adapters/parser/markup.py +93 -0
- ish/adapters/parser/plugins.py +135 -0
- ish/adapters/parser/python.py +127 -0
- ish/adapters/parser/structured.py +234 -0
- ish/adapters/parser/tree_sitter.py +249 -0
- ish/adapters/vcs/__init__.py +0 -0
- ish/adapters/vcs/git.py +90 -0
- ish/adapters/vector_store/__init__.py +0 -0
- ish/adapters/vector_store/federated.py +117 -0
- ish/adapters/vector_store/pure_python.py +157 -0
- ish/adapters/vector_store/sqlite.py +538 -0
- ish/application/__init__.py +0 -0
- ish/application/index.py +227 -0
- ish/application/ports/__init__.py +0 -0
- ish/application/ports/embedder.py +31 -0
- ish/application/ports/parser.py +39 -0
- ish/application/ports/vector_store.py +165 -0
- ish/application/preview.py +39 -0
- ish/application/scan.py +179 -0
- ish/application/search.py +367 -0
- ish/bootstrap.py +412 -0
- ish/domain/__init__.py +0 -0
- ish/domain/chunk.py +30 -0
- ish/interfaces/__init__.py +0 -0
- ish/interfaces/cli/__init__.py +0 -0
- ish/interfaces/cli/args.py +165 -0
- ish/interfaces/cli/complete.py +48 -0
- ish/interfaces/cli/log.py +141 -0
- ish/interfaces/cli/main.py +168 -0
- ish/interfaces/complete.py +115 -0
- ish/interfaces/format.py +50 -0
- ish/interfaces/mcp/__init__.py +0 -0
- ish/interfaces/mcp/protocol.py +171 -0
- ish/interfaces/mcp/server.py +384 -0
- ish/interfaces/python/__init__.py +0 -0
- ish/interfaces/python/api.py +195 -0
- ish/interfaces/tui/__init__.py +0 -0
- ish/interfaces/tui/app.py +393 -0
- ish/settings.py +387 -0
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: interactive-semantic-hunt
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Interactive semantic search for code, inspired by fzf
|
|
5
|
+
Author-email: David Kristiansen <david@kristiansen.tech>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/davetothek/interactive-semantic-hunt
|
|
8
|
+
Project-URL: Repository, https://github.com/davetothek/interactive-semantic-hunt
|
|
9
|
+
Project-URL: Issues, https://github.com/davetothek/interactive-semantic-hunt/issues
|
|
10
|
+
Keywords: search,semantic,embeddings,code,fzf,tui,mcp
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
18
|
+
Classifier: Topic :: Software Development
|
|
19
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
20
|
+
Classifier: Typing :: Typed
|
|
21
|
+
Requires-Python: >=3.12
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: numpy>=2.5.2
|
|
25
|
+
Requires-Dist: pyyaml>=6.0
|
|
26
|
+
Requires-Dist: textual>=0.70.0
|
|
27
|
+
Requires-Dist: tree-sitter>=0.26.0
|
|
28
|
+
Requires-Dist: tree-sitter-c>=0.24.2
|
|
29
|
+
Requires-Dist: tree-sitter-cpp>=0.23.4
|
|
30
|
+
Provides-Extra: llama
|
|
31
|
+
Requires-Dist: llama-cpp-python>=0.2.75; extra == "llama"
|
|
32
|
+
Requires-Dist: huggingface-hub>=0.23.0; extra == "llama"
|
|
33
|
+
Provides-Extra: st
|
|
34
|
+
Requires-Dist: sentence-transformers; extra == "st"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# ish — Interactive Semantic Hunt
|
|
38
|
+
|
|
39
|
+
Semantic search for code, inspired by `fzf`. Point `ish` at a directory. It parses
|
|
40
|
+
each source file into named chunks, embeds them with a local model, and ranks them
|
|
41
|
+
against your query. Everything runs on your machine.
|
|
42
|
+
|
|
43
|
+
Languages: Python, C and C++, Markdown, and AsciiDoc. Documentation is indexed
|
|
44
|
+
beside the code it describes, so one query searches both.
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
pip install interactive-semantic-hunt
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Nothing in that install compiles: the default backend reaches Ollama over
|
|
53
|
+
HTTP with the standard library. A backend that runs the model in this
|
|
54
|
+
process is an extra — `[llama]` for llama.cpp, `[st]` for
|
|
55
|
+
sentence-transformers.
|
|
56
|
+
|
|
57
|
+
To work on ish itself, clone it and run `uv sync`.
|
|
58
|
+
|
|
59
|
+
The default embedding backend is Ollama, which keeps the model resident so no
|
|
60
|
+
run pays a model load. Start it once and pull the embedding model:
|
|
61
|
+
|
|
62
|
+
```sh
|
|
63
|
+
ollama serve
|
|
64
|
+
ollama pull nomic-embed-text
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Set `OLLAMA_HOST` to reach a daemon elsewhere.
|
|
68
|
+
|
|
69
|
+
Two other backends need no daemon:
|
|
70
|
+
|
|
71
|
+
- `--embedder llama.cpp` downloads a GGUF model and loads it per run. Slower per
|
|
72
|
+
query, faster for a first index of a large tree. Needs the `[llama]` extra.
|
|
73
|
+
- `--embedder st` uses sentence-transformers. Needs the `[st]` extra.
|
|
74
|
+
|
|
75
|
+
## Use
|
|
76
|
+
|
|
77
|
+
List every chunk under a path:
|
|
78
|
+
|
|
79
|
+
```sh
|
|
80
|
+
ish src/
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
src/ish/domain/chunk.py:7-28 class Chunk
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Search for a query:
|
|
88
|
+
|
|
89
|
+
```sh
|
|
90
|
+
ish "parse a python file" src/
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
[0.71] src/ish/adapters/parser/python.py:14-31 method PythonParser.parse
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Run the interactive picker and open the selection in your editor. Type to
|
|
98
|
+
search, `up`/`down` or `ctrl+p`/`ctrl+n` to move, `enter` to choose, `escape`
|
|
99
|
+
to quit. Narrow without leaving the query line:
|
|
100
|
+
|
|
101
|
+
```text
|
|
102
|
+
state machine transitions every language
|
|
103
|
+
lang:cpp state machine transitions the implementation
|
|
104
|
+
lang:yaml under:/10.System/ state the tests that cover it
|
|
105
|
+
type:doc how do I configure this the prose, not the code
|
|
106
|
+
type:test,doc retry backoff the tests and what they document
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Press Tab to finish a filter word. `ty` becomes `type:`, `lang:cp` becomes
|
|
110
|
+
`lang:cpp`, and `under:/s` becomes `under:/src/`; a word with several answers
|
|
111
|
+
grows as far as they agree and names the rest. `ish-complete` does the work, so
|
|
112
|
+
any picker can call it.
|
|
113
|
+
|
|
114
|
+
`lang:`, `under:`, and `type:` work in the query line of every interface —
|
|
115
|
+
the command line, the picker, Neovim, and MCP. The words are taken out
|
|
116
|
+
before the query is embedded, so the model sees the question rather than
|
|
117
|
+
how it was narrowed.
|
|
118
|
+
|
|
119
|
+
A language may be named however it comes to mind. `c`, `c++`, `cxx`, `h`, and
|
|
120
|
+
`hpp` all mean `cpp`, because one parser reads them all; `adoc` means
|
|
121
|
+
`asciidoc`, `md` means `markdown`, `py` means `python`, and `yml` means `yaml`.
|
|
122
|
+
|
|
123
|
+
`type:` sorts every chunk into exactly one kind. A path decides before a
|
|
124
|
+
language does, so a YAML fixture counts as a test rather than as config:
|
|
125
|
+
|
|
126
|
+
| kind | what it holds |
|
|
127
|
+
|---|---|
|
|
128
|
+
| `code` | Python, C, C++, and anything a plugin parser adds |
|
|
129
|
+
| `doc` | Markdown and AsciiDoc |
|
|
130
|
+
| `test` | anything under `tests/`, `spec/`, `fixtures/`, plus `test_*` and `conftest.py` |
|
|
131
|
+
| `config` | YAML, JSON, and TOML outside a test path |
|
|
132
|
+
|
|
133
|
+
A repository that names its trees its own way can say so, in
|
|
134
|
+
`.ish/config.toml`:
|
|
135
|
+
|
|
136
|
+
```toml
|
|
137
|
+
type_patterns = [
|
|
138
|
+
"test:/[0-9.]*(Tests|Verification)/",
|
|
139
|
+
"doc:/[0-9.]*Specification/",
|
|
140
|
+
]
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
The first match wins; anything unmatched keeps the reading above.
|
|
144
|
+
|
|
145
|
+
A config beside a subtree adds to the one above it, so a tree settles only
|
|
146
|
+
what it names and inherits the rest. Searching inside an already-indexed tree
|
|
147
|
+
reads that tree's index and narrows the answers to the path, rather than
|
|
148
|
+
starting a second index of the same files.
|
|
149
|
+
|
|
150
|
+
```sh
|
|
151
|
+
nvim $(ish -i src/)
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
### Options
|
|
155
|
+
|
|
156
|
+
| Flag | Purpose |
|
|
157
|
+
|---|---|
|
|
158
|
+
| `-i`, `--interactive` | Run the TUI picker |
|
|
159
|
+
| `--embedder {llama.cpp,ollama,st}` | Select the embedding backend (default: ollama) |
|
|
160
|
+
| `-v`, `-vv` | Increase log detail |
|
|
161
|
+
| `--color {auto,always,never}` | Control log color |
|
|
162
|
+
| `--limit N` | Maximum search results |
|
|
163
|
+
| `--ignore DIR ...` | Directory names to skip (default `.git .venv venv __pycache__`) |
|
|
164
|
+
| `--include REGEX ...` | Index only paths matching these patterns |
|
|
165
|
+
| `--exclude REGEX ...` | Never index paths matching these patterns |
|
|
166
|
+
| `--git`, `--no-git` | Skip files git ignores (default: on) |
|
|
167
|
+
| `--lang LANG ...` | Show results only from these languages |
|
|
168
|
+
| `--under REGEX` | Show results only from matching paths |
|
|
169
|
+
| `--type TYPE ...` | Show results only of these kinds: `code`, `doc`, `test`, `config` |
|
|
170
|
+
| `--type-patterns TYPE:REGEX ...` | Say what a path holds, overriding the built-in reading |
|
|
171
|
+
| `--model NAME` | Override the backend model |
|
|
172
|
+
| `--refresh` | Bring every stored index at or below the path up to date first |
|
|
173
|
+
| `--reindex` | Discard the stored index and build it again |
|
|
174
|
+
| `--no-cache` | Index in memory only, leaving nothing on disk |
|
|
175
|
+
|
|
176
|
+
Logs go to stderr, so you can pipe stdout safely. A file that cannot be read
|
|
177
|
+
or parsed is counted in one line; `-v` names them.
|
|
178
|
+
|
|
179
|
+
## Use from Neovim
|
|
180
|
+
|
|
181
|
+
`contrib/nvim/ish.lua` is an fzf-lua picker. Copy it to `lua/utils/ish.lua` and
|
|
182
|
+
bind it:
|
|
183
|
+
|
|
184
|
+
```lua
|
|
185
|
+
map('n', '<leader>fi', function() require('utils.ish').search() end,
|
|
186
|
+
{ desc = 'Semantic search (ish)' })
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
It reads `--format grep`, so the built-in previewer opens each result at its
|
|
190
|
+
line, and prints the rank in the leftmost column. `search_lang({'cpp'})`,
|
|
191
|
+
`search_type({'doc'})`, and `search_here()` narrow it, as does a `lang:`,
|
|
192
|
+
`type:`, or `under:` word typed into the query.
|
|
193
|
+
|
|
194
|
+
While the index refreshes, `require('utils.ish').statusline()` renders a bar
|
|
195
|
+
for a statusline — `ish ███░░░░░ 38%` — and an empty string when idle. It reads `vim.g.ish_index_status`, which the picker keeps up to date, and
|
|
196
|
+
shows `ish ✓` briefly when a refresh finishes. Nothing is reported through
|
|
197
|
+
`vim.notify`: with `cmdheight = 0` there is no command line to put a message
|
|
198
|
+
in, so nvim draws one over the last screen row — the statusline itself.
|
|
199
|
+
|
|
200
|
+
The picker never blocks the editor: results are written as they arrive, so
|
|
201
|
+
typing stays smooth however long a search takes.
|
|
202
|
+
|
|
203
|
+
`contrib/nvim/ish_server.lua` keeps one `ish-mcp` process per session. It
|
|
204
|
+
starts on the first search and is reused after that, which cuts a keystroke
|
|
205
|
+
from about 500 ms to about 150 ms. Copy it beside the picker.
|
|
206
|
+
|
|
207
|
+
## Use from Python
|
|
208
|
+
|
|
209
|
+
```python
|
|
210
|
+
from ish.interfaces.python.api import Ish
|
|
211
|
+
|
|
212
|
+
with Ish("src/") as ish:
|
|
213
|
+
for chunk, score in ish.search("type:doc how to configure", limit=5):
|
|
214
|
+
print(score, chunk.path, chunk.symbol)
|
|
215
|
+
print(ish.status())
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
`Ish` holds the index open, so a second query costs a search rather than a
|
|
219
|
+
process start. It offers `search()`, `chunks()`, `index()`, `refresh_all()`,
|
|
220
|
+
and `status()`, and reads `lang:`, `type:`, and `under:` out of the query
|
|
221
|
+
exactly as the other interfaces do.
|
|
222
|
+
|
|
223
|
+
## Use from an agent
|
|
224
|
+
|
|
225
|
+
`ish-mcp` serves the same search over the Model Context Protocol, so an agent
|
|
226
|
+
can query the index directly. Add it to a project with `.mcp.json`:
|
|
227
|
+
|
|
228
|
+
```json
|
|
229
|
+
{
|
|
230
|
+
"mcpServers": {
|
|
231
|
+
"ish": { "command": "uv", "args": ["run", "ish-mcp"] }
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
It offers `search_code`, `list_chunks`, `index_status`, and `refresh_index`. The server stays
|
|
237
|
+
resident, so a query costs about 58 ms rather than a process start.
|
|
238
|
+
|
|
239
|
+
A call may narrow one search with `lang`, `under`, `type`, and `limit`, or
|
|
240
|
+
write the same filters into the query text. It cannot change
|
|
241
|
+
what is indexed — those settings come from `ish.toml` only, so no single call can
|
|
242
|
+
shrink an index that another call depends on.
|
|
243
|
+
|
|
244
|
+
## Index
|
|
245
|
+
|
|
246
|
+
The index persists in SQLite under `$XDG_DATA_HOME/ish/`, one file per scanned
|
|
247
|
+
tree. **A search of a parent reads the indexes below it and refreshes none** —
|
|
248
|
+
choosing one of them to write to would be wrong — so it warns and offers
|
|
249
|
+
`--refresh`, which visits each tree in turn. A repeated query reuses it, so only changed files are parsed and only new
|
|
250
|
+
text is embedded. A renamed file re-embeds nothing.
|
|
251
|
+
|
|
252
|
+
Each index records the tree it was built from, so searching a directory also
|
|
253
|
+
searches every index below it. Index the parts of a large project separately and
|
|
254
|
+
search the whole from its root:
|
|
255
|
+
|
|
256
|
+
```sh
|
|
257
|
+
ish "warm" project/docs # index one part
|
|
258
|
+
ish "warm" project/firmware # and another
|
|
259
|
+
ish "how is exposure set" project # searches both
|
|
260
|
+
```
|
|
261
|
+
|
|
262
|
+
Searching a parent never rewrites an index below it. Pass `--no-federate` to use
|
|
263
|
+
only the index of the exact path.
|
|
264
|
+
|
|
265
|
+
The index records where each chunk is — its path, line range, kind, and name —
|
|
266
|
+
together with the embedding vector. It does not store the source, so it is not a
|
|
267
|
+
second readable copy of your code. Previews are read from the file, which also
|
|
268
|
+
means they always show the current content.
|
|
269
|
+
|
|
270
|
+
## Configure
|
|
271
|
+
|
|
272
|
+
Every command-line option is also a key in `ish.toml`, under the same name.
|
|
273
|
+
Put project settings in `ish.toml` at the root of your repository:
|
|
274
|
+
|
|
275
|
+
```toml
|
|
276
|
+
embedder = "ollama"
|
|
277
|
+
model = "mxbai-embed-large"
|
|
278
|
+
limit = 10
|
|
279
|
+
ignore = [".git", ".venv", "build", "node_modules"]
|
|
280
|
+
|
|
281
|
+
# Regular expressions, searched against the path.
|
|
282
|
+
exclude = ["/vendor/", "_pb2\\.py$", "(_test|_spec)\\.py$"]
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
`include` and `exclude` take regular expressions rather than globs, so `/vendor/`
|
|
286
|
+
matches at any depth and alternation works. `exclude` wins over `include`.
|
|
287
|
+
|
|
288
|
+
`--git` is on by default, so anything a `.gitignore` covers stays out of the
|
|
289
|
+
index. Pass `--no-git` to index it anyway.
|
|
290
|
+
|
|
291
|
+
`--lang` and `--under` narrow what a search *returns*. They never change what is
|
|
292
|
+
indexed, so a narrowed query cannot shrink the index:
|
|
293
|
+
|
|
294
|
+
```sh
|
|
295
|
+
ish "how is ranking done" --lang python
|
|
296
|
+
ish "installation steps" --lang markdown asciidoc
|
|
297
|
+
ish "parse a header" --under '/include/'
|
|
298
|
+
```
|
|
299
|
+
|
|
300
|
+
User-level defaults go in `~/.config/ish/ish.toml`. Later sources win:
|
|
301
|
+
|
|
302
|
+
```
|
|
303
|
+
defaults < ~/.config/ish/ish.toml < ./ish.toml < ISH_* environment < command line
|
|
304
|
+
```
|
|
305
|
+
|
|
306
|
+
Set any option from the environment with the `ISH_` prefix, for example
|
|
307
|
+
`ISH_LIMIT=20` or `ISH_IGNORE=build,dist`.
|
|
308
|
+
|
|
309
|
+
## Develop
|
|
310
|
+
|
|
311
|
+
```sh
|
|
312
|
+
uv run poe check # lint, typecheck, test
|
|
313
|
+
```
|
|
314
|
+
|
|
315
|
+
The architecture is ports and adapters. `spec.md` holds the requirements, and
|
|
316
|
+
`.claude/CLAUDE.md` describes the layers and the composition root.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
interactive_semantic_hunt-0.1.0.dist-info/licenses/LICENSE,sha256=_tIgsjXEpm2J-zkIN9m7SkL-K2LkgHWfBcS_Y7aFzqY,1074
|
|
2
|
+
ish/__init__.py,sha256=W9wJYhAv0PwnXHkhf6caJfYO5OXH73Y-g_PoehyPLWU,294
|
|
3
|
+
ish/bootstrap.py,sha256=4IiaxP6UaRhSts30Nu4jWWpIX69-pKce0xFMJbWEJhw,13770
|
|
4
|
+
ish/settings.py,sha256=2LhHGDWyhZXTbNq-MlW1sClJjEz5a0Fxs-W1qWHaWmM,12521
|
|
5
|
+
ish/adapters/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
ish/adapters/embedder/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
ish/adapters/embedder/llama_cpp.py,sha256=_GnRmYcTiBZ-reMFFMnRuSvWdhIMJTUb33pOXTD224I,1746
|
|
8
|
+
ish/adapters/embedder/ollama.py,sha256=cnLcTgR6mkNf5I4LLcVxXqkpPcBn3gLtBL70SB7BxyM,4209
|
|
9
|
+
ish/adapters/embedder/prefixes.py,sha256=M7ThGfjDL8sV_eVZPiCmXR5IR5CT8d9iT8bEOvm3vKk,3076
|
|
10
|
+
ish/adapters/embedder/sentence_transformer.py,sha256=z9z5CiF1WRZ-_bNVYXBm2jWp58nxw0FbuTd4KYejFKM,1796
|
|
11
|
+
ish/adapters/parser/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
12
|
+
ish/adapters/parser/limits.py,sha256=Vgv9gzsZRpJMcORKgNeJJND6kAfLSN79xzI-6_GQb3g,3161
|
|
13
|
+
ish/adapters/parser/markup.py,sha256=QCwK5ExyqXTkQkmuEfJ3h-7j9OG1mJE4vuuyiZlzQdU,3312
|
|
14
|
+
ish/adapters/parser/plugins.py,sha256=8Ob6J6kPHtp5qVEU9a2Q7AqoIg8ZWvR8EIPNkk9uOZI,4166
|
|
15
|
+
ish/adapters/parser/python.py,sha256=2Oguh249O1EHeuk0Gq1PthSqDFnHFf48xBQ_DzVbU2s,4293
|
|
16
|
+
ish/adapters/parser/structured.py,sha256=HyuKUCVdaz2sJJNWXCcLy1FWob_KywLOJ5pUh4qytRg,8124
|
|
17
|
+
ish/adapters/parser/tree_sitter.py,sha256=KHnWKdMnNke2Y4l3MyeJJDpImtOdZVoyxGFmNUPDjOY,8673
|
|
18
|
+
ish/adapters/vcs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
19
|
+
ish/adapters/vcs/git.py,sha256=nzXiLJHwHTYDaHoAjsEFpNYua-CdRE_MqnEkA7FEW9k,2986
|
|
20
|
+
ish/adapters/vector_store/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
21
|
+
ish/adapters/vector_store/federated.py,sha256=ZmVDXE3LLe225uGNuYCqe-2Drfd-Ef-w5C_0yV-TWzs,4328
|
|
22
|
+
ish/adapters/vector_store/pure_python.py,sha256=fZBmbFlb7zLAEI5wpAHDt3wh9wDpHZBoe4x4ojyt0So,5672
|
|
23
|
+
ish/adapters/vector_store/sqlite.py,sha256=VrnC-Q5rKL0J7Prd7JqVoaOtA89jvhUsLFgpo4mjzzw,20171
|
|
24
|
+
ish/application/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
25
|
+
ish/application/index.py,sha256=AyE8ka8SLBlpS6HPhzvZqp4by2k-lzEUTSB3bFYwFwo,8022
|
|
26
|
+
ish/application/preview.py,sha256=sSIW4vfqpZW6b4Q9CSqdrsl7qyVRqshqFeKQwStLgtU,1320
|
|
27
|
+
ish/application/scan.py,sha256=WVvkownLTP3jBt1VsAaVzw5N3tUxtKDg0zghs6ATwnI,6589
|
|
28
|
+
ish/application/search.py,sha256=0VpiJwUUfVvTWeODy5kXSX5tSEuT15pKdQnqejZnqTM,12265
|
|
29
|
+
ish/application/ports/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
30
|
+
ish/application/ports/embedder.py,sha256=ZBiQ-8gz31XcPRRKUvbICSrTo2Ya8xecogKhimsSRWs,987
|
|
31
|
+
ish/application/ports/parser.py,sha256=KjCaB0YDdKna0O9mdab5B5c5MvntWJMTmzWxXiamapg,1059
|
|
32
|
+
ish/application/ports/vector_store.py,sha256=GlbQc_b0ZEwJaKGegzkDIK7G224BKb3uu-CLemMEFAI,5494
|
|
33
|
+
ish/domain/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
34
|
+
ish/domain/chunk.py,sha256=O5xo3KAaumkScYgGVtX7EAv1oICGKLb1eyjmBAbo66c,1058
|
|
35
|
+
ish/interfaces/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
36
|
+
ish/interfaces/complete.py,sha256=77f0ZK7ozvEFvST6rQtF4n8YqVO58IueQouSjh69_mg,3947
|
|
37
|
+
ish/interfaces/format.py,sha256=EupJWKg65lSAHqvfB6hu-TzlX9Qu5IP9DYfyVIpg_2U,1584
|
|
38
|
+
ish/interfaces/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
39
|
+
ish/interfaces/cli/args.py,sha256=3m_aD8BlmG4steJInPxdkpxeHDwtj1OW6T_ua9HjUOU,5406
|
|
40
|
+
ish/interfaces/cli/complete.py,sha256=4STpIDg17JykuJQvYLDSVYOFM6iyTSP7R3IzrjV1l64,1663
|
|
41
|
+
ish/interfaces/cli/log.py,sha256=UCLJslCcQa-zbBwnYde5KtjmHRZEbtJFzLgEbHSy9j0,3988
|
|
42
|
+
ish/interfaces/cli/main.py,sha256=TCYnKgEmnjc-3xCsYUpX5tR5__GP2KS-VGikPzKvx9A,5188
|
|
43
|
+
ish/interfaces/mcp/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
44
|
+
ish/interfaces/mcp/protocol.py,sha256=b8iPjd7u78HR6ss066l2Iew85chjWa0LORwHBLdCpmM,6042
|
|
45
|
+
ish/interfaces/mcp/server.py,sha256=Td-2FA7ZknXcw_-_-xGkX56bihJclbAHhVGEU_qopRg,14628
|
|
46
|
+
ish/interfaces/python/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
47
|
+
ish/interfaces/python/api.py,sha256=PA69SyJ6N2k6ZBVOwzCneCu24D9lSDp69CKElqmOSe4,6687
|
|
48
|
+
ish/interfaces/tui/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
49
|
+
ish/interfaces/tui/app.py,sha256=aFwwbxFR4iku8H3UxecxTowL_nWrLz51dHbgthcD5VY,14840
|
|
50
|
+
interactive_semantic_hunt-0.1.0.dist-info/METADATA,sha256=M9JZEOFrQJgQ2-28tksaswATOwuvc--jJ6sqGlsOPAc,11636
|
|
51
|
+
interactive_semantic_hunt-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
52
|
+
interactive_semantic_hunt-0.1.0.dist-info/entry_points.txt,sha256=Gw9kcf_Ijny8kgKQ1gDZTdtKrTkp3P1ahGkH7_qRALA,142
|
|
53
|
+
interactive_semantic_hunt-0.1.0.dist-info/top_level.txt,sha256=iEzghqES9lDeQgVwT6pm7BwRfGDSnYFvuQ6jmqbyM50,4
|
|
54
|
+
interactive_semantic_hunt-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 David Kristiansen
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ish
|
ish/__init__.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""ish — Interactive Semantic Hunt. A semantic search tool for code.
|
|
2
|
+
|
|
3
|
+
Keep this module free of layer imports. The package root must not pull
|
|
4
|
+
in interfaces, application, or adapter code — import from the layer
|
|
5
|
+
modules directly, for example ``from ish.interfaces.python.api import Ish``.
|
|
6
|
+
"""
|
ish/adapters/__init__.py
ADDED
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""llama.cpp adapter for the Embedder protocol."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Sequence
|
|
4
|
+
|
|
5
|
+
from ish.adapters.embedder.prefixes import PrefixingEmbedder
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class LlamaCppEmbedder(PrefixingEmbedder):
|
|
9
|
+
"""Generate embeddings using a local GGUF model via llama.cpp.
|
|
10
|
+
|
|
11
|
+
Automatically downloads `nomic-embed-text` from Hugging Face if not present.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
repo_id: str = "nomic-ai/nomic-embed-text-v1.5-GGUF",
|
|
17
|
+
filename: str = "nomic-embed-text-v1.5.Q4_K_M.gguf",
|
|
18
|
+
) -> None:
|
|
19
|
+
# Expose the model identity for cache keying.
|
|
20
|
+
self.model_name = f"{repo_id}/{filename}"
|
|
21
|
+
|
|
22
|
+
import os
|
|
23
|
+
|
|
24
|
+
from huggingface_hub.utils.logging import set_verbosity_error
|
|
25
|
+
|
|
26
|
+
# Suppress huggingface_hub network warnings
|
|
27
|
+
os.environ["HF_HUB_DISABLE_TELEMETRY"] = "1"
|
|
28
|
+
os.environ["HF_HUB_DISABLE_SYMLINKS_WARNING"] = "1"
|
|
29
|
+
set_verbosity_error()
|
|
30
|
+
|
|
31
|
+
from huggingface_hub import hf_hub_download
|
|
32
|
+
from llama_cpp import Llama
|
|
33
|
+
|
|
34
|
+
# 1. Download/find the model on disk
|
|
35
|
+
model_path = hf_hub_download(repo_id=repo_id, filename=filename)
|
|
36
|
+
|
|
37
|
+
# 2. Instantiate the engine. verbose=False hides the massive C++ startup logs.
|
|
38
|
+
self._model = Llama(model_path=model_path, embedding=True, verbose=False)
|
|
39
|
+
|
|
40
|
+
def _embed(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
|
|
41
|
+
"""Encode texts into vectors via llama.cpp."""
|
|
42
|
+
text_list = list(texts)
|
|
43
|
+
|
|
44
|
+
# create_embedding accepts a single string or a list of strings
|
|
45
|
+
result = self._model.create_embedding(text_list)
|
|
46
|
+
|
|
47
|
+
from typing import cast
|
|
48
|
+
|
|
49
|
+
embeddings = [item["embedding"] for item in result["data"]]
|
|
50
|
+
return cast("Sequence[Sequence[float]]", embeddings)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Ollama adapter for the Embedder protocol.
|
|
2
|
+
|
|
3
|
+
Call the local Ollama daemon over HTTP with the standard library. The
|
|
4
|
+
daemon holds the model resident, so no process pays a model load, and
|
|
5
|
+
the adapter needs no third-party package.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import logging
|
|
10
|
+
import os
|
|
11
|
+
import urllib.error
|
|
12
|
+
import urllib.request
|
|
13
|
+
from collections.abc import Sequence
|
|
14
|
+
|
|
15
|
+
from ish.adapters.embedder.prefixes import PrefixingEmbedder
|
|
16
|
+
|
|
17
|
+
log = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
DEFAULT_HOST = "http://localhost:11434"
|
|
20
|
+
DEFAULT_MODEL = "nomic-embed-text"
|
|
21
|
+
|
|
22
|
+
# Send this many texts per request. One request for a whole repository
|
|
23
|
+
# would hold the daemon for minutes and risk the timeout.
|
|
24
|
+
DEFAULT_BATCH_SIZE = 64
|
|
25
|
+
# A batch of large definitions can take minutes on a busy daemon.
|
|
26
|
+
TIMEOUT_SECONDS = 600
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _normalize_host(host: str) -> str:
|
|
30
|
+
"""Accept a bare ``host:port`` as well as a full URL."""
|
|
31
|
+
host = host.rstrip("/")
|
|
32
|
+
if not host.startswith(("http://", "https://")):
|
|
33
|
+
return f"http://{host}"
|
|
34
|
+
return host
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class OllamaEmbedder(PrefixingEmbedder):
|
|
38
|
+
"""Generate embeddings through a running Ollama daemon.
|
|
39
|
+
|
|
40
|
+
Read ``OLLAMA_HOST`` when no host is given, matching the Ollama
|
|
41
|
+
command-line tools.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
def __init__(
|
|
45
|
+
self,
|
|
46
|
+
model_name: str = DEFAULT_MODEL,
|
|
47
|
+
*,
|
|
48
|
+
host: str | None = None,
|
|
49
|
+
batch_size: int = DEFAULT_BATCH_SIZE,
|
|
50
|
+
) -> None:
|
|
51
|
+
self.model_name = model_name
|
|
52
|
+
chosen = host or os.environ.get("OLLAMA_HOST") or DEFAULT_HOST
|
|
53
|
+
self.host = _normalize_host(chosen)
|
|
54
|
+
self._batch_size = max(1, batch_size)
|
|
55
|
+
|
|
56
|
+
def _embed(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
|
|
57
|
+
"""Encode texts into vectors, one batch of requests at a time."""
|
|
58
|
+
items = list(texts)
|
|
59
|
+
vectors: list[Sequence[float]] = []
|
|
60
|
+
for start in range(0, len(items), self._batch_size):
|
|
61
|
+
vectors.extend(self._embed_batch(items[start : start + self._batch_size]))
|
|
62
|
+
return vectors
|
|
63
|
+
|
|
64
|
+
# ------------------------------------------------------------------
|
|
65
|
+
# Internal helpers
|
|
66
|
+
# ------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
def _embed_batch(self, batch: list[str]) -> list[Sequence[float]]:
|
|
69
|
+
"""Send one batch and return its vectors."""
|
|
70
|
+
payload = json.dumps({"model": self.model_name, "input": batch}).encode("utf-8")
|
|
71
|
+
request = urllib.request.Request(
|
|
72
|
+
f"{self.host}/api/embed",
|
|
73
|
+
data=payload,
|
|
74
|
+
headers={"Content-Type": "application/json"},
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
try:
|
|
78
|
+
with urllib.request.urlopen(request, timeout=TIMEOUT_SECONDS) as response:
|
|
79
|
+
body = json.loads(response.read())
|
|
80
|
+
except urllib.error.HTTPError as exc:
|
|
81
|
+
detail = exc.read().decode("utf-8", "replace").strip()
|
|
82
|
+
if exc.code == 404:
|
|
83
|
+
hint = f"Pull it with 'ollama pull {self.model_name}'."
|
|
84
|
+
else:
|
|
85
|
+
hint = (
|
|
86
|
+
f"Confirm that {self.model_name!r} is an embedding model, "
|
|
87
|
+
f"not a generation model."
|
|
88
|
+
)
|
|
89
|
+
raise RuntimeError(
|
|
90
|
+
f"Ollama refused the request ({exc.code}): {detail}. {hint}"
|
|
91
|
+
) from exc
|
|
92
|
+
except urllib.error.URLError as exc:
|
|
93
|
+
raise RuntimeError(
|
|
94
|
+
f"Cannot reach Ollama at {self.host}: {exc.reason}. "
|
|
95
|
+
f"Start it with 'ollama serve', or install another backend "
|
|
96
|
+
f"and select it, as in "
|
|
97
|
+
f"'pip install interactive-semantic-hunt[llama]' "
|
|
98
|
+
f"then '--embedder llama.cpp'."
|
|
99
|
+
) from exc
|
|
100
|
+
except json.JSONDecodeError as exc:
|
|
101
|
+
raise RuntimeError(
|
|
102
|
+
f"Ollama at {self.host} returned a reply that is not JSON."
|
|
103
|
+
) from exc
|
|
104
|
+
|
|
105
|
+
embeddings = body.get("embeddings")
|
|
106
|
+
if not isinstance(embeddings, list) or len(embeddings) != len(batch):
|
|
107
|
+
got = len(embeddings) if isinstance(embeddings, list) else 0
|
|
108
|
+
raise RuntimeError(
|
|
109
|
+
f"Ollama returned {got} vectors for {len(batch)} texts. "
|
|
110
|
+
f"Confirm that {self.model_name!r} is an embedding model."
|
|
111
|
+
)
|
|
112
|
+
return embeddings
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Apply the task prefixes an embedding model was trained with.
|
|
2
|
+
|
|
3
|
+
Several retrieval models expect the caller to say whether a text is a
|
|
4
|
+
stored document or a search query, and they lose accuracy without it.
|
|
5
|
+
The convention belongs to the model, not to the backend serving it, so
|
|
6
|
+
both the llama.cpp and the Ollama adapter read the same table.
|
|
7
|
+
|
|
8
|
+
Measured on this repository with nomic-embed-text, 16 queries: the
|
|
9
|
+
prefixes moved top-1 accuracy from 62% to 75%.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from collections import OrderedDict
|
|
13
|
+
from collections.abc import Sequence
|
|
14
|
+
|
|
15
|
+
# Model name prefix -> (document prefix, query prefix).
|
|
16
|
+
# Match on the start of the name, so a tag such as ":latest" still hits.
|
|
17
|
+
_CONVENTIONS: dict[str, tuple[str, str]] = {
|
|
18
|
+
"nomic-embed-text": ("search_document: ", "search_query: "),
|
|
19
|
+
"nomic-ai/nomic-embed-text": ("search_document: ", "search_query: "),
|
|
20
|
+
"mxbai-embed-large": (
|
|
21
|
+
"",
|
|
22
|
+
"Represent this sentence for searching relevant passages: ",
|
|
23
|
+
),
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def prefixes_for(model_name: str) -> tuple[str, str]:
|
|
28
|
+
"""Return the (document, query) prefixes for *model_name*.
|
|
29
|
+
|
|
30
|
+
Return empty strings for a model with no known convention, which
|
|
31
|
+
leaves the text untouched.
|
|
32
|
+
"""
|
|
33
|
+
name = model_name.lower()
|
|
34
|
+
for known, pair in _CONVENTIONS.items():
|
|
35
|
+
if known in name:
|
|
36
|
+
return pair
|
|
37
|
+
return ("", "")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# How many recent queries to keep. Typing walks over the same text as
|
|
41
|
+
# characters are added and removed, so a small cache turns a backspace
|
|
42
|
+
# from a request into a lookup.
|
|
43
|
+
QUERY_CACHE_SIZE = 128
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class PrefixingEmbedder:
|
|
47
|
+
"""Add task prefixes, then hand the text to the concrete backend.
|
|
48
|
+
|
|
49
|
+
Subclasses set ``model_name`` and implement ``_embed``.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
model_name: str
|
|
53
|
+
|
|
54
|
+
def embed_documents(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
|
|
55
|
+
"""Embed texts that will be stored and searched over."""
|
|
56
|
+
if not texts:
|
|
57
|
+
return []
|
|
58
|
+
prefix, _ = prefixes_for(self.model_name)
|
|
59
|
+
prepared = [f"{prefix}{text}" for text in texts] if prefix else list(texts)
|
|
60
|
+
return self._embed(prepared)
|
|
61
|
+
|
|
62
|
+
def embed_query(self, text: str) -> Sequence[float]:
|
|
63
|
+
"""Embed one search query, reusing a recent answer.
|
|
64
|
+
|
|
65
|
+
An interactive search embeds a query on every keystroke, and
|
|
66
|
+
deleting a character asks for text already seen.
|
|
67
|
+
"""
|
|
68
|
+
cache = getattr(self, "_query_cache", None)
|
|
69
|
+
if cache is None:
|
|
70
|
+
cache = self._query_cache = OrderedDict()
|
|
71
|
+
held = cache.get(text)
|
|
72
|
+
if held is not None:
|
|
73
|
+
cache.move_to_end(text)
|
|
74
|
+
return held
|
|
75
|
+
|
|
76
|
+
_, prefix = prefixes_for(self.model_name)
|
|
77
|
+
vectors = self._embed([f"{prefix}{text}" if prefix else text])
|
|
78
|
+
vector = vectors[0] if vectors else []
|
|
79
|
+
|
|
80
|
+
cache[text] = vector
|
|
81
|
+
if len(cache) > QUERY_CACHE_SIZE:
|
|
82
|
+
cache.popitem(last=False)
|
|
83
|
+
return vector
|
|
84
|
+
|
|
85
|
+
def _embed(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
|
|
86
|
+
"""Encode already-prepared texts. Implemented by each adapter."""
|
|
87
|
+
raise NotImplementedError
|