dirsql-plugin-embeddings 0.1.7__tar.gz → 0.1.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/PKG-INFO +13 -5
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/README.md +12 -4
- dirsql_plugin_embeddings-0.1.7/e2e-attestations/claude-700-route-pdf.json → dirsql_plugin_embeddings-0.1.8/e2e-attestations/claude-706-widen-glob.json +3 -3
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/dirsql.toml +7 -1
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/.gitignore +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/pyproject.toml +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/__init__.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/cache.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/config.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/embedder.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/on_file/__init__.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/on_file/build_rows.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/on_file/on_file.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/on_file/read_content.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/on_file/read_pdf.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/on_file/read_text.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/src/dirsql_plugin_embeddings/pre_query.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/testing-conventions.toml +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/tests/conftest.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/tests/e2e/__init__.py +0 -0
- {dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/tests/integration/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dirsql-plugin-embeddings
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.8
|
|
4
4
|
Summary: First-party dirsql plugin: semantic search via an OpenAI-compatible embeddings endpoint.
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Requires-Dist: cachetta>=0.6.15
|
|
@@ -10,7 +10,8 @@ Description-Content-Type: text/markdown
|
|
|
10
10
|
# dirsql-plugin-embeddings
|
|
11
11
|
|
|
12
12
|
A first-party [`dirsql`](https://github.com/thekevinscott/dirsql) plugin that
|
|
13
|
-
adds **semantic search** over a directory of
|
|
13
|
+
adds **semantic search** over a directory of documents -- Markdown, plain
|
|
14
|
+
text, reStructuredText and PDFs. It is the worked
|
|
14
15
|
implementation behind the [Search documents by
|
|
15
16
|
meaning](https://thekevinscott.github.io/dirsql/howto/search-by-meaning) how-to,
|
|
16
17
|
swapping that guide's local `model2vec` model for any OpenAI-compatible
|
|
@@ -30,19 +31,26 @@ is installed alongside it. The fragment declares:
|
|
|
30
31
|
|
|
31
32
|
- the [`sqlite-vec`](https://github.com/asg017/sqlite-vec) extension, for
|
|
32
33
|
`vec_distance_cosine()`;
|
|
33
|
-
- a `documents` table whose `on-file` hook embeds each
|
|
34
|
-
a TEXT `embedding` column;
|
|
34
|
+
- a `documents` table whose `on-file` hook embeds each
|
|
35
|
+
`**/*.{md,markdown,mdx,rst,txt,pdf}` file into a TEXT `embedding` column;
|
|
35
36
|
- a `pre-query` hook that embeds the incoming question and emits the
|
|
36
37
|
nearest-neighbor SQL.
|
|
37
38
|
|
|
38
39
|
Both hooks are console scripts that call the same embedder.
|
|
39
40
|
|
|
40
|
-
|
|
41
|
+
Every matched extension except `.pdf` is read as UTF-8 text; a `.pdf` is read with
|
|
41
42
|
[pypdf](https://pypdf.readthedocs.io), whose per-page extracted text is joined
|
|
42
43
|
and embedded like any other document. The extension check is case-insensitive
|
|
43
44
|
(`.PDF` is a PDF), though the glob above is not — globset matches case-sensitively,
|
|
44
45
|
so an uppercase-suffixed file needs its own `glob` entry to be picked up at all.
|
|
45
46
|
|
|
47
|
+
The glob is an allowlist rather than `**/*` for a reason worth knowing: a hook
|
|
48
|
+
that exits non-zero aborts the entire scan
|
|
49
|
+
([dirsql#697](https://github.com/thekevinscott/dirsql/issues/697)), and reading a
|
|
50
|
+
PNG as UTF-8 does exactly that. Matching everything would mean one image anywhere
|
|
51
|
+
under the root produces no index at all, so the list stays limited to what the
|
|
52
|
+
plugin can actually read.
|
|
53
|
+
|
|
46
54
|
A PDF that cannot be parsed aborts the scan rather than being skipped: the hook
|
|
47
55
|
exits non-zero and dirsql reports the path and the pypdf reason. A *scanned*,
|
|
48
56
|
image-only PDF is not a failure — pypdf yields no text, and the file is indexed
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
# dirsql-plugin-embeddings
|
|
2
2
|
|
|
3
3
|
A first-party [`dirsql`](https://github.com/thekevinscott/dirsql) plugin that
|
|
4
|
-
adds **semantic search** over a directory of
|
|
4
|
+
adds **semantic search** over a directory of documents -- Markdown, plain
|
|
5
|
+
text, reStructuredText and PDFs. It is the worked
|
|
5
6
|
implementation behind the [Search documents by
|
|
6
7
|
meaning](https://thekevinscott.github.io/dirsql/howto/search-by-meaning) how-to,
|
|
7
8
|
swapping that guide's local `model2vec` model for any OpenAI-compatible
|
|
@@ -21,19 +22,26 @@ is installed alongside it. The fragment declares:
|
|
|
21
22
|
|
|
22
23
|
- the [`sqlite-vec`](https://github.com/asg017/sqlite-vec) extension, for
|
|
23
24
|
`vec_distance_cosine()`;
|
|
24
|
-
- a `documents` table whose `on-file` hook embeds each
|
|
25
|
-
a TEXT `embedding` column;
|
|
25
|
+
- a `documents` table whose `on-file` hook embeds each
|
|
26
|
+
`**/*.{md,markdown,mdx,rst,txt,pdf}` file into a TEXT `embedding` column;
|
|
26
27
|
- a `pre-query` hook that embeds the incoming question and emits the
|
|
27
28
|
nearest-neighbor SQL.
|
|
28
29
|
|
|
29
30
|
Both hooks are console scripts that call the same embedder.
|
|
30
31
|
|
|
31
|
-
|
|
32
|
+
Every matched extension except `.pdf` is read as UTF-8 text; a `.pdf` is read with
|
|
32
33
|
[pypdf](https://pypdf.readthedocs.io), whose per-page extracted text is joined
|
|
33
34
|
and embedded like any other document. The extension check is case-insensitive
|
|
34
35
|
(`.PDF` is a PDF), though the glob above is not — globset matches case-sensitively,
|
|
35
36
|
so an uppercase-suffixed file needs its own `glob` entry to be picked up at all.
|
|
36
37
|
|
|
38
|
+
The glob is an allowlist rather than `**/*` for a reason worth knowing: a hook
|
|
39
|
+
that exits non-zero aborts the entire scan
|
|
40
|
+
([dirsql#697](https://github.com/thekevinscott/dirsql/issues/697)), and reading a
|
|
41
|
+
PNG as UTF-8 does exactly that. Matching everything would mean one image anywhere
|
|
42
|
+
under the root produces no index at all, so the list stays limited to what the
|
|
43
|
+
plugin can actually read.
|
|
44
|
+
|
|
37
45
|
A PDF that cannot be parsed aborts the scan rather than being skipped: the hook
|
|
38
46
|
exits non-zero and dirsql reports the path and the pypdf reason. A *scanned*,
|
|
39
47
|
image-only PDF is not a failure — pypdf yields no text, and the file is indexed
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"command": "uv run --with sqlite-vec --with-editable . --with-editable ../../packages/python python -m pytest tests/e2e -x -q",
|
|
3
|
-
"ran_at":
|
|
3
|
+
"ran_at": 1785339217,
|
|
4
4
|
"exit_code": 0,
|
|
5
|
-
"commit": "
|
|
6
|
-
"branch": "claude/
|
|
5
|
+
"commit": "f96d82085395d4e38bd74992a6ae3c7b029d5f60",
|
|
6
|
+
"branch": "claude/706-widen-glob"
|
|
7
7
|
}
|
|
@@ -15,7 +15,13 @@ entrypoint = "sqlite3_vec_init"
|
|
|
15
15
|
# The brace is globset alternation, not a dirsql `{name}` placeholder: the core
|
|
16
16
|
# rewrites those to `*`, but only when the braces hold a bare identifier, so a
|
|
17
17
|
# comma-separated list reaches globset intact.
|
|
18
|
+
#
|
|
19
|
+
# An allowlist rather than `**/*`, and not for tidiness: a hook that fails takes
|
|
20
|
+
# the whole scan down (dirsql#697), and reading a PNG as UTF-8 fails. Matching
|
|
21
|
+
# everything would mean one image anywhere under the root yields no index at
|
|
22
|
+
# all. The list is therefore what `read_content` can actually read -- prose, not
|
|
23
|
+
# source -- and widening it is safe only as far as that stays true.
|
|
18
24
|
[[table]]
|
|
19
25
|
ddl = "CREATE TABLE documents (path TEXT, text TEXT, embedding TEXT)"
|
|
20
|
-
glob = "**/*.{md,pdf}"
|
|
26
|
+
glob = "**/*.{md,markdown,mdx,rst,txt,pdf}"
|
|
21
27
|
on-file = "dirsql-embeddings-on-file {path}"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dirsql_plugin_embeddings-0.1.7 → dirsql_plugin_embeddings-0.1.8}/tests/integration/__init__.py
RENAMED
|
File without changes
|