rayzin 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. rayzin-0.0.1/.github/workflows/ci.yml +51 -0
  2. rayzin-0.0.1/.github/workflows/release.yml +56 -0
  3. rayzin-0.0.1/.gitignore +28 -0
  4. rayzin-0.0.1/.pre-commit-config.yaml +29 -0
  5. rayzin-0.0.1/CHANGELOG.md +9 -0
  6. rayzin-0.0.1/PKG-INFO +112 -0
  7. rayzin-0.0.1/README.md +98 -0
  8. rayzin-0.0.1/pixi.lock +3617 -0
  9. rayzin-0.0.1/pyproject.toml +125 -0
  10. rayzin-0.0.1/src/rayzin/__init__.py +26 -0
  11. rayzin-0.0.1/src/rayzin/enums.py +17 -0
  12. rayzin-0.0.1/src/rayzin/manifest/__init__.py +7 -0
  13. rayzin-0.0.1/src/rayzin/manifest/build.py +166 -0
  14. rayzin-0.0.1/src/rayzin/manifest/filtering.py +60 -0
  15. rayzin-0.0.1/src/rayzin/manifest/schema.py +77 -0
  16. rayzin-0.0.1/src/rayzin/manifest/spatial.py +137 -0
  17. rayzin-0.0.1/src/rayzin/metrics.py +156 -0
  18. rayzin-0.0.1/src/rayzin/pipeline.py +234 -0
  19. rayzin-0.0.1/src/rayzin/readers/__init__.py +21 -0
  20. rayzin-0.0.1/src/rayzin/readers/cog_reader.py +17 -0
  21. rayzin-0.0.1/src/rayzin/readers/protocol.py +17 -0
  22. rayzin-0.0.1/src/rayzin/readers/zarr_layout.py +65 -0
  23. rayzin-0.0.1/src/rayzin/readers/zarr_reader.py +85 -0
  24. rayzin-0.0.1/src/rayzin/search/__init__.py +20 -0
  25. rayzin-0.0.1/src/rayzin/search/backends/__init__.py +43 -0
  26. rayzin-0.0.1/src/rayzin/search/backends/faiss.py +292 -0
  27. rayzin-0.0.1/src/rayzin/search/backends/numpy.py +174 -0
  28. rayzin-0.0.1/src/rayzin/search/backends/protocols.py +42 -0
  29. rayzin-0.0.1/src/rayzin/search/block_searcher.py +211 -0
  30. rayzin-0.0.1/src/rayzin/search/heap_actor.py +25 -0
  31. rayzin-0.0.1/src/rayzin/search/results.py +37 -0
  32. rayzin-0.0.1/src/rayzin/types.py +83 -0
  33. rayzin-0.0.1/tests/__init__.py +0 -0
  34. rayzin-0.0.1/tests/unit/conftest.py +65 -0
  35. rayzin-0.0.1/tests/unit/manifest/test_build.py +98 -0
  36. rayzin-0.0.1/tests/unit/manifest/test_filtering.py +66 -0
  37. rayzin-0.0.1/tests/unit/manifest/test_spatial.py +23 -0
  38. rayzin-0.0.1/tests/unit/readers/test_zarr_reader.py +73 -0
  39. rayzin-0.0.1/tests/unit/search/test_backends.py +106 -0
  40. rayzin-0.0.1/tests/unit/search/test_block_searcher.py +282 -0
  41. rayzin-0.0.1/tests/unit/search/test_heap.py +65 -0
  42. rayzin-0.0.1/tests/unit/test_metrics.py +109 -0
  43. rayzin-0.0.1/tests/unit/test_pipeline.py +238 -0
@@ -0,0 +1,51 @@
1
+ name: CI
2
+
3
+ on:
4
+ pull_request:
5
+ push:
6
+ branches: [main]
7
+
8
+ concurrency:
9
+ group: ci-${{ github.ref }}
10
+ cancel-in-progress: true
11
+
12
+ jobs:
13
+ pre-commit:
14
+ runs-on: ubuntu-latest
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+
18
+ - uses: actions/setup-python@v5
19
+ with:
20
+ python-version: "3.12"
21
+
22
+ - uses: prefix-dev/setup-pixi@v0.9.5
23
+ with:
24
+ pixi-version: v0.62.2
25
+ manifest-path: pyproject.toml
26
+ environments: dev
27
+ locked: true
28
+ cache: true
29
+
30
+ - name: Run pre-commit
31
+ run: pixi run -e dev pre-commit run --all-files
32
+
33
+ test:
34
+ runs-on: ubuntu-latest
35
+ steps:
36
+ - uses: actions/checkout@v4
37
+
38
+ - uses: actions/setup-python@v5
39
+ with:
40
+ python-version: "3.12"
41
+
42
+ - uses: prefix-dev/setup-pixi@v0.9.5
43
+ with:
44
+ pixi-version: v0.62.2
45
+ manifest-path: pyproject.toml
46
+ environments: dev
47
+ locked: true
48
+ cache: true
49
+
50
+ - name: Run tests
51
+ run: pixi run -e dev pytest --cov=rayzin --cov-report=term-missing
@@ -0,0 +1,56 @@
1
+ name: Release
2
+
3
+ on:
4
+ workflow_run:
5
+ workflows: ["CI"]
6
+ types: [completed]
7
+
8
+ jobs:
9
+ release:
10
+ if: >
11
+ github.event.workflow_run.conclusion == 'success' &&
12
+ github.event.workflow_run.event == 'push' &&
13
+ github.event.workflow_run.head_branch == 'main'
14
+ runs-on: ubuntu-latest
15
+ concurrency: release
16
+ environment:
17
+ name: pypi
18
+ url: https://pypi.org/project/rayzin/
19
+ permissions:
20
+ id-token: write
21
+ contents: write
22
+
23
+ steps:
24
+ - uses: actions/checkout@v4
25
+ with:
26
+ fetch-depth: 0
27
+ ref: ${{ github.event.workflow_run.head_sha }}
28
+
29
+ - name: Attach checkout to source branch
30
+ run: git switch -C "${{ github.event.workflow_run.head_branch }}"
31
+
32
+ - uses: actions/setup-python@v5
33
+ with:
34
+ python-version: "3.12"
35
+
36
+ - name: Python Semantic Release
37
+ id: release
38
+ uses: python-semantic-release/python-semantic-release@v9.21.1
39
+ with:
40
+ github_token: ${{ secrets.GITHUB_TOKEN }}
41
+
42
+ - name: Install build tools
43
+ if: steps.release.outputs.released == 'true'
44
+ run: python -m pip install --upgrade pip build twine
45
+
46
+ - name: Build distributions
47
+ if: steps.release.outputs.released == 'true'
48
+ run: python -m build
49
+
50
+ - name: Check distributions
51
+ if: steps.release.outputs.released == 'true'
52
+ run: python -m twine check dist/*
53
+
54
+ - name: Publish to PyPI
55
+ if: steps.release.outputs.released == 'true'
56
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,28 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .pytest_cache/
5
+ .mypy_cache/
6
+ .ruff_cache/
7
+ .venv/
8
+ dist/
9
+ .coverage
10
+ .DS_Store
11
+
12
+ # data artifacts
13
+ *.zarr/
14
+ *.parquet
15
+ *.zstd
16
+
17
+ # model weights
18
+ *.pth
19
+ *.pt
20
+ *.ckpt
21
+ *.onnx
22
+
23
+ # IDE
24
+ .vscode/
25
+ .idea/
26
+
27
+ # pixi
28
+ .pixi/
@@ -0,0 +1,29 @@
1
+ repos:
2
+ - repo: https://github.com/astral-sh/ruff-pre-commit
3
+ rev: v0.4.0
4
+ hooks:
5
+ - id: ruff
6
+ args: [--fix]
7
+ - id: ruff-format
8
+
9
+ - repo: https://github.com/pre-commit/pre-commit-hooks
10
+ rev: v4.6.0
11
+ hooks:
12
+ - id: check-yaml
13
+ - id: end-of-file-fixer
14
+ - id: trailing-whitespace
15
+
16
+ - repo: local
17
+ hooks:
18
+ - id: mypy
19
+ name: mypy
20
+ language: system
21
+ entry: pixi run -e dev mypy --cache-dir=/tmp/rayzin-mypy-cache src/rayzin tests
22
+ pass_filenames: false
23
+
24
+ - id: pixi-lock
25
+ name: pixi-lock
26
+ language: system
27
+ entry: pixi install --locked
28
+ pass_filenames: false
29
+ files: pyproject.toml
@@ -0,0 +1,9 @@
1
+ # CHANGELOG
2
+
3
+
4
+ ## v0.0.1 (2026-04-18)
5
+
6
+ ### Bug Fixes
7
+
8
+ - Simplify __init__.py
9
+ ([`802be91`](https://github.com/ljstrnadiii/rayzin/commit/802be91a730fa361cabaa5ac57b15a2bce38a955))
rayzin-0.0.1/PKG-INFO ADDED
@@ -0,0 +1,112 @@
1
+ Metadata-Version: 2.4
2
+ Name: rayzin
3
+ Version: 0.0.1
4
+ Summary: Exact KNN search over massive Zarr/COG datasets using Ray Data
5
+ Author: rayzin contributors
6
+ Requires-Python: >=3.11
7
+ Requires-Dist: numpy>=1.26.0
8
+ Requires-Dist: obstore>=0.6
9
+ Requires-Dist: pyarrow>=16.0.0
10
+ Requires-Dist: ray[data]>=2.30.0
11
+ Requires-Dist: shapely>=2.0.0
12
+ Requires-Dist: zarr>=3.0.0
13
+ Description-Content-Type: text/markdown
14
+
15
+ # `rayzin`
16
+ [![CI](https://github.com/ljstrnadiii/rayzin/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/ljstrnadiii/rayzin/actions/workflows/ci.yml)
17
+ [![Release](https://github.com/ljstrnadiii/rayzin/actions/workflows/release.yml/badge.svg?branch=main)](https://github.com/ljstrnadiii/rayzin/actions/workflows/release.yml)
18
+
19
+ `rayzin` is a Ray project for exact k-nearest-neighbor search over chunked embedding arrays stored
20
+ in Zarr.
21
+
22
+ It is designed for workflows where:
23
+ - Embeddings are stored as chunked arrays, typically with a trailing feature dimension.
24
+ - Search should prune whole chunks before loading vector payloads whenever possible.
25
+ - Relevant chunks are read once, searched in parallel with Ray, and merged into a global top-k.
26
+
27
+ The project uses a chunk-first execution model:
28
+ 1. Build or read a manifest describing each chunk by slice, count, centroid, and radius.
29
+ 2. Apply manifest-level pushdown with Ray Data using predicate filters and optional AOI filtering.
30
+ 3. Compute query-to-chunk lower bounds in vectorized batches.
31
+ 4. Search only the active query subset for each surviving chunk after pruning.
32
+ 5. Merge local top-k results into a global heap actor and return flat result rows.
33
+
34
+ This architecture keeps chunk reads bounded, avoids rescanning the same data for each query, and
35
+ fits naturally into Ray Data execution.
36
+
37
+ ## Features
38
+
39
+ - **Manifest-driven pruning:** Chunk summaries let the pipeline skip work before loading dense
40
+ vectors, which is the main lever for scaling exact search.
41
+ - **Ray-native distributed execution:** Manifest filtering, chunk reads, and search fan out through
42
+ Ray Data while final top-k merging stays centralized in a heap actor.
43
+ - **Pluggable search backends:** NumPy and FAISS backends share the same pipeline so pruning and
44
+ result handling stay consistent across implementations.
45
+
46
+ ## Current Scope
47
+
48
+ - Zarr-backed search path.
49
+ - Batched query input shaped `(1, d)` or `(nq, d)`.
50
+ - Euclidean pruning path.
51
+ - NumPy backend by default, with FAISS supported when installed separately.
52
+ - Early COG entry points are present, but the COG path is not the main supported flow yet.
53
+
54
+ ## Installation
55
+
56
+ Install the base package from PyPI:
57
+
58
+ ```bash
59
+ pip install rayzin
60
+ ```
61
+
62
+ That gives you the core Zarr search path and the NumPy backend.
63
+
64
+ If you want the FAISS backend, install FAISS separately after installing `rayzin`:
65
+
66
+ ```bash
67
+ pip install faiss-cpu
68
+ ```
69
+
70
+ For FAISS GPU, use your platform's recommended FAISS installation method. In practice that is
71
+ often conda or Pixi on `linux-64`.
72
+
73
+ If you are working from a local checkout and want an editable install:
74
+
75
+ ```bash
76
+ pip install -e .
77
+ ```
78
+
79
+ ## Development
80
+
81
+ - Install Pixi: `curl -fsSL https://pixi.sh/install.sh | sh`
82
+ - Install dependencies: `pixi install -e dev`
83
+ - Run lint: `pixi run -e dev ruff check .`
84
+ - Run format check: `pixi run -e dev ruff format --check .`
85
+ - Run type check: `pixi run -e dev mypy src/rayzin tests`
86
+ - Run tests: `pixi run -e dev pytest`
87
+
88
+ ## Release
89
+
90
+ Releases are cut automatically from `main` after the `CI` workflow passes. Use Conventional
91
+ Commits for changes that should trigger a release, and semantic-release will:
92
+
93
+ - compute the next version
94
+ - update `pyproject.toml` and `src/rayzin/__init__.py`
95
+ - refresh `pixi.lock`
96
+ - create the version tag and GitHub release
97
+ - build and publish the package to PyPI via GitHub Actions trusted publishing
98
+
99
+ ## Roadmap
100
+
101
+ 1. Keep pruning metrics and search-time metrics aligned across all supported backends.
102
+ 2. Generalize pruning beyond a single radius or tau so multiple pivots or metric-independent
103
+ pruning are possible.
104
+ 3. Simplify filtering to chunk existence, AOI intersection, and straightforward coordinate
105
+ subsetting.
106
+ 4. Add support for native dtypes when no scale or offset or other CF decoding is required.
107
+ 5. Add support for pre-transforms such as dequantization before search.
108
+ 6. Prefer `from __future__ import annotations` over quoted forward references.
109
+ 7. Add logging and timing around pruning, loading, decoding, and per-block search.
110
+ 8. Revisit float32 assumptions and document where they are required by FAISS.
111
+ 9. Make radius-based manifest data optional if pruning mode does not require it.
112
+ 10. Revisit rows-per-block tuning and execution sizing once the main search path stabilizes.
rayzin-0.0.1/README.md ADDED
@@ -0,0 +1,98 @@
1
+ # `rayzin`
2
+ [![CI](https://github.com/ljstrnadiii/rayzin/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/ljstrnadiii/rayzin/actions/workflows/ci.yml)
3
+ [![Release](https://github.com/ljstrnadiii/rayzin/actions/workflows/release.yml/badge.svg?branch=main)](https://github.com/ljstrnadiii/rayzin/actions/workflows/release.yml)
4
+
5
+ `rayzin` is a Ray project for exact k-nearest-neighbor search over chunked embedding arrays stored
6
+ in Zarr.
7
+
8
+ It is designed for workflows where:
9
+ - Embeddings are stored as chunked arrays, typically with a trailing feature dimension.
10
+ - Search should prune whole chunks before loading vector payloads whenever possible.
11
+ - Relevant chunks are read once, searched in parallel with Ray, and merged into a global top-k.
12
+
13
+ The project uses a chunk-first execution model:
14
+ 1. Build or read a manifest describing each chunk by slice, count, centroid, and radius.
15
+ 2. Apply manifest-level pushdown with Ray Data using predicate filters and optional AOI filtering.
16
+ 3. Compute query-to-chunk lower bounds in vectorized batches.
17
+ 4. Search only the active query subset for each surviving chunk after pruning.
18
+ 5. Merge local top-k results into a global heap actor and return flat result rows.
19
+
20
+ This architecture keeps chunk reads bounded, avoids rescanning the same data for each query, and
21
+ fits naturally into Ray Data execution.
22
+
23
+ ## Features
24
+
25
+ - **Manifest-driven pruning:** Chunk summaries let the pipeline skip work before loading dense
26
+ vectors, which is the main lever for scaling exact search.
27
+ - **Ray-native distributed execution:** Manifest filtering, chunk reads, and search fan out through
28
+ Ray Data while final top-k merging stays centralized in a heap actor.
29
+ - **Pluggable search backends:** NumPy and FAISS backends share the same pipeline so pruning and
30
+ result handling stay consistent across implementations.
31
+
32
+ ## Current Scope
33
+
34
+ - Zarr-backed search path.
35
+ - Batched query input shaped `(1, d)` or `(nq, d)`.
36
+ - Euclidean pruning path.
37
+ - NumPy backend by default, with FAISS supported when installed separately.
38
+ - Early COG entry points are present, but the COG path is not the main supported flow yet.
39
+
40
+ ## Installation
41
+
42
+ Install the base package from PyPI:
43
+
44
+ ```bash
45
+ pip install rayzin
46
+ ```
47
+
48
+ That gives you the core Zarr search path and the NumPy backend.
49
+
50
+ If you want the FAISS backend, install FAISS separately after installing `rayzin`:
51
+
52
+ ```bash
53
+ pip install faiss-cpu
54
+ ```
55
+
56
+ For FAISS GPU, use your platform's recommended FAISS installation method. In practice that is
57
+ often conda or Pixi on `linux-64`.
58
+
59
+ If you are working from a local checkout and want an editable install:
60
+
61
+ ```bash
62
+ pip install -e .
63
+ ```
64
+
65
+ ## Development
66
+
67
+ - Install Pixi: `curl -fsSL https://pixi.sh/install.sh | sh`
68
+ - Install dependencies: `pixi install -e dev`
69
+ - Run lint: `pixi run -e dev ruff check .`
70
+ - Run format check: `pixi run -e dev ruff format --check .`
71
+ - Run type check: `pixi run -e dev mypy src/rayzin tests`
72
+ - Run tests: `pixi run -e dev pytest`
73
+
74
+ ## Release
75
+
76
+ Releases are cut automatically from `main` after the `CI` workflow passes. Use Conventional
77
+ Commits for changes that should trigger a release, and semantic-release will:
78
+
79
+ - compute the next version
80
+ - update `pyproject.toml` and `src/rayzin/__init__.py`
81
+ - refresh `pixi.lock`
82
+ - create the version tag and GitHub release
83
+ - build and publish the package to PyPI via GitHub Actions trusted publishing
84
+
85
+ ## Roadmap
86
+
87
+ 1. Keep pruning metrics and search-time metrics aligned across all supported backends.
88
+ 2. Generalize pruning beyond a single radius or tau so multiple pivots or metric-independent
89
+ pruning are possible.
90
+ 3. Simplify filtering to chunk existence, AOI intersection, and straightforward coordinate
91
+ subsetting.
92
+ 4. Add support for native dtypes when no scale or offset or other CF decoding is required.
93
+ 5. Add support for pre-transforms such as dequantization before search.
94
+ 6. Prefer `from __future__ import annotations` over quoted forward references.
95
+ 7. Add logging and timing around pruning, loading, decoding, and per-block search.
96
+ 8. Revisit float32 assumptions and document where they are required by FAISS.
97
+ 9. Make radius-based manifest data optional if pruning mode does not require it.
98
+ 10. Revisit rows-per-block tuning and execution sizing once the main search path stabilizes.