rgapi 0.1.22__tar.gz → 0.1.24__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rgapi-0.1.22 → rgapi-0.1.24}/.github/workflows/ci.yml +7 -1
- {rgapi-0.1.22 → rgapi-0.1.24}/Cargo.lock +14 -14
- {rgapi-0.1.22 → rgapi-0.1.24}/Cargo.toml +18 -1
- {rgapi-0.1.22 → rgapi-0.1.24}/DEV.md +3 -4
- {rgapi-0.1.22 → rgapi-0.1.24}/PKG-INFO +12 -3
- {rgapi-0.1.22 → rgapi-0.1.24}/README.md +11 -2
- {rgapi-0.1.22 → rgapi-0.1.24}/pyproject.toml +3 -0
- rgapi-0.1.24/python/rgapi/_cli.py +28 -0
- rgapi-0.1.24/python/rgapi/skill.py +35 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/rustfmt.toml +1 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/src/block.rs +13 -46
- {rgapi-0.1.22 → rgapi-0.1.24}/src/lib.rs +6 -20
- {rgapi-0.1.22 → rgapi-0.1.24}/src/nb.rs +17 -68
- {rgapi-0.1.22 → rgapi-0.1.24}/src/python.rs +32 -97
- {rgapi-0.1.22 → rgapi-0.1.24}/src/search.rs +25 -92
- {rgapi-0.1.22 → rgapi-0.1.24}/src/walk.rs +37 -116
- rgapi-0.1.22/python/rgapi/skill.py +0 -51
- {rgapi-0.1.22 → rgapi-0.1.24}/.gitignore +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/LICENSE +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/_config.yml +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/_layouts/default.html +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/python/rgapi/__init__.py +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/python/rgapi/block.py +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/python/rgapi/nb.py +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/tests/test_async.py +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/tests/test_rgapi.py +0 -0
- {rgapi-0.1.22 → rgapi-0.1.24}/tools/bench.py +0 -0
|
@@ -29,7 +29,7 @@ jobs:
|
|
|
29
29
|
- uses: actions/checkout@v7
|
|
30
30
|
- uses: PyO3/maturin-action@v1
|
|
31
31
|
with:
|
|
32
|
-
args: --
|
|
32
|
+
args: --profile dist --out dist -i python3.10 -i python3.11 -i python3.12 -i python3.13
|
|
33
33
|
manylinux: auto
|
|
34
34
|
- uses: actions/upload-artifact@v7
|
|
35
35
|
with:
|
|
@@ -58,6 +58,12 @@ jobs:
|
|
|
58
58
|
contents: write
|
|
59
59
|
steps:
|
|
60
60
|
- uses: actions/checkout@v7
|
|
61
|
+
- uses: dtolnay/rust-toolchain@stable
|
|
62
|
+
- id: crates-auth
|
|
63
|
+
uses: rust-lang/crates-io-auth-action@v1
|
|
64
|
+
- run: cargo publish
|
|
65
|
+
env:
|
|
66
|
+
CARGO_REGISTRY_TOKEN: ${{ steps.crates-auth.outputs.token }}
|
|
61
67
|
- uses: actions/download-artifact@v8
|
|
62
68
|
with:
|
|
63
69
|
path: dist
|
|
@@ -30,18 +30,18 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
|
|
|
30
30
|
|
|
31
31
|
[[package]]
|
|
32
32
|
name = "crc32fast"
|
|
33
|
-
version = "1.5.
|
|
33
|
+
version = "1.5.1"
|
|
34
34
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
35
|
-
checksum = "
|
|
35
|
+
checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550"
|
|
36
36
|
dependencies = [
|
|
37
37
|
"cfg-if",
|
|
38
38
|
]
|
|
39
39
|
|
|
40
40
|
[[package]]
|
|
41
41
|
name = "crossbeam-deque"
|
|
42
|
-
version = "0.8.
|
|
42
|
+
version = "0.8.8"
|
|
43
43
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
44
|
-
checksum = "
|
|
44
|
+
checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a"
|
|
45
45
|
dependencies = [
|
|
46
46
|
"crossbeam-epoch",
|
|
47
47
|
"crossbeam-utils",
|
|
@@ -49,18 +49,18 @@ dependencies = [
|
|
|
49
49
|
|
|
50
50
|
[[package]]
|
|
51
51
|
name = "crossbeam-epoch"
|
|
52
|
-
version = "0.9.
|
|
52
|
+
version = "0.9.21"
|
|
53
53
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
54
|
-
checksum = "
|
|
54
|
+
checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d"
|
|
55
55
|
dependencies = [
|
|
56
56
|
"crossbeam-utils",
|
|
57
57
|
]
|
|
58
58
|
|
|
59
59
|
[[package]]
|
|
60
60
|
name = "crossbeam-utils"
|
|
61
|
-
version = "0.8.
|
|
61
|
+
version = "0.8.23"
|
|
62
62
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
63
|
-
checksum = "
|
|
63
|
+
checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
|
|
64
64
|
|
|
65
65
|
[[package]]
|
|
66
66
|
name = "encoding_rs"
|
|
@@ -166,9 +166,9 @@ checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
|
|
166
166
|
|
|
167
167
|
[[package]]
|
|
168
168
|
name = "log"
|
|
169
|
-
version = "0.4.
|
|
169
|
+
version = "0.4.34"
|
|
170
170
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
171
|
-
checksum = "
|
|
171
|
+
checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6"
|
|
172
172
|
|
|
173
173
|
[[package]]
|
|
174
174
|
name = "memchr"
|
|
@@ -291,7 +291,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
|
|
|
291
291
|
|
|
292
292
|
[[package]]
|
|
293
293
|
name = "rgapi"
|
|
294
|
-
version = "0.1.
|
|
294
|
+
version = "0.1.24"
|
|
295
295
|
dependencies = [
|
|
296
296
|
"crc32fast",
|
|
297
297
|
"globset",
|
|
@@ -340,7 +340,7 @@ checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
|
|
340
340
|
dependencies = [
|
|
341
341
|
"proc-macro2",
|
|
342
342
|
"quote",
|
|
343
|
-
"syn 3.0.
|
|
343
|
+
"syn 3.0.5",
|
|
344
344
|
]
|
|
345
345
|
|
|
346
346
|
[[package]]
|
|
@@ -369,9 +369,9 @@ dependencies = [
|
|
|
369
369
|
|
|
370
370
|
[[package]]
|
|
371
371
|
name = "syn"
|
|
372
|
-
version = "3.0.
|
|
372
|
+
version = "3.0.5"
|
|
373
373
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
374
|
-
checksum = "
|
|
374
|
+
checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9"
|
|
375
375
|
dependencies = [
|
|
376
376
|
"proc-macro2",
|
|
377
377
|
"quote",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "rgapi"
|
|
3
|
-
version = "0.1.
|
|
3
|
+
version = "0.1.24"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
rust-version = "1.91"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -31,3 +31,20 @@ extension-module = ["python", "pyo3/extension-module"]
|
|
|
31
31
|
|
|
32
32
|
[lints.clippy]
|
|
33
33
|
too_many_arguments = "allow"
|
|
34
|
+
|
|
35
|
+
[profile.release]
|
|
36
|
+
lto = false
|
|
37
|
+
codegen-units = 16
|
|
38
|
+
|
|
39
|
+
[profile.release.package.rgapi]
|
|
40
|
+
incremental = true
|
|
41
|
+
|
|
42
|
+
[profile.dist]
|
|
43
|
+
inherits = "release"
|
|
44
|
+
lto = true
|
|
45
|
+
incremental = false
|
|
46
|
+
codegen-units = 1
|
|
47
|
+
strip = true
|
|
48
|
+
|
|
49
|
+
[profile.dist.package.rgapi]
|
|
50
|
+
incremental = false
|
|
@@ -9,7 +9,7 @@ src/walk.rs ignore/globset/grep-regex-backed path walking and filtering
|
|
|
9
9
|
src/search.rs grep-regex/grep-searcher-backed searching
|
|
10
10
|
src/block.rs blank-line-delimited block grouping, matching, and block context
|
|
11
11
|
src/python.rs PyO3 classes and private core functions
|
|
12
|
-
python/rgapi/ public Python wrappers over `rgapi._core`
|
|
12
|
+
python/rgapi/ public Python wrappers over `rgapi._core`, plus the `rgapi-nbrg` CLI
|
|
13
13
|
tests/ pytest coverage for the Python API
|
|
14
14
|
```
|
|
15
15
|
|
|
@@ -34,14 +34,13 @@ Release flow is: release first, then bump - `ship-release` does both.
|
|
|
34
34
|
2. Confirm the release version in `Cargo.toml` (`[package].version`).
|
|
35
35
|
3. Run `ship-release`. It tags `v<version>`, pushes branch and tag (CI builds and publishes), then bumps `Cargo.toml`, refreshes the editable install, and pushes the bump without a tag.
|
|
36
36
|
|
|
37
|
-
The GitHub workflow builds wheels for Python 3.10-3.13 on Linux and macOS and publishes
|
|
37
|
+
The GitHub workflow builds wheels for Python 3.10-3.13 on Linux and macOS and publishes the Rust crate, GitHub release artifacts, and PyPI package when a `v*` tag is pushed.
|
|
38
38
|
|
|
39
39
|
## Design notes
|
|
40
40
|
|
|
41
41
|
Paths in `fd`, `walk`, `rg`, and `rg_iter` results are relative to the requested root and use `/` separators. Traversal uses `ignore::WalkParallel`, so result order is not part of the API contract. Search results are structured rows; collected result lists use rg-style `str()` and notebook display. `SearchLine.lnhash` is computed with the same CRC-32-based line-content hash format as exhash (`lineno|hash|`, low 16 bits of CRC-32 over the line's UTF-8 bytes); `lnhashs=True` only changes row display, not `line_number` or matching behavior. Path regexes filter returned/searched paths; `skip_dir` and `skip_dir_re` prune traversal through `ignore::WalkBuilder::filter_entry`. Depth, size, symlink, filesystem, hidden, and ignore options are direct `ignore::WalkBuilder` settings. `rg_iter` exposes the same parallel search stream that `rg` collects by default; `paths=True` and `count=True` consume that stream with different reducers. Binary files and invalid UTF-8 are skipped for now.
|
|
42
42
|
|
|
43
|
-
Streaming engine: `walk.rs` owns the generic machinery. `StreamIter<T>` is the worker-thread-plus-bounded-channel iterator (`sync_channel(8192)`, so producers block rather than buffer without limit when a consumer lags), and `spawn_walk` owns the shared scaffold: walker config, panic catching, cancel flag, and worker thread. `rg_iter` (`T = SearchLine`), `block_iter` (`T = SearchBlock`), `nb_iter` (`T = NbCell`), and `find_iter` (`T = String`, the path walk) plug entry closures into that engine. Block search reads each file once, searches it once, groups nonblank lines into blocks, maps matching lines to their blocks, and expands context by block index.
|
|
44
|
-
Each `SearchBlock` carries numeric boundaries plus hashes for its first and last source lines. Python keeps both and chooses the displayed address without another file read.
|
|
43
|
+
Streaming engine: `walk.rs` owns the generic machinery. `StreamIter<T>` is the worker-thread-plus-bounded-channel iterator (`sync_channel(8192)`, so producers block rather than buffer without limit when a consumer lags), and `spawn_walk` owns the shared scaffold: walker config, panic catching, cancel flag, and worker thread. `rg_iter` (`T = SearchLine`), `block_iter` (`T = SearchBlock`), `nb_iter` (`T = NbCell`), and `find_iter` (`T = String`, the path walk) plug entry closures into that engine. Block search reads each file once, searches it once, groups nonblank lines into blocks, maps matching lines to their blocks, and expands context by block index. Each `SearchBlock` carries numeric boundaries plus hashes for its first and last source lines. Python keeps both and chooses the displayed address without another file read.
|
|
45
44
|
|
|
46
45
|
Async API: `fda`, `fda_iter`, `rga`, `rga_iter`, `nbrga`, and `nbrga_iter` wrap the corresponding private core operations. `rga(summary=True)` uses `_core.block_search_async`; ordinary `rga` uses `_core.rg_async`. Each collected core function takes a Python callback, runs on Rust threads through the generic `stream_async` helper, and delivers with one GIL attach at the end. Iterator forms use `stream_iter_async` and attach once per batch. The Python side settles an `asyncio.Future` or feeds an `asyncio.Queue` via `loop.call_soon_threadsafe`; no Python thread blocks and `asyncio.to_thread` is not involved. `AsyncHandle.cancel()` sets the same atomic flag used by the Rust iterators.
|
|
47
46
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rgapi
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.24
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
6
6
|
Requires-Dist: fastcore>=1.14.6
|
|
@@ -77,8 +77,7 @@ pip install rgapi
|
|
|
77
77
|
|
|
78
78
|
## Semantics
|
|
79
79
|
|
|
80
|
-
`fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters.
|
|
81
|
-
`root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
|
|
80
|
+
`fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
|
|
82
81
|
|
|
83
82
|
`fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
|
|
84
83
|
|
|
@@ -154,6 +153,16 @@ matches list of SearchLine rows for the matched lines within the cell
|
|
|
154
153
|
|
|
155
154
|
Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
|
|
156
155
|
|
|
156
|
+
`rgapi-nbrg` exposes notebook search without requiring a Python kernel:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
rgapi-nbrg 'read_csv' .
|
|
160
|
+
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
161
|
+
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Run `rgapi-nbrg --help` for its discovery, matching, and output options.
|
|
165
|
+
|
|
157
166
|
## Async
|
|
158
167
|
|
|
159
168
|
`fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
|
|
@@ -56,8 +56,7 @@ pip install rgapi
|
|
|
56
56
|
|
|
57
57
|
## Semantics
|
|
58
58
|
|
|
59
|
-
`fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters.
|
|
60
|
-
`root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
|
|
59
|
+
`fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
|
|
61
60
|
|
|
62
61
|
`fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
|
|
63
62
|
|
|
@@ -133,6 +132,16 @@ matches list of SearchLine rows for the matched lines within the cell
|
|
|
133
132
|
|
|
134
133
|
Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
|
|
135
134
|
|
|
135
|
+
`rgapi-nbrg` exposes notebook search without requiring a Python kernel:
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
rgapi-nbrg 'read_csv' .
|
|
139
|
+
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
140
|
+
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Run `rgapi-nbrg --help` for its discovery, matching, and output options.
|
|
144
|
+
|
|
136
145
|
## Async
|
|
137
146
|
|
|
138
147
|
`fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"Command-line access to notebook-aware search."
|
|
2
|
+
|
|
3
|
+
from fastcore.script import call_parse
|
|
4
|
+
from . import nbrg
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@call_parse(pos=['root'])
|
|
8
|
+
def nbrg_cli(
|
|
9
|
+
pattern:str, # Regex pattern to search for
|
|
10
|
+
root:str='.', # File or directory to search
|
|
11
|
+
cell_context:int=0, # Neighbouring cells to include before and after matches
|
|
12
|
+
multiline:bool=False, # Allow matches across lines within a cell?
|
|
13
|
+
smart_case:bool=False, # Use case-sensitive matching when the pattern contains uppercase?
|
|
14
|
+
case:bool=False, # Force case-sensitive matching?
|
|
15
|
+
paths:bool=False, # Return only matching notebook paths?
|
|
16
|
+
count:bool=False, # Return the number of matching cells?
|
|
17
|
+
max_results:int=None, # Maximum matching cells to return
|
|
18
|
+
maxlen:int=180, # Maximum source characters displayed per cell
|
|
19
|
+
glob:str=None, # Glob selecting notebook paths
|
|
20
|
+
exclude:str=None, # Glob excluding notebook paths
|
|
21
|
+
hidden:bool=False, # Search hidden files and directories?
|
|
22
|
+
max_depth:int=None, # Maximum directory depth to search
|
|
23
|
+
timeout_ms:int=None, # Stop searching after this many milliseconds
|
|
24
|
+
):
|
|
25
|
+
"Search notebook cell sources and report stable cell IDs."
|
|
26
|
+
print(nbrg(pattern, root, cell_context=cell_context, multiline=multiline, smart_case=smart_case,
|
|
27
|
+
case_sensitive=True if case else None, paths=paths, count=count, max_results=max_results, maxlen=maxlen,
|
|
28
|
+
glob=glob, exclude=exclude, hidden=hidden, max_depth=max_depth, timeout_ms=timeout_ms))
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Find files, search text, and search notebook cell sources from Python with ripgrep semantics and structured results. Use for fd-style discovery, regex searches, and notebook searches returning stable cell IDs rather than escaped JSON.
|
|
2
|
+
|
|
3
|
+
Uses ripgrep's `ignore`, `grep-regex`, and `grep-searcher` crates, including ignore files, hidden-file handling, globs, extensions, and regex semantics. Prefer these APIs to shell parsing or manual file scans in Python; use `rgstr` for held text instead of split-line loops, and `ls`/`fd` for kernel-side listings.
|
|
4
|
+
|
|
5
|
+
For orientation, start with `rg(summary=True)`; use line-level results where needed and `lnhashs=True` when edits may follow. Summary blocks suit prose/config paragraphs as well as code. Display results bare; narrow oversized results with API parameters rather than joining, slicing, or reformatting them.
|
|
6
|
+
|
|
7
|
+
## Search units
|
|
8
|
+
|
|
9
|
+
- `rg`: lines, or `SearchBlock` rows with `summary=True`. Blank/whitespace-only lines separate blocks; multiple matches in one block yield one row. Context counts the selected unit. Summary mode cannot combine with `paths` or `count`, but supports hashed block boundaries.
|
|
10
|
+
- `nbrg`: `NbResults` of `NbCell` rows from source only, never metadata/outputs. `multiline=True` matches across cell lines while `^`/`$` remain line anchors; ordinary line-oriented `rg` rejects newline patterns.
|
|
11
|
+
- Traversal/search run in parallel in Rust; sort when stable order is required. `path_re`/`skip_path_re` filter paths without pruning; `skip_dir`/`skip_dir_re` prune subtrees.
|
|
12
|
+
|
|
13
|
+
## Result fields and display
|
|
14
|
+
|
|
15
|
+
`FileEntry` is a slash-separated relative-path `str` with lazy `size`/`mtime`/`is_dir`/`stat`. Path lists render as ls-style tables capped at `MAX_REPR`; `str(res)`/`list(res)` yield plain paths. Unfollowed symlinks remain, marked `l`; `link_target` is their target or `None` for non-links. `ls(hidden=True)` corresponds to `ls -a`.
|
|
16
|
+
|
|
17
|
+
Search rows provide `asdict()`. All paths are relative to the search root:
|
|
18
|
+
|
|
19
|
+
| Row | Location | Content/matches | `kind` |
|
|
20
|
+
|---|---|---|---|
|
|
21
|
+
| `SearchLine` | `path`, 1-based `line_number`, `lnhash` | `line` without trailing newline; `matches` as byte-offset `(start, end)` pairs on match rows | match/before/after/context |
|
|
22
|
+
| `SearchBlock` | `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash` | full `source`; `matches` as matching `SearchLine`s | match/context |
|
|
23
|
+
| `NbCell` | `path`, `cell_index`, `cell_id`, `cell_type` (code/markdown/raw) | full `source`; `matches` as matching `SearchLine`s | match/context |
|
|
24
|
+
|
|
25
|
+
Block displays use `path:start-end:source`; notebook displays use `path:cell_id:source`. Context replaces the last separator colon with `-`. Hashed block locations are `start_lnhash,end_lnhash`, or one hash for a single line. Newline runs display as ¶, retaining indentation; `maxlen` limits displayed source, not stored `source`.
|
|
26
|
+
|
|
27
|
+
Notebook match previews start at the first matched line, marking omitted earlier lines as `…[Ln]` (1-based), but retain a leading directive: `#| export…[L4]needle here`. `cell_context` counts neighbouring cells. Consult individual function docs for parameters and reduction modes.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from . import RgIter, fd, ls, nbrg, rg, rg_iter, rgstr
|
|
31
|
+
|
|
32
|
+
__all__ = [ "RgIter", "fd", "ls", "rg", "rg_iter", "nbrg", "rgstr" ]
|
|
33
|
+
|
|
34
|
+
__pyskill_params__ = {'walk_params': ('glob', 'include', 'exclude', 'hidden', 'min_depth', 'max_filesize',
|
|
35
|
+
'follow_links', 'same_file_system', 'path_re', 'skip_path_re', 'skip_dir', 'skip_dir_re')}
|
|
@@ -23,11 +23,7 @@ pub struct SearchBlock {
|
|
|
23
23
|
pub matches: Vec<SearchLine>,
|
|
24
24
|
}
|
|
25
25
|
|
|
26
|
-
struct BlockInfo {
|
|
27
|
-
start_line: u64,
|
|
28
|
-
end_line: u64,
|
|
29
|
-
source: String,
|
|
30
|
-
}
|
|
26
|
+
struct BlockInfo { start_line: u64, end_line: u64, source: String }
|
|
31
27
|
|
|
32
28
|
fn split_blocks(text: &str) -> Vec<BlockInfo> {
|
|
33
29
|
let mut blocks = Vec::new();
|
|
@@ -41,31 +37,20 @@ fn split_blocks(text: &str) -> Vec<BlockInfo> {
|
|
|
41
37
|
lines.clear();
|
|
42
38
|
}
|
|
43
39
|
} else {
|
|
44
|
-
if start.is_none() {
|
|
45
|
-
start = Some(line_no);
|
|
46
|
-
}
|
|
40
|
+
if start.is_none() { start = Some(line_no); }
|
|
47
41
|
lines.push(line);
|
|
48
42
|
}
|
|
49
43
|
}
|
|
50
|
-
if let Some(first) = start {
|
|
51
|
-
blocks.push(BlockInfo { start_line: first, end_line: text.lines().count() as u64, source: lines.join("\n") });
|
|
52
|
-
}
|
|
44
|
+
if let Some(first) = start { blocks.push(BlockInfo { start_line: first, end_line: text.lines().count() as u64, source: lines.join("\n") }); }
|
|
53
45
|
blocks
|
|
54
46
|
}
|
|
55
47
|
|
|
56
48
|
fn process_file(disp: String, bytes: &[u8], matcher: &RegexMatcher, before_context: usize, after_context: usize) -> Result<Vec<SearchBlock>, RgApiError> {
|
|
57
|
-
if bytes.contains(&0) {
|
|
58
|
-
|
|
59
|
-
}
|
|
60
|
-
let text = match std::str::from_utf8(bytes) {
|
|
61
|
-
Ok(text) => text,
|
|
62
|
-
Err(_) => return Ok(Vec::new()),
|
|
63
|
-
};
|
|
49
|
+
if bytes.contains(&0) { return Ok(Vec::new()); }
|
|
50
|
+
let text = match std::str::from_utf8(bytes) { Ok(text) => text, Err(_) => return Ok(Vec::new()) };
|
|
64
51
|
let blocks = split_blocks(text);
|
|
65
52
|
let hits = search_text(disp.clone(), text, matcher.clone(), 0, 0, false)?;
|
|
66
|
-
if hits.is_empty() {
|
|
67
|
-
return Ok(Vec::new());
|
|
68
|
-
}
|
|
53
|
+
if hits.is_empty() { return Ok(Vec::new()); }
|
|
69
54
|
let mut matched: HashMap<usize, Vec<SearchLine>> = HashMap::new();
|
|
70
55
|
for hit in hits {
|
|
71
56
|
if let Some((i, _)) = blocks.iter().enumerate().find(|(_, b)| b.start_line <= hit.line_number && hit.line_number <= b.end_line) {
|
|
@@ -77,9 +62,7 @@ fn process_file(disp: String, bytes: &[u8], matcher: &RegexMatcher, before_conte
|
|
|
77
62
|
emit.insert(*i, true);
|
|
78
63
|
let start = i.saturating_sub(before_context);
|
|
79
64
|
let end = (i + after_context + 1).min(blocks.len());
|
|
80
|
-
for j in start..end {
|
|
81
|
-
emit.entry(j).or_insert(false);
|
|
82
|
-
}
|
|
65
|
+
for j in start..end { emit.entry(j).or_insert(false); }
|
|
83
66
|
}
|
|
84
67
|
Ok(emit
|
|
85
68
|
.into_iter()
|
|
@@ -109,25 +92,13 @@ fn block_entry(
|
|
|
109
92
|
after_context: usize,
|
|
110
93
|
max_depth: Option<usize>,
|
|
111
94
|
) -> Result<Vec<SearchBlock>, RgApiError> {
|
|
112
|
-
let dent = match entry {
|
|
113
|
-
Ok(dent) => dent,
|
|
114
|
-
Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err),
|
|
115
|
-
};
|
|
95
|
+
let dent = match entry { Ok(dent) => dent, Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err) };
|
|
116
96
|
let path = dent.path();
|
|
117
|
-
let Some(ft) = dent.file_type() else {
|
|
118
|
-
|
|
119
|
-
};
|
|
120
|
-
if !ft.is_file() {
|
|
121
|
-
return Ok(Vec::new());
|
|
122
|
-
}
|
|
97
|
+
let Some(ft) = dent.file_type() else { return Ok(Vec::new()); };
|
|
98
|
+
if !ft.is_file() { return Ok(Vec::new()); }
|
|
123
99
|
let rel = rel_path(root, path);
|
|
124
|
-
if !filters.path_allowed(&rel) {
|
|
125
|
-
|
|
126
|
-
}
|
|
127
|
-
let bytes = match std::fs::read(path) {
|
|
128
|
-
Ok(bytes) => bytes,
|
|
129
|
-
Err(_) => return Ok(Vec::new()),
|
|
130
|
-
};
|
|
100
|
+
if !filters.path_allowed(&rel) { return Ok(Vec::new()); }
|
|
101
|
+
let bytes = match std::fs::read(path) { Ok(bytes) => bytes, Err(_) => return Ok(Vec::new()) };
|
|
131
102
|
process_file(rel, &bytes, matcher, before_context, after_context)
|
|
132
103
|
}
|
|
133
104
|
|
|
@@ -159,11 +130,7 @@ pub fn block_iter(opts: &RgOptions) -> Result<BlockIter, RgApiError> {
|
|
|
159
130
|
filters,
|
|
160
131
|
move |dent, root, filters, tx, cancel| match block_entry(dent, root, filters, &matcher, before_context, after_context, max_depth) {
|
|
161
132
|
Ok(blocks) => {
|
|
162
|
-
for block in blocks {
|
|
163
|
-
if cancel.load(Ordering::Relaxed) || tx.send(Ok(block)).is_err() {
|
|
164
|
-
return WalkState::Quit;
|
|
165
|
-
}
|
|
166
|
-
}
|
|
133
|
+
for block in blocks { if cancel.load(Ordering::Relaxed) || tx.send(Ok(block)).is_err() { return WalkState::Quit; } }
|
|
167
134
|
WalkState::Continue
|
|
168
135
|
}
|
|
169
136
|
Err(err) => {
|
|
@@ -14,26 +14,12 @@ pub use search::{MatchSpan, RgIter, RgOptions, SearchKind, SearchLine, compile_r
|
|
|
14
14
|
pub use walk::{FindIter, FindOptions, StreamIter, find, find_iter};
|
|
15
15
|
|
|
16
16
|
#[derive(Debug, Clone)]
|
|
17
|
-
pub struct RgApiError {
|
|
18
|
-
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
impl RgApiError {
|
|
22
|
-
pub(crate) fn new(msg: impl Into<String>) -> Self {
|
|
23
|
-
Self { msg: msg.into() }
|
|
24
|
-
}
|
|
25
|
-
}
|
|
26
|
-
|
|
27
|
-
impl std::fmt::Display for RgApiError {
|
|
28
|
-
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
|
29
|
-
write!(f, "{}", self.msg)
|
|
30
|
-
}
|
|
31
|
-
}
|
|
17
|
+
pub struct RgApiError { msg: String }
|
|
18
|
+
|
|
19
|
+
impl RgApiError { pub(crate) fn new(msg: impl Into<String>) -> Self { Self { msg: msg.into() } } }
|
|
20
|
+
|
|
21
|
+
impl std::fmt::Display for RgApiError { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { write!(f, "{}", self.msg) } }
|
|
32
22
|
|
|
33
23
|
impl std::error::Error for RgApiError {}
|
|
34
24
|
|
|
35
|
-
impl From<std::io::Error> for RgApiError {
|
|
36
|
-
fn from(err: std::io::Error) -> Self {
|
|
37
|
-
Self::new(err.to_string())
|
|
38
|
-
}
|
|
39
|
-
}
|
|
25
|
+
impl From<std::io::Error> for RgApiError { fn from(err: std::io::Error) -> Self { Self::new(err.to_string()) } }
|
|
@@ -49,10 +49,7 @@ pub struct NbCell {
|
|
|
49
49
|
// Lean notebook model: only the fields we search; outputs/metadata are skipped by serde
|
|
50
50
|
// without being allocated, which is the whole memory win over materializing the JSON in Python.
|
|
51
51
|
#[derive(Deserialize)]
|
|
52
|
-
struct RawNb {
|
|
53
|
-
#[serde(default)]
|
|
54
|
-
cells: Vec<RawCell>,
|
|
55
|
-
}
|
|
52
|
+
struct RawNb { #[serde(default)] cells: Vec<RawCell> }
|
|
56
53
|
|
|
57
54
|
#[derive(Deserialize)]
|
|
58
55
|
struct RawCell {
|
|
@@ -66,11 +63,7 @@ struct RawCell {
|
|
|
66
63
|
|
|
67
64
|
impl RawCell {
|
|
68
65
|
fn id_string(&self, index: usize) -> String {
|
|
69
|
-
match &self.id {
|
|
70
|
-
Some(serde_json::Value::String(s)) => s.clone(),
|
|
71
|
-
Some(v) => v.to_string(),
|
|
72
|
-
None => index.to_string(),
|
|
73
|
-
}
|
|
66
|
+
match &self.id { Some(serde_json::Value::String(s)) => s.clone(), Some(v) => v.to_string(), None => index.to_string() }
|
|
74
67
|
}
|
|
75
68
|
}
|
|
76
69
|
|
|
@@ -84,46 +77,25 @@ enum Source {
|
|
|
84
77
|
Empty,
|
|
85
78
|
}
|
|
86
79
|
|
|
87
|
-
impl Source {
|
|
88
|
-
fn text(&self) -> String {
|
|
89
|
-
match self {
|
|
90
|
-
Source::Lines(v) => v.concat(),
|
|
91
|
-
Source::Text(s) => s.clone(),
|
|
92
|
-
Source::Empty => String::new(),
|
|
93
|
-
}
|
|
94
|
-
}
|
|
95
|
-
}
|
|
80
|
+
impl Source { fn text(&self) -> String { match self { Source::Lines(v) => v.concat(), Source::Text(s) => s.clone(), Source::Empty => String::new() } } }
|
|
96
81
|
|
|
97
82
|
fn process_file(disp: String, bytes: &[u8], matcher: &RegexMatcher, cell_context: usize, multiline: bool) -> Result<Vec<NbCell>, RgApiError> {
|
|
98
83
|
// Not a parseable notebook (bad JSON, or JSON that isn't a notebook): skip, like a binary file.
|
|
99
|
-
let nb: RawNb = match serde_json::from_slice(bytes) {
|
|
100
|
-
Ok(nb) => nb,
|
|
101
|
-
Err(_) => return Ok(Vec::new()),
|
|
102
|
-
};
|
|
84
|
+
let nb: RawNb = match serde_json::from_slice(bytes) { Ok(nb) => nb, Err(_) => return Ok(Vec::new()) };
|
|
103
85
|
let n = nb.cells.len();
|
|
104
86
|
let mut info = Vec::with_capacity(n);
|
|
105
87
|
let mut matched: Vec<(usize, Vec<SearchLine>)> = Vec::new();
|
|
106
88
|
for (i, cell) in nb.cells.iter().enumerate() {
|
|
107
89
|
let src = cell.source.text();
|
|
108
90
|
let hits = search_text(disp.clone(), &src, matcher.clone(), 0, 0, multiline)?;
|
|
109
|
-
if !hits.is_empty() {
|
|
110
|
-
matched.push((i, hits));
|
|
111
|
-
}
|
|
91
|
+
if !hits.is_empty() { matched.push((i, hits)); }
|
|
112
92
|
info.push((cell.id_string(i), cell.cell_type.clone().unwrap_or_default(), src));
|
|
113
93
|
}
|
|
114
|
-
if matched.is_empty() {
|
|
115
|
-
return Ok(Vec::new());
|
|
116
|
-
}
|
|
94
|
+
if matched.is_empty() { return Ok(Vec::new()); }
|
|
117
95
|
let mut emit: BTreeMap<usize, bool> = BTreeMap::new(); // index -> is_match
|
|
118
|
-
for (i, _) in &matched {
|
|
119
|
-
emit.insert(*i, true);
|
|
120
|
-
}
|
|
96
|
+
for (i, _) in &matched { emit.insert(*i, true); }
|
|
121
97
|
if cell_context > 0 {
|
|
122
|
-
for (i, _) in &matched {
|
|
123
|
-
for j in i.saturating_sub(cell_context)..(i + cell_context + 1).min(n) {
|
|
124
|
-
emit.entry(j).or_insert(false);
|
|
125
|
-
}
|
|
126
|
-
}
|
|
98
|
+
for (i, _) in &matched { for j in i.saturating_sub(cell_context)..(i + cell_context + 1).min(n) { emit.entry(j).or_insert(false); } }
|
|
127
99
|
}
|
|
128
100
|
let mut matched: HashMap<usize, Vec<SearchLine>> = matched.into_iter().collect();
|
|
129
101
|
let mut out = Vec::with_capacity(emit.len());
|
|
@@ -139,9 +111,7 @@ fn compile_nb_regex(pattern: &str, case_sensitive: Option<bool>, smart_case: boo
|
|
|
139
111
|
compile_regex(pattern, case_sensitive, smart_case, multiline).map_err(|e| {
|
|
140
112
|
if !multiline && e.to_string().contains("not allowed in a regex") {
|
|
141
113
|
RgApiError::new(format!("{e}; pass multiline=True to let the pattern match across lines within a cell"))
|
|
142
|
-
} else {
|
|
143
|
-
e
|
|
144
|
-
}
|
|
114
|
+
} else { e }
|
|
145
115
|
})
|
|
146
116
|
}
|
|
147
117
|
|
|
@@ -155,10 +125,7 @@ pub fn nb_search_file(
|
|
|
155
125
|
multiline: bool,
|
|
156
126
|
) -> Result<Vec<NbCell>, RgApiError> {
|
|
157
127
|
let matcher = compile_nb_regex(pattern, case_sensitive, smart_case, multiline)?;
|
|
158
|
-
let bytes = match std::fs::read(path) {
|
|
159
|
-
Ok(b) => b,
|
|
160
|
-
Err(_) => return Ok(Vec::new()),
|
|
161
|
-
};
|
|
128
|
+
let bytes = match std::fs::read(path) { Ok(b) => b, Err(_) => return Ok(Vec::new()) };
|
|
162
129
|
process_file(display_path, &bytes, &matcher, cell_context, multiline)
|
|
163
130
|
}
|
|
164
131
|
|
|
@@ -171,25 +138,13 @@ fn nb_entry(
|
|
|
171
138
|
multiline: bool,
|
|
172
139
|
max_depth: Option<usize>,
|
|
173
140
|
) -> Result<Vec<NbCell>, RgApiError> {
|
|
174
|
-
let dent = match entry {
|
|
175
|
-
Ok(dent) => dent,
|
|
176
|
-
Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err),
|
|
177
|
-
};
|
|
141
|
+
let dent = match entry { Ok(dent) => dent, Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err) };
|
|
178
142
|
let path = dent.path();
|
|
179
|
-
let Some(ft) = dent.file_type() else {
|
|
180
|
-
|
|
181
|
-
};
|
|
182
|
-
if !ft.is_file() {
|
|
183
|
-
return Ok(Vec::new());
|
|
184
|
-
}
|
|
143
|
+
let Some(ft) = dent.file_type() else { return Ok(Vec::new()); };
|
|
144
|
+
if !ft.is_file() { return Ok(Vec::new()); }
|
|
185
145
|
let rel = rel_path(root, path);
|
|
186
|
-
if !filters.path_allowed(&rel) {
|
|
187
|
-
|
|
188
|
-
}
|
|
189
|
-
let bytes = match std::fs::read(path) {
|
|
190
|
-
Ok(b) => b,
|
|
191
|
-
Err(_) => return Ok(Vec::new()),
|
|
192
|
-
};
|
|
146
|
+
if !filters.path_allowed(&rel) { return Ok(Vec::new()); }
|
|
147
|
+
let bytes = match std::fs::read(path) { Ok(b) => b, Err(_) => return Ok(Vec::new()) };
|
|
193
148
|
process_file(rel, &bytes, matcher, cell_context, multiline)
|
|
194
149
|
}
|
|
195
150
|
|
|
@@ -223,11 +178,7 @@ pub fn nb_iter(opts: &NbOptions) -> Result<NbIter, RgApiError> {
|
|
|
223
178
|
filters,
|
|
224
179
|
move |dent, root, filters, tx, cancel| match nb_entry(dent, root, filters, &matcher, cell_context, multiline, max_depth) {
|
|
225
180
|
Ok(cells) => {
|
|
226
|
-
for cell in cells {
|
|
227
|
-
if cancel.load(Ordering::Relaxed) || tx.send(Ok(cell)).is_err() {
|
|
228
|
-
return WalkState::Quit;
|
|
229
|
-
}
|
|
230
|
-
}
|
|
181
|
+
for cell in cells { if cancel.load(Ordering::Relaxed) || tx.send(Ok(cell)).is_err() { return WalkState::Quit; } }
|
|
231
182
|
WalkState::Continue
|
|
232
183
|
}
|
|
233
184
|
Err(err) => {
|
|
@@ -238,6 +189,4 @@ pub fn nb_iter(opts: &NbOptions) -> Result<NbIter, RgApiError> {
|
|
|
238
189
|
))
|
|
239
190
|
}
|
|
240
191
|
|
|
241
|
-
pub fn nb_search(opts: &NbOptions) -> Result<Vec<NbCell>, RgApiError> {
|
|
242
|
-
nb_iter(opts)?.collect()
|
|
243
|
-
}
|
|
192
|
+
pub fn nb_search(opts: &NbOptions) -> Result<Vec<NbCell>, RgApiError> { nb_iter(opts)?.collect() }
|