rgapi 0.1.22__tar.gz → 0.1.24__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,7 +29,7 @@ jobs:
29
29
  - uses: actions/checkout@v7
30
30
  - uses: PyO3/maturin-action@v1
31
31
  with:
32
- args: --release --out dist -i python3.10 -i python3.11 -i python3.12 -i python3.13
32
+ args: --profile dist --out dist -i python3.10 -i python3.11 -i python3.12 -i python3.13
33
33
  manylinux: auto
34
34
  - uses: actions/upload-artifact@v7
35
35
  with:
@@ -58,6 +58,12 @@ jobs:
58
58
  contents: write
59
59
  steps:
60
60
  - uses: actions/checkout@v7
61
+ - uses: dtolnay/rust-toolchain@stable
62
+ - id: crates-auth
63
+ uses: rust-lang/crates-io-auth-action@v1
64
+ - run: cargo publish
65
+ env:
66
+ CARGO_REGISTRY_TOKEN: ${{ steps.crates-auth.outputs.token }}
61
67
  - uses: actions/download-artifact@v8
62
68
  with:
63
69
  path: dist
@@ -30,18 +30,18 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
30
30
 
31
31
  [[package]]
32
32
  name = "crc32fast"
33
- version = "1.5.0"
33
+ version = "1.5.1"
34
34
  source = "registry+https://github.com/rust-lang/crates.io-index"
35
- checksum = "9481c1c90cbf2ac953f07c8d4a58aa3945c425b7185c9154d67a65e4230da511"
35
+ checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550"
36
36
  dependencies = [
37
37
  "cfg-if",
38
38
  ]
39
39
 
40
40
  [[package]]
41
41
  name = "crossbeam-deque"
42
- version = "0.8.7"
42
+ version = "0.8.8"
43
43
  source = "registry+https://github.com/rust-lang/crates.io-index"
44
- checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb"
44
+ checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a"
45
45
  dependencies = [
46
46
  "crossbeam-epoch",
47
47
  "crossbeam-utils",
@@ -49,18 +49,18 @@ dependencies = [
49
49
 
50
50
  [[package]]
51
51
  name = "crossbeam-epoch"
52
- version = "0.9.20"
52
+ version = "0.9.21"
53
53
  source = "registry+https://github.com/rust-lang/crates.io-index"
54
- checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f"
54
+ checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d"
55
55
  dependencies = [
56
56
  "crossbeam-utils",
57
57
  ]
58
58
 
59
59
  [[package]]
60
60
  name = "crossbeam-utils"
61
- version = "0.8.22"
61
+ version = "0.8.23"
62
62
  source = "registry+https://github.com/rust-lang/crates.io-index"
63
- checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17"
63
+ checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
64
64
 
65
65
  [[package]]
66
66
  name = "encoding_rs"
@@ -166,9 +166,9 @@ checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
166
166
 
167
167
  [[package]]
168
168
  name = "log"
169
- version = "0.4.33"
169
+ version = "0.4.34"
170
170
  source = "registry+https://github.com/rust-lang/crates.io-index"
171
- checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
171
+ checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6"
172
172
 
173
173
  [[package]]
174
174
  name = "memchr"
@@ -291,7 +291,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
291
291
 
292
292
  [[package]]
293
293
  name = "rgapi"
294
- version = "0.1.22"
294
+ version = "0.1.24"
295
295
  dependencies = [
296
296
  "crc32fast",
297
297
  "globset",
@@ -340,7 +340,7 @@ checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
340
340
  dependencies = [
341
341
  "proc-macro2",
342
342
  "quote",
343
- "syn 3.0.3",
343
+ "syn 3.0.5",
344
344
  ]
345
345
 
346
346
  [[package]]
@@ -369,9 +369,9 @@ dependencies = [
369
369
 
370
370
  [[package]]
371
371
  name = "syn"
372
- version = "3.0.3"
372
+ version = "3.0.5"
373
373
  source = "registry+https://github.com/rust-lang/crates.io-index"
374
- checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3"
374
+ checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9"
375
375
  dependencies = [
376
376
  "proc-macro2",
377
377
  "quote",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "rgapi"
3
- version = "0.1.22"
3
+ version = "0.1.24"
4
4
  edition = "2024"
5
5
  rust-version = "1.91"
6
6
  license = "Apache-2.0"
@@ -31,3 +31,20 @@ extension-module = ["python", "pyo3/extension-module"]
31
31
 
32
32
  [lints.clippy]
33
33
  too_many_arguments = "allow"
34
+
35
+ [profile.release]
36
+ lto = false
37
+ codegen-units = 16
38
+
39
+ [profile.release.package.rgapi]
40
+ incremental = true
41
+
42
+ [profile.dist]
43
+ inherits = "release"
44
+ lto = true
45
+ incremental = false
46
+ codegen-units = 1
47
+ strip = true
48
+
49
+ [profile.dist.package.rgapi]
50
+ incremental = false
@@ -9,7 +9,7 @@ src/walk.rs ignore/globset/grep-regex-backed path walking and filtering
9
9
  src/search.rs grep-regex/grep-searcher-backed searching
10
10
  src/block.rs blank-line-delimited block grouping, matching, and block context
11
11
  src/python.rs PyO3 classes and private core functions
12
- python/rgapi/ public Python wrappers over `rgapi._core`
12
+ python/rgapi/ public Python wrappers over `rgapi._core`, plus the `rgapi-nbrg` CLI
13
13
  tests/ pytest coverage for the Python API
14
14
  ```
15
15
 
@@ -34,14 +34,13 @@ Release flow is: release first, then bump - `ship-release` does both.
34
34
  2. Confirm the release version in `Cargo.toml` (`[package].version`).
35
35
  3. Run `ship-release`. It tags `v<version>`, pushes branch and tag (CI builds and publishes), then bumps `Cargo.toml`, refreshes the editable install, and pushes the bump without a tag.
36
36
 
37
- The GitHub workflow builds wheels for Python 3.10-3.13 on Linux and macOS and publishes artifacts to GitHub Releases and PyPI when a `v*` tag is pushed.
37
+ The GitHub workflow builds wheels for Python 3.10-3.13 on Linux and macOS and publishes the Rust crate, GitHub release artifacts, and PyPI package when a `v*` tag is pushed.
38
38
 
39
39
  ## Design notes
40
40
 
41
41
  Paths in `fd`, `walk`, `rg`, and `rg_iter` results are relative to the requested root and use `/` separators. Traversal uses `ignore::WalkParallel`, so result order is not part of the API contract. Search results are structured rows; collected result lists use rg-style `str()` and notebook display. `SearchLine.lnhash` is computed with the same CRC-32-based line-content hash format as exhash (`lineno|hash|`, low 16 bits of CRC-32 over the line's UTF-8 bytes); `lnhashs=True` only changes row display, not `line_number` or matching behavior. Path regexes filter returned/searched paths; `skip_dir` and `skip_dir_re` prune traversal through `ignore::WalkBuilder::filter_entry`. Depth, size, symlink, filesystem, hidden, and ignore options are direct `ignore::WalkBuilder` settings. `rg_iter` exposes the same parallel search stream that `rg` collects by default; `paths=True` and `count=True` consume that stream with different reducers. Binary files and invalid UTF-8 are skipped for now.
42
42
 
43
- Streaming engine: `walk.rs` owns the generic machinery. `StreamIter<T>` is the worker-thread-plus-bounded-channel iterator (`sync_channel(8192)`, so producers block rather than buffer without limit when a consumer lags), and `spawn_walk` owns the shared scaffold: walker config, panic catching, cancel flag, and worker thread. `rg_iter` (`T = SearchLine`), `block_iter` (`T = SearchBlock`), `nb_iter` (`T = NbCell`), and `find_iter` (`T = String`, the path walk) plug entry closures into that engine. Block search reads each file once, searches it once, groups nonblank lines into blocks, maps matching lines to their blocks, and expands context by block index.
44
- Each `SearchBlock` carries numeric boundaries plus hashes for its first and last source lines. Python keeps both and chooses the displayed address without another file read.
43
+ Streaming engine: `walk.rs` owns the generic machinery. `StreamIter<T>` is the worker-thread-plus-bounded-channel iterator (`sync_channel(8192)`, so producers block rather than buffer without limit when a consumer lags), and `spawn_walk` owns the shared scaffold: walker config, panic catching, cancel flag, and worker thread. `rg_iter` (`T = SearchLine`), `block_iter` (`T = SearchBlock`), `nb_iter` (`T = NbCell`), and `find_iter` (`T = String`, the path walk) plug entry closures into that engine. Block search reads each file once, searches it once, groups nonblank lines into blocks, maps matching lines to their blocks, and expands context by block index. Each `SearchBlock` carries numeric boundaries plus hashes for its first and last source lines. Python keeps both and chooses the displayed address without another file read.
45
44
 
46
45
  Async API: `fda`, `fda_iter`, `rga`, `rga_iter`, `nbrga`, and `nbrga_iter` wrap the corresponding private core operations. `rga(summary=True)` uses `_core.block_search_async`; ordinary `rga` uses `_core.rg_async`. Each collected core function takes a Python callback, runs on Rust threads through the generic `stream_async` helper, and delivers with one GIL attach at the end. Iterator forms use `stream_iter_async` and attach once per batch. The Python side settles an `asyncio.Future` or feeds an `asyncio.Queue` via `loop.call_soon_threadsafe`; no Python thread blocks and `asyncio.to_thread` is not involved. `AsyncHandle.cancel()` sets the same atomic flag used by the Rust iterators.
47
46
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rgapi
3
- Version: 0.1.22
3
+ Version: 0.1.24
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: Implementation :: CPython
6
6
  Requires-Dist: fastcore>=1.14.6
@@ -77,8 +77,7 @@ pip install rgapi
77
77
 
78
78
  ## Semantics
79
79
 
80
- `fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters.
81
- `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
80
+ `fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
82
81
 
83
82
  `fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
84
83
 
@@ -154,6 +153,16 @@ matches list of SearchLine rows for the matched lines within the cell
154
153
 
155
154
  Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
156
155
 
156
+ `rgapi-nbrg` exposes notebook search without requiring a Python kernel:
157
+
158
+ ```bash
159
+ rgapi-nbrg 'read_csv' .
160
+ rgapi-nbrg 'read_csv' . --cell-context 1
161
+ rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
162
+ ```
163
+
164
+ Run `rgapi-nbrg --help` for its discovery, matching, and output options.
165
+
157
166
  ## Async
158
167
 
159
168
  `fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
@@ -56,8 +56,7 @@ pip install rgapi
56
56
 
57
57
  ## Semantics
58
58
 
59
- `fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters.
60
- `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
59
+ `fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
61
60
 
62
61
  `fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
63
62
 
@@ -133,6 +132,16 @@ matches list of SearchLine rows for the matched lines within the cell
133
132
 
134
133
  Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
135
134
 
135
+ `rgapi-nbrg` exposes notebook search without requiring a Python kernel:
136
+
137
+ ```bash
138
+ rgapi-nbrg 'read_csv' .
139
+ rgapi-nbrg 'read_csv' . --cell-context 1
140
+ rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
141
+ ```
142
+
143
+ Run `rgapi-nbrg --help` for its discovery, matching, and output options.
144
+
136
145
  ## Async
137
146
 
138
147
  `fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
@@ -30,6 +30,9 @@ rgapi = "rgapi.skill"
30
30
  [project.entry-points.fastaudit_safe_native]
31
31
  rgapi = "rgapi"
32
32
 
33
+ [project.scripts]
34
+ rgapi-nbrg = "rgapi._cli:nbrg_cli"
35
+
33
36
  [tool.maturin]
34
37
  features = ["extension-module"]
35
38
  python-source = "python"
@@ -0,0 +1,28 @@
1
+ "Command-line access to notebook-aware search."
2
+
3
+ from fastcore.script import call_parse
4
+ from . import nbrg
5
+
6
+
7
+ @call_parse(pos=['root'])
8
+ def nbrg_cli(
9
+ pattern:str, # Regex pattern to search for
10
+ root:str='.', # File or directory to search
11
+ cell_context:int=0, # Neighbouring cells to include before and after matches
12
+ multiline:bool=False, # Allow matches across lines within a cell?
13
+ smart_case:bool=False, # Use case-sensitive matching when the pattern contains uppercase?
14
+ case:bool=False, # Force case-sensitive matching?
15
+ paths:bool=False, # Return only matching notebook paths?
16
+ count:bool=False, # Return the number of matching cells?
17
+ max_results:int=None, # Maximum matching cells to return
18
+ maxlen:int=180, # Maximum source characters displayed per cell
19
+ glob:str=None, # Glob selecting notebook paths
20
+ exclude:str=None, # Glob excluding notebook paths
21
+ hidden:bool=False, # Search hidden files and directories?
22
+ max_depth:int=None, # Maximum directory depth to search
23
+ timeout_ms:int=None, # Stop searching after this many milliseconds
24
+ ):
25
+ "Search notebook cell sources and report stable cell IDs."
26
+ print(nbrg(pattern, root, cell_context=cell_context, multiline=multiline, smart_case=smart_case,
27
+ case_sensitive=True if case else None, paths=paths, count=count, max_results=max_results, maxlen=maxlen,
28
+ glob=glob, exclude=exclude, hidden=hidden, max_depth=max_depth, timeout_ms=timeout_ms))
@@ -0,0 +1,35 @@
1
+ """Find files, search text, and search notebook cell sources from Python with ripgrep semantics and structured results. Use for fd-style discovery, regex searches, and notebook searches returning stable cell IDs rather than escaped JSON.
2
+
3
+ Uses ripgrep's `ignore`, `grep-regex`, and `grep-searcher` crates, including ignore files, hidden-file handling, globs, extensions, and regex semantics. Prefer these APIs to shell parsing or manual file scans in Python; use `rgstr` for held text instead of split-line loops, and `ls`/`fd` for kernel-side listings.
4
+
5
+ For orientation, start with `rg(summary=True)`; use line-level results where needed and `lnhashs=True` when edits may follow. Summary blocks suit prose/config paragraphs as well as code. Display results bare; narrow oversized results with API parameters rather than joining, slicing, or reformatting them.
6
+
7
+ ## Search units
8
+
9
+ - `rg`: lines, or `SearchBlock` rows with `summary=True`. Blank/whitespace-only lines separate blocks; multiple matches in one block yield one row. Context counts the selected unit. Summary mode cannot combine with `paths` or `count`, but supports hashed block boundaries.
10
+ - `nbrg`: `NbResults` of `NbCell` rows from source only, never metadata/outputs. `multiline=True` matches across cell lines while `^`/`$` remain line anchors; ordinary line-oriented `rg` rejects newline patterns.
11
+ - Traversal/search run in parallel in Rust; sort when stable order is required. `path_re`/`skip_path_re` filter paths without pruning; `skip_dir`/`skip_dir_re` prune subtrees.
12
+
13
+ ## Result fields and display
14
+
15
+ `FileEntry` is a slash-separated relative-path `str` with lazy `size`/`mtime`/`is_dir`/`stat`. Path lists render as ls-style tables capped at `MAX_REPR`; `str(res)`/`list(res)` yield plain paths. Unfollowed symlinks remain, marked `l`; `link_target` is their target or `None` for non-links. `ls(hidden=True)` corresponds to `ls -a`.
16
+
17
+ Search rows provide `asdict()`. All paths are relative to the search root:
18
+
19
+ | Row | Location | Content/matches | `kind` |
20
+ |---|---|---|---|
21
+ | `SearchLine` | `path`, 1-based `line_number`, `lnhash` | `line` without trailing newline; `matches` as byte-offset `(start, end)` pairs on match rows | match/before/after/context |
22
+ | `SearchBlock` | `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash` | full `source`; `matches` as matching `SearchLine`s | match/context |
23
+ | `NbCell` | `path`, `cell_index`, `cell_id`, `cell_type` (code/markdown/raw) | full `source`; `matches` as matching `SearchLine`s | match/context |
24
+
25
+ Block displays use `path:start-end:source`; notebook displays use `path:cell_id:source`. Context replaces the last separator colon with `-`. Hashed block locations are `start_lnhash,end_lnhash`, or one hash for a single line. Newline runs display as ¶, retaining indentation; `maxlen` limits displayed source, not stored `source`.
26
+
27
+ Notebook match previews start at the first matched line, marking omitted earlier lines as `…[Ln]` (1-based), but retain a leading directive: `#| export…[L4]needle here`. `cell_context` counts neighbouring cells. Consult individual function docs for parameters and reduction modes.
28
+ """
29
+
30
+ from . import RgIter, fd, ls, nbrg, rg, rg_iter, rgstr
31
+
32
+ __all__ = [ "RgIter", "fd", "ls", "rg", "rg_iter", "nbrg", "rgstr" ]
33
+
34
+ __pyskill_params__ = {'walk_params': ('glob', 'include', 'exclude', 'hidden', 'min_depth', 'max_filesize',
35
+ 'follow_links', 'same_file_system', 'path_re', 'skip_path_re', 'skip_dir', 'skip_dir_re')}
@@ -1,3 +1,4 @@
1
1
  max_width = 160
2
2
  use_small_heuristics = "Max"
3
3
  use_field_init_shorthand = true
4
+ disable_all_formatting = true
@@ -23,11 +23,7 @@ pub struct SearchBlock {
23
23
  pub matches: Vec<SearchLine>,
24
24
  }
25
25
 
26
- struct BlockInfo {
27
- start_line: u64,
28
- end_line: u64,
29
- source: String,
30
- }
26
+ struct BlockInfo { start_line: u64, end_line: u64, source: String }
31
27
 
32
28
  fn split_blocks(text: &str) -> Vec<BlockInfo> {
33
29
  let mut blocks = Vec::new();
@@ -41,31 +37,20 @@ fn split_blocks(text: &str) -> Vec<BlockInfo> {
41
37
  lines.clear();
42
38
  }
43
39
  } else {
44
- if start.is_none() {
45
- start = Some(line_no);
46
- }
40
+ if start.is_none() { start = Some(line_no); }
47
41
  lines.push(line);
48
42
  }
49
43
  }
50
- if let Some(first) = start {
51
- blocks.push(BlockInfo { start_line: first, end_line: text.lines().count() as u64, source: lines.join("\n") });
52
- }
44
+ if let Some(first) = start { blocks.push(BlockInfo { start_line: first, end_line: text.lines().count() as u64, source: lines.join("\n") }); }
53
45
  blocks
54
46
  }
55
47
 
56
48
  fn process_file(disp: String, bytes: &[u8], matcher: &RegexMatcher, before_context: usize, after_context: usize) -> Result<Vec<SearchBlock>, RgApiError> {
57
- if bytes.contains(&0) {
58
- return Ok(Vec::new());
59
- }
60
- let text = match std::str::from_utf8(bytes) {
61
- Ok(text) => text,
62
- Err(_) => return Ok(Vec::new()),
63
- };
49
+ if bytes.contains(&0) { return Ok(Vec::new()); }
50
+ let text = match std::str::from_utf8(bytes) { Ok(text) => text, Err(_) => return Ok(Vec::new()) };
64
51
  let blocks = split_blocks(text);
65
52
  let hits = search_text(disp.clone(), text, matcher.clone(), 0, 0, false)?;
66
- if hits.is_empty() {
67
- return Ok(Vec::new());
68
- }
53
+ if hits.is_empty() { return Ok(Vec::new()); }
69
54
  let mut matched: HashMap<usize, Vec<SearchLine>> = HashMap::new();
70
55
  for hit in hits {
71
56
  if let Some((i, _)) = blocks.iter().enumerate().find(|(_, b)| b.start_line <= hit.line_number && hit.line_number <= b.end_line) {
@@ -77,9 +62,7 @@ fn process_file(disp: String, bytes: &[u8], matcher: &RegexMatcher, before_conte
77
62
  emit.insert(*i, true);
78
63
  let start = i.saturating_sub(before_context);
79
64
  let end = (i + after_context + 1).min(blocks.len());
80
- for j in start..end {
81
- emit.entry(j).or_insert(false);
82
- }
65
+ for j in start..end { emit.entry(j).or_insert(false); }
83
66
  }
84
67
  Ok(emit
85
68
  .into_iter()
@@ -109,25 +92,13 @@ fn block_entry(
109
92
  after_context: usize,
110
93
  max_depth: Option<usize>,
111
94
  ) -> Result<Vec<SearchBlock>, RgApiError> {
112
- let dent = match entry {
113
- Ok(dent) => dent,
114
- Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err),
115
- };
95
+ let dent = match entry { Ok(dent) => dent, Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err) };
116
96
  let path = dent.path();
117
- let Some(ft) = dent.file_type() else {
118
- return Ok(Vec::new());
119
- };
120
- if !ft.is_file() {
121
- return Ok(Vec::new());
122
- }
97
+ let Some(ft) = dent.file_type() else { return Ok(Vec::new()); };
98
+ if !ft.is_file() { return Ok(Vec::new()); }
123
99
  let rel = rel_path(root, path);
124
- if !filters.path_allowed(&rel) {
125
- return Ok(Vec::new());
126
- }
127
- let bytes = match std::fs::read(path) {
128
- Ok(bytes) => bytes,
129
- Err(_) => return Ok(Vec::new()),
130
- };
100
+ if !filters.path_allowed(&rel) { return Ok(Vec::new()); }
101
+ let bytes = match std::fs::read(path) { Ok(bytes) => bytes, Err(_) => return Ok(Vec::new()) };
131
102
  process_file(rel, &bytes, matcher, before_context, after_context)
132
103
  }
133
104
 
@@ -159,11 +130,7 @@ pub fn block_iter(opts: &RgOptions) -> Result<BlockIter, RgApiError> {
159
130
  filters,
160
131
  move |dent, root, filters, tx, cancel| match block_entry(dent, root, filters, &matcher, before_context, after_context, max_depth) {
161
132
  Ok(blocks) => {
162
- for block in blocks {
163
- if cancel.load(Ordering::Relaxed) || tx.send(Ok(block)).is_err() {
164
- return WalkState::Quit;
165
- }
166
- }
133
+ for block in blocks { if cancel.load(Ordering::Relaxed) || tx.send(Ok(block)).is_err() { return WalkState::Quit; } }
167
134
  WalkState::Continue
168
135
  }
169
136
  Err(err) => {
@@ -14,26 +14,12 @@ pub use search::{MatchSpan, RgIter, RgOptions, SearchKind, SearchLine, compile_r
14
14
  pub use walk::{FindIter, FindOptions, StreamIter, find, find_iter};
15
15
 
16
16
  #[derive(Debug, Clone)]
17
- pub struct RgApiError {
18
- msg: String,
19
- }
20
-
21
- impl RgApiError {
22
- pub(crate) fn new(msg: impl Into<String>) -> Self {
23
- Self { msg: msg.into() }
24
- }
25
- }
26
-
27
- impl std::fmt::Display for RgApiError {
28
- fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
29
- write!(f, "{}", self.msg)
30
- }
31
- }
17
+ pub struct RgApiError { msg: String }
18
+
19
+ impl RgApiError { pub(crate) fn new(msg: impl Into<String>) -> Self { Self { msg: msg.into() } } }
20
+
21
+ impl std::fmt::Display for RgApiError { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { write!(f, "{}", self.msg) } }
32
22
 
33
23
  impl std::error::Error for RgApiError {}
34
24
 
35
- impl From<std::io::Error> for RgApiError {
36
- fn from(err: std::io::Error) -> Self {
37
- Self::new(err.to_string())
38
- }
39
- }
25
+ impl From<std::io::Error> for RgApiError { fn from(err: std::io::Error) -> Self { Self::new(err.to_string()) } }
@@ -49,10 +49,7 @@ pub struct NbCell {
49
49
  // Lean notebook model: only the fields we search; outputs/metadata are skipped by serde
50
50
  // without being allocated, which is the whole memory win over materializing the JSON in Python.
51
51
  #[derive(Deserialize)]
52
- struct RawNb {
53
- #[serde(default)]
54
- cells: Vec<RawCell>,
55
- }
52
+ struct RawNb { #[serde(default)] cells: Vec<RawCell> }
56
53
 
57
54
  #[derive(Deserialize)]
58
55
  struct RawCell {
@@ -66,11 +63,7 @@ struct RawCell {
66
63
 
67
64
  impl RawCell {
68
65
  fn id_string(&self, index: usize) -> String {
69
- match &self.id {
70
- Some(serde_json::Value::String(s)) => s.clone(),
71
- Some(v) => v.to_string(),
72
- None => index.to_string(),
73
- }
66
+ match &self.id { Some(serde_json::Value::String(s)) => s.clone(), Some(v) => v.to_string(), None => index.to_string() }
74
67
  }
75
68
  }
76
69
 
@@ -84,46 +77,25 @@ enum Source {
84
77
  Empty,
85
78
  }
86
79
 
87
- impl Source {
88
- fn text(&self) -> String {
89
- match self {
90
- Source::Lines(v) => v.concat(),
91
- Source::Text(s) => s.clone(),
92
- Source::Empty => String::new(),
93
- }
94
- }
95
- }
80
+ impl Source { fn text(&self) -> String { match self { Source::Lines(v) => v.concat(), Source::Text(s) => s.clone(), Source::Empty => String::new() } } }
96
81
 
97
82
  fn process_file(disp: String, bytes: &[u8], matcher: &RegexMatcher, cell_context: usize, multiline: bool) -> Result<Vec<NbCell>, RgApiError> {
98
83
  // Not a parseable notebook (bad JSON, or JSON that isn't a notebook): skip, like a binary file.
99
- let nb: RawNb = match serde_json::from_slice(bytes) {
100
- Ok(nb) => nb,
101
- Err(_) => return Ok(Vec::new()),
102
- };
84
+ let nb: RawNb = match serde_json::from_slice(bytes) { Ok(nb) => nb, Err(_) => return Ok(Vec::new()) };
103
85
  let n = nb.cells.len();
104
86
  let mut info = Vec::with_capacity(n);
105
87
  let mut matched: Vec<(usize, Vec<SearchLine>)> = Vec::new();
106
88
  for (i, cell) in nb.cells.iter().enumerate() {
107
89
  let src = cell.source.text();
108
90
  let hits = search_text(disp.clone(), &src, matcher.clone(), 0, 0, multiline)?;
109
- if !hits.is_empty() {
110
- matched.push((i, hits));
111
- }
91
+ if !hits.is_empty() { matched.push((i, hits)); }
112
92
  info.push((cell.id_string(i), cell.cell_type.clone().unwrap_or_default(), src));
113
93
  }
114
- if matched.is_empty() {
115
- return Ok(Vec::new());
116
- }
94
+ if matched.is_empty() { return Ok(Vec::new()); }
117
95
  let mut emit: BTreeMap<usize, bool> = BTreeMap::new(); // index -> is_match
118
- for (i, _) in &matched {
119
- emit.insert(*i, true);
120
- }
96
+ for (i, _) in &matched { emit.insert(*i, true); }
121
97
  if cell_context > 0 {
122
- for (i, _) in &matched {
123
- for j in i.saturating_sub(cell_context)..(i + cell_context + 1).min(n) {
124
- emit.entry(j).or_insert(false);
125
- }
126
- }
98
+ for (i, _) in &matched { for j in i.saturating_sub(cell_context)..(i + cell_context + 1).min(n) { emit.entry(j).or_insert(false); } }
127
99
  }
128
100
  let mut matched: HashMap<usize, Vec<SearchLine>> = matched.into_iter().collect();
129
101
  let mut out = Vec::with_capacity(emit.len());
@@ -139,9 +111,7 @@ fn compile_nb_regex(pattern: &str, case_sensitive: Option<bool>, smart_case: boo
139
111
  compile_regex(pattern, case_sensitive, smart_case, multiline).map_err(|e| {
140
112
  if !multiline && e.to_string().contains("not allowed in a regex") {
141
113
  RgApiError::new(format!("{e}; pass multiline=True to let the pattern match across lines within a cell"))
142
- } else {
143
- e
144
- }
114
+ } else { e }
145
115
  })
146
116
  }
147
117
 
@@ -155,10 +125,7 @@ pub fn nb_search_file(
155
125
  multiline: bool,
156
126
  ) -> Result<Vec<NbCell>, RgApiError> {
157
127
  let matcher = compile_nb_regex(pattern, case_sensitive, smart_case, multiline)?;
158
- let bytes = match std::fs::read(path) {
159
- Ok(b) => b,
160
- Err(_) => return Ok(Vec::new()),
161
- };
128
+ let bytes = match std::fs::read(path) { Ok(b) => b, Err(_) => return Ok(Vec::new()) };
162
129
  process_file(display_path, &bytes, &matcher, cell_context, multiline)
163
130
  }
164
131
 
@@ -171,25 +138,13 @@ fn nb_entry(
171
138
  multiline: bool,
172
139
  max_depth: Option<usize>,
173
140
  ) -> Result<Vec<NbCell>, RgApiError> {
174
- let dent = match entry {
175
- Ok(dent) => dent,
176
- Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err),
177
- };
141
+ let dent = match entry { Ok(dent) => dent, Err(err) => return entry_err(err, max_depth).map_or(Ok(Vec::new()), Err) };
178
142
  let path = dent.path();
179
- let Some(ft) = dent.file_type() else {
180
- return Ok(Vec::new());
181
- };
182
- if !ft.is_file() {
183
- return Ok(Vec::new());
184
- }
143
+ let Some(ft) = dent.file_type() else { return Ok(Vec::new()); };
144
+ if !ft.is_file() { return Ok(Vec::new()); }
185
145
  let rel = rel_path(root, path);
186
- if !filters.path_allowed(&rel) {
187
- return Ok(Vec::new());
188
- }
189
- let bytes = match std::fs::read(path) {
190
- Ok(b) => b,
191
- Err(_) => return Ok(Vec::new()),
192
- };
146
+ if !filters.path_allowed(&rel) { return Ok(Vec::new()); }
147
+ let bytes = match std::fs::read(path) { Ok(b) => b, Err(_) => return Ok(Vec::new()) };
193
148
  process_file(rel, &bytes, matcher, cell_context, multiline)
194
149
  }
195
150
 
@@ -223,11 +178,7 @@ pub fn nb_iter(opts: &NbOptions) -> Result<NbIter, RgApiError> {
223
178
  filters,
224
179
  move |dent, root, filters, tx, cancel| match nb_entry(dent, root, filters, &matcher, cell_context, multiline, max_depth) {
225
180
  Ok(cells) => {
226
- for cell in cells {
227
- if cancel.load(Ordering::Relaxed) || tx.send(Ok(cell)).is_err() {
228
- return WalkState::Quit;
229
- }
230
- }
181
+ for cell in cells { if cancel.load(Ordering::Relaxed) || tx.send(Ok(cell)).is_err() { return WalkState::Quit; } }
231
182
  WalkState::Continue
232
183
  }
233
184
  Err(err) => {
@@ -238,6 +189,4 @@ pub fn nb_iter(opts: &NbOptions) -> Result<NbIter, RgApiError> {
238
189
  ))
239
190
  }
240
191
 
241
- pub fn nb_search(opts: &NbOptions) -> Result<Vec<NbCell>, RgApiError> {
242
- nb_iter(opts)?.collect()
243
- }
192
+ pub fn nb_search(opts: &NbOptions) -> Result<Vec<NbCell>, RgApiError> { nb_iter(opts)?.collect() }