rgapi 0.1.24__tar.gz → 0.1.26__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,11 +28,17 @@ version = "1.0.4"
28
28
  source = "registry+https://github.com/rust-lang/crates.io-index"
29
29
  checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
30
30
 
31
+ [[package]]
32
+ name = "core_detect"
33
+ version = "1.0.0"
34
+ source = "registry+https://github.com/rust-lang/crates.io-index"
35
+ checksum = "7f8f80099a98041a3d1622845c271458a2d73e688351bf3cb999266764b81d48"
36
+
31
37
  [[package]]
32
38
  name = "crc32fast"
33
- version = "1.5.1"
39
+ version = "1.5.2"
34
40
  source = "registry+https://github.com/rust-lang/crates.io-index"
35
- checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550"
41
+ checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
36
42
  dependencies = [
37
43
  "cfg-if",
38
44
  ]
@@ -64,11 +70,17 @@ checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
64
70
 
65
71
  [[package]]
66
72
  name = "encoding_rs"
67
- version = "0.8.35"
73
+ version = "0.8.41"
68
74
  source = "registry+https://github.com/rust-lang/crates.io-index"
69
- checksum = "75030f3c4f45dafd7586dd6780965a8c7e8e285a5ecb86713e63a79c5b2766f3"
75
+ checksum = "7b5ef0006ac9ab233c38522f5ae99cae3625151de8f706cacee1cba4b8e2832a"
70
76
  dependencies = [
71
77
  "cfg-if",
78
+ "core_detect",
79
+ "multiversion",
80
+ "multiversion_no_op",
81
+ "rustversion",
82
+ "scopeguard",
83
+ "simdutf8",
72
84
  ]
73
85
 
74
86
  [[package]]
@@ -185,6 +197,33 @@ dependencies = [
185
197
  "libc",
186
198
  ]
187
199
 
200
+ [[package]]
201
+ name = "multiversion"
202
+ version = "0.9.0"
203
+ source = "registry+https://github.com/rust-lang/crates.io-index"
204
+ checksum = "b4ca4bea16ffc3f443cf7d866912118196bfef4c6a1556ca00f9f9b00bb43f7c"
205
+ dependencies = [
206
+ "multiversion-macros",
207
+ ]
208
+
209
+ [[package]]
210
+ name = "multiversion-macros"
211
+ version = "0.9.0"
212
+ source = "registry+https://github.com/rust-lang/crates.io-index"
213
+ checksum = "0d416831a7317ef4b08bee00b69cbbb9c8763da7959a7026244d6266869f9c83"
214
+ dependencies = [
215
+ "proc-macro2",
216
+ "quote",
217
+ "rustversion",
218
+ "syn 3.0.5",
219
+ ]
220
+
221
+ [[package]]
222
+ name = "multiversion_no_op"
223
+ version = "1.0.0"
224
+ source = "registry+https://github.com/rust-lang/crates.io-index"
225
+ checksum = "743fb55ba31b18fb1ecef6bdc9aa2743314978ac084044301a7eee33fb99a20d"
226
+
188
227
  [[package]]
189
228
  name = "once_cell"
190
229
  version = "1.21.4"
@@ -291,7 +330,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
291
330
 
292
331
  [[package]]
293
332
  name = "rgapi"
294
- version = "0.1.24"
333
+ version = "0.1.26"
295
334
  dependencies = [
296
335
  "crc32fast",
297
336
  "globset",
@@ -304,6 +343,12 @@ dependencies = [
304
343
  "serde_json",
305
344
  ]
306
345
 
346
+ [[package]]
347
+ name = "rustversion"
348
+ version = "1.0.23"
349
+ source = "registry+https://github.com/rust-lang/crates.io-index"
350
+ checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f"
351
+
307
352
  [[package]]
308
353
  name = "same-file"
309
354
  version = "1.0.6"
@@ -313,6 +358,12 @@ dependencies = [
313
358
  "winapi-util",
314
359
  ]
315
360
 
361
+ [[package]]
362
+ name = "scopeguard"
363
+ version = "1.2.0"
364
+ source = "registry+https://github.com/rust-lang/crates.io-index"
365
+ checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
366
+
316
367
  [[package]]
317
368
  name = "serde"
318
369
  version = "1.0.229"
@@ -356,6 +407,12 @@ dependencies = [
356
407
  "zmij",
357
408
  ]
358
409
 
410
+ [[package]]
411
+ name = "simdutf8"
412
+ version = "0.1.5"
413
+ source = "registry+https://github.com/rust-lang/crates.io-index"
414
+ checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
415
+
359
416
  [[package]]
360
417
  name = "syn"
361
418
  version = "2.0.119"
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "rgapi"
3
- version = "0.1.24"
3
+ version = "0.1.26"
4
4
  edition = "2024"
5
5
  rust-version = "1.91"
6
6
  license = "Apache-2.0"
@@ -48,4 +48,8 @@ Truncation is recorded on collected results: `max_results` sets `stop_reason="ma
48
48
 
49
49
  Path results are `FileEntry` rows: a `str` subclass carrying the walk root, so paths stay plain strings for compatibility while stat info loads lazily (one cached `os.lstat` per entry, read only on attribute access). The wrapping happens at result construction on the Python side; Rust still streams plain strings. `PathResults.__repr__` shows an `ls -l`-style listing capped at `MAX_REPR` rows, so a huge result never stats everything, while `str()` stays one plain path per line. `ls` is `fd` with shell-style defaults (one level, dirs, ignore rules off), re-sorted with `stop_reason` preserved.
50
50
 
51
+ Rust callers can consume a `StreamIter` with `cancel_and_join()` to cancel, drain queued results, and wait for the walk's workers to finish. `Drop` remains nonblocking. Since filesystem calls already in progress must return before joining completes, keep `cancel_and_join()` off async executors.
52
+
53
+ Notebook hierarchy uses `heading_level(source)`, `section_range(levels, idx)` and `ancestor_indices(levels, idx)`. Callers supply zero for non-heading cells. Heading detection skips blank lines and lines starting with `#|`, then checks the first remaining line against `^#{1,6} \w`. It does not search past ordinary text. Section ranges include the addressed cell and end before the next equal-or-higher heading; non-headings select themselves. Ancestors exclude the addressed cell and are returned outermost first. These are calculations over the supplied levels, without retained outline state. Rustygate uses them for its cell selectors.
54
+
51
55
  This package intentionally has no CLI. Python is the interface.
rgapi-0.1.26/PKG-INFO ADDED
@@ -0,0 +1,255 @@
1
+ Metadata-Version: 2.4
2
+ Name: rgapi
3
+ Version: 0.1.26
4
+ Classifier: Programming Language :: Rust
5
+ Classifier: Programming Language :: Python :: Implementation :: CPython
6
+ Requires-Dist: fastcore>=1.14.6
7
+ Requires-Dist: fastship>=0.0.12 ; extra == 'dev'
8
+ Requires-Dist: maturin>=1.0,<2.0 ; extra == 'dev'
9
+ Requires-Dist: pytest ; extra == 'dev'
10
+ Provides-Extra: dev
11
+ License-File: LICENSE
12
+ Summary: Python API for ripgrep-style file walking and searching
13
+ Home-Page: https://github.com/AnswerDotAI/rgapi
14
+ Author-email: Jeremy Howard <j@fast.ai>
15
+ License: Apache-2.0
16
+ Requires-Python: >=3.10
17
+ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
18
+ Project-URL: Homepage, https://github.com/AnswerDotAI/rgapi
19
+ Project-URL: Issues, https://github.com/AnswerDotAI/rgapi/issues
20
+ Project-URL: Repository, https://github.com/AnswerDotAI/rgapi
21
+
22
+ # rgapi
23
+
24
+ `rgapi` provides `fd`-style file discovery and `rg`-style text search from Python without starting a shell command.
25
+
26
+ It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
27
+
28
+ ## Overview
29
+
30
+ For common file discovery and search:
31
+
32
+ ```python
33
+ from rgapi import fd, ls, rg, rg_iter
34
+
35
+ fd(".", ext="py", exclude="test_*.py")
36
+ ls("src")
37
+ for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
38
+ rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
39
+ ```
40
+
41
+ For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
42
+
43
+ ```python
44
+ from rgapi import nbrg
45
+
46
+ nbrg("read_csv", ".", cell_context=1)
47
+ ```
48
+
49
+ Walk and search functions have async versions. Streaming versions yield results as the search finds them. See [Async](#async):
50
+
51
+ ```python
52
+ from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
53
+
54
+ await rga("TODO", ".", ext="py", timeout_ms=200)
55
+ async for row in rga_iter("TODO", "."): print(row)
56
+ ```
57
+
58
+ Use the lower-level functions to compile a regex, search text or a single file, or walk a directory:
59
+
60
+ ```python
61
+ from rgapi import compile, search_path, search_text, walk
62
+
63
+ matcher = compile("TODO")
64
+ matcher.is_match("TODO")
65
+ matcher.finditer("TODO TODO")
66
+
67
+ walk(".")
68
+ search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
69
+ search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
70
+ ```
71
+
72
+ ## Install
73
+
74
+ ```bash
75
+ pip install rgapi
76
+ ```
77
+
78
+ ## File discovery
79
+
80
+ `fd` and `walk` return slash-separated paths relative to `root`. Pass `root` as a `str` or `pathlib.Path`. The sync and async APIs expand `~` and accept `.`, `./`, and paths containing `..`.
81
+
82
+ Discovery uses the `ignore` crate with ripgrep's default filters. It reads `.gitignore`, `.ignore`, and `.rgignore` files. `.rgignore` takes precedence over `.gitignore`. Pass `ignore=False` to disable all ignore-file filtering, including `.rgignore`.
83
+
84
+ Hidden files are skipped unless `hidden=True`. Symlinks are followed only with `follow_links=True`. Use `same_file_system=True` to avoid crossing filesystem boundaries.
85
+
86
+ Traversal runs in parallel without guaranteed result order. Use `sorted(...)` when order matters.
87
+
88
+ `fd` adds filename filters to `walk`. Its `pattern` is a smart-case regex matched against each basename. Lowercase patterns match case-insensitively. A pattern containing uppercase letters is case-sensitive. Use `path_re` to match the slash-separated relative path instead.
89
+
90
+ `include` and `exclude` use glob syntax. `glob=` is an alias for `include=`. A basename glob such as `*.py` also matches nested paths such as `src/app.py`.
91
+
92
+ Filter extensions with `ext="py"` or `ext=["py", "rs"]`. Extension and glob filters must both match. For example, `include="src/*", ext="py"` requires `src/*` and `*.py`, like combining `rg -g` with `-t`.
93
+
94
+ Set `min_depth` and `max_depth` to bound recursion. `max_filesize` skips files above a byte limit.
95
+
96
+ `ls` follows the shell command's listing conventions. It uses `fd` with `max_depth=1`, includes directories, disables ignore rules, and sorts by name. Set `hidden=True` for `ls -a` behaviour. All `fd` filters remain available.
97
+
98
+ `fd_iter` yields `FileEntry` paths as the walk finds them. It accepts every `fd` filter. Stopping iteration ends the walk. It does not accept `timeout_ms`.
99
+
100
+ `path_re` and `skip_path_re` filter slash-separated relative paths using regexes. They select returned paths or searched files without changing traversal. To skip entire subtrees, use `skip_dir` with a glob or `skip_dir_re` with a regex.
101
+
102
+ ## Text search
103
+
104
+ `rg` and `rg_iter` return structured `SearchLine` rows. They accept the same filters as `fd`: `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system`.
105
+
106
+ Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behaviour. Use `case_sensitive=False` to force case-insensitive matching.
107
+
108
+ Each `SearchLine` has these fields:
109
+
110
+ ```text
111
+ kind 'match', 'before', 'after', or 'context'
112
+ path path relative to root
113
+ line_number 1-based line number
114
+ lnhash exhash-style `lineno|hash|` address for the line
115
+ line line text without the trailing newline
116
+ matches list of (start, end) byte offsets for match rows
117
+ ```
118
+
119
+ `rg`, `search_text`, and `search_path` return `SearchResults` by default. This list subclass displays as rg-style multiline text in `str()` and notebook pretty output. `rg_iter` yields rows lazily.
120
+
121
+ `SearchLine` has a structured `repr` and an rg-style `str`. The string display truncates `line` to 180 characters with a trailing `…`. `repr` and `asdict()` retain the full line. `SearchLine.asdict()` returns the fields as a plain Python dictionary.
122
+
123
+ Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of line numbers. The `line_number` field remains available.
124
+
125
+ For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
126
+
127
+ ### Path results
128
+
129
+ `fd`, `walk`, and `ls` return `PathResults`. So do `rg` and `nbrg` with `paths=True`. This is a list of `FileEntry` rows.
130
+
131
+ A `FileEntry` is a `str` subclass containing a relative path. Its `stat` property calls `os.lstat` on first access and caches the result. It returns `None` if the path has vanished. `size`, `mtime`, and `is_dir` use the cached stat result. The object also supports ordinary string operations.
132
+
133
+ `PathResults` displays as an `ls -l`-style listing of at most `rgapi.MAX_REPR` rows. A final `… N more` line reports omitted rows. Only displayed rows need stat calls. `str()` returns one plain path per line. `list(res)` also displays plain paths.
134
+
135
+ ### Limits and timeouts
136
+
137
+ Set `timeout_ms` on `rg`, `fd`, `walk`, or `ls` to stop at a deadline and return the results collected so far. Their async versions accept it too. Results report why the operation stopped:
138
+
139
+ - `stop_reason=None` means the result is complete.
140
+ - `stop_reason="max_results"` means `max_results` truncated the result.
141
+ - `stop_reason="timeout"` means the deadline was reached.
142
+
143
+ `complete` is true exactly when `stop_reason` is `None`. `count=True` returns a plain integer without a completion flag. It cannot be combined with `timeout_ms`.
144
+
145
+ ### Context lines
146
+
147
+ `before_context`, `after_context`, and `context` correspond to `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
148
+
149
+ ### Block summaries
150
+
151
+ `rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
152
+
153
+ ```python
154
+ rg("TODO", ".", summary=True, context=1, maxlen=120)
155
+ ```
156
+
157
+ The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`.
158
+
159
+ Matches display as `path:start-end:source`. Context displays as `path:start-end-source`. With `lnhashs=True`, the range uses copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Newline runs display as `¶`. `maxlen` limits the displayed text without changing `source` or `asdict()`.
160
+
161
+ In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their context. `summary=True` cannot be combined with `paths` or `count`. It can be combined with `lnhash` for copyable block boundaries.
162
+
163
+ ## Notebooks
164
+
165
+ `nbrg` searches cell source in Jupyter `.ipynb` files and returns matching cells. Each result identifies the cell by its nbformat cell/message id, which stays stable across edits.
166
+
167
+ Plain `rg` searches escaped notebook JSON, including outputs and metadata. Its line numbers refer to that JSON file. `nbrg` searches the reconstructed cell source and identifies the cell you would edit.
168
+
169
+ ```python
170
+ from rgapi import nbrg
171
+
172
+ nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
173
+ nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
174
+ ```
175
+
176
+ Notebook discovery, parsing, and matching run together in one parallel Rust pass. Matching uses `rg`'s regex engine with the same `case_sensitive` and `smart_case` behaviour. `nbrg` accepts the discovery filters from `fd` and `rg`, including `include`, `exclude`, `glob`, `hidden`, `max_depth`, and `skip_dir`.
177
+
178
+ `nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
179
+
180
+ ```text
181
+ path notebook path relative to root
182
+ cell_index 0-based position of the cell in the notebook
183
+ cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
184
+ cell_type 'code', 'markdown', or 'raw'
185
+ kind 'match' or 'context'
186
+ source full cell source
187
+ matches list of SearchLine rows for the matched lines within the cell
188
+ ```
189
+
190
+ `NbCell.asdict()` returns these fields as a plain dictionary. Its `matches` field contains `SearchLine` dictionaries.
191
+
192
+ `str()` and pretty display show one line per cell. Matches use `path:cell_id:source`. Context uses `path:cell_id-source`. Newline runs display as `¶`.
193
+
194
+ A matching cell's display starts at its first matched line. Earlier lines are replaced by `…[Ln]`, where `n` is the matched line's one-based number within the cell. A leading `#|` directive is retained, as in `#| export…[L4]needle here`.
195
+
196
+ `maxlen` limits displayed source and defaults to 120. The full source remains in `source` and `asdict()`. Each cell appears once even when it has multiple matches. All hits remain in `matches`.
197
+
198
+ `cell_context=N` includes the `N` cells before and after each match as `kind="context"` rows. Context cells are deduplicated within each notebook.
199
+
200
+ `nbrg_iter` yields `NbCell` rows as notebooks are parsed. For collected results, `nbrg` accepts these limits and result options:
201
+
202
+ - `max_results` returns at most that many cells after sorting by path and cell index.
203
+ - `count=True` returns the number of matching cells.
204
+ - `timeout_ms` applies a deadline with the same `stop_reason` values as `rg`.
205
+
206
+ The parser reads only each cell's `id`, `cell_type`, and `source`. It skips outputs and metadata without loading embedded images or plots. `search_nb(pattern, path, ...)` searches a single notebook file in the same way.
207
+
208
+ `rgapi-nbrg` exposes notebook search without requiring a Python kernel:
209
+
210
+ ```bash
211
+ rgapi-nbrg 'read_csv' .
212
+ rgapi-nbrg 'read_csv' . --cell-context 1
213
+ rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
214
+ ```
215
+
216
+ Run `rgapi-nbrg --help` for its discovery, matching, and output options.
217
+
218
+ ## Async
219
+
220
+ `fda`, `rga`, and `nbrga` are async versions of `fd`, `rg`, and `nbrg`. The async generators `fda_iter`, `rga_iter`, and `nbrga_iter` yield rows as the search finds them. Async functions accept the same arguments and return the same types as their synchronous equivalents.
221
+
222
+ ```python
223
+ from rgapi import fda, rga, rga_iter
224
+
225
+ await fda(".", ext="py")
226
+ res = await rga("TODO", ".", timeout_ms=200)
227
+ if not res.complete: print(f"partial results: {res.stop_reason}")
228
+ async for row in rga_iter("TODO", "."): ...
229
+ ```
230
+
231
+ Walking and searching use Rust threads, without `asyncio.to_thread` or the event loop's executor. A callback uses `loop.call_soon_threadsafe` to complete the awaited future or supply results to the generator's queue. The event loop remains unblocked, with normal `contextvars` behaviour.
232
+
233
+ Cancellation stops the Rust workers within about one row. This includes timeouts from `asyncio.wait_for` or `asyncio.timeout`. It also includes task cancellation, such as starlette cancelling a disconnected client's request.
234
+
235
+ Wrap an async iterator in `contextlib.aclosing` when leaving its loop early. `break` alone delays generator finalization until garbage collection. The context manager provides prompt cleanup:
236
+
237
+ ```python
238
+ async with aclosing(rga_iter("TODO", ".")) as it:
239
+ async for row in it:
240
+ if enough(row): break
241
+ ```
242
+
243
+ Use streaming results for incremental display, such as sending batches to a browser as they arrive. Collected results with `timeout_ms` return what was found before the deadline. Use `asyncio.wait_for` when a timeout should raise instead.
244
+
245
+
246
+ ## Benchmarks
247
+
248
+ `tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
249
+
250
+ | fixture | rg | rgapi |
251
+ | --- | ---: | ---: |
252
+ | 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
253
+ | 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
254
+ | tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
255
+
rgapi-0.1.26/README.md ADDED
@@ -0,0 +1,233 @@
1
+ # rgapi
2
+
3
+ `rgapi` provides `fd`-style file discovery and `rg`-style text search from Python without starting a shell command.
4
+
5
+ It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
6
+
7
+ ## Overview
8
+
9
+ For common file discovery and search:
10
+
11
+ ```python
12
+ from rgapi import fd, ls, rg, rg_iter
13
+
14
+ fd(".", ext="py", exclude="test_*.py")
15
+ ls("src")
16
+ for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
17
+ rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
18
+ ```
19
+
20
+ For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
21
+
22
+ ```python
23
+ from rgapi import nbrg
24
+
25
+ nbrg("read_csv", ".", cell_context=1)
26
+ ```
27
+
28
+ Walk and search functions have async versions. Streaming versions yield results as the search finds them. See [Async](#async):
29
+
30
+ ```python
31
+ from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
32
+
33
+ await rga("TODO", ".", ext="py", timeout_ms=200)
34
+ async for row in rga_iter("TODO", "."): print(row)
35
+ ```
36
+
37
+ Use the lower-level functions to compile a regex, search text or a single file, or walk a directory:
38
+
39
+ ```python
40
+ from rgapi import compile, search_path, search_text, walk
41
+
42
+ matcher = compile("TODO")
43
+ matcher.is_match("TODO")
44
+ matcher.finditer("TODO TODO")
45
+
46
+ walk(".")
47
+ search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
48
+ search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
49
+ ```
50
+
51
+ ## Install
52
+
53
+ ```bash
54
+ pip install rgapi
55
+ ```
56
+
57
+ ## File discovery
58
+
59
+ `fd` and `walk` return slash-separated paths relative to `root`. Pass `root` as a `str` or `pathlib.Path`. The sync and async APIs expand `~` and accept `.`, `./`, and paths containing `..`.
60
+
61
+ Discovery uses the `ignore` crate with ripgrep's default filters. It reads `.gitignore`, `.ignore`, and `.rgignore` files. `.rgignore` takes precedence over `.gitignore`. Pass `ignore=False` to disable all ignore-file filtering, including `.rgignore`.
62
+
63
+ Hidden files are skipped unless `hidden=True`. Symlinks are followed only with `follow_links=True`. Use `same_file_system=True` to avoid crossing filesystem boundaries.
64
+
65
+ Traversal runs in parallel without guaranteed result order. Use `sorted(...)` when order matters.
66
+
67
+ `fd` adds filename filters to `walk`. Its `pattern` is a smart-case regex matched against each basename. Lowercase patterns match case-insensitively. A pattern containing uppercase letters is case-sensitive. Use `path_re` to match the slash-separated relative path instead.
68
+
69
+ `include` and `exclude` use glob syntax. `glob=` is an alias for `include=`. A basename glob such as `*.py` also matches nested paths such as `src/app.py`.
70
+
71
+ Filter extensions with `ext="py"` or `ext=["py", "rs"]`. Extension and glob filters must both match. For example, `include="src/*", ext="py"` requires `src/*` and `*.py`, like combining `rg -g` with `-t`.
72
+
73
+ Set `min_depth` and `max_depth` to bound recursion. `max_filesize` skips files above a byte limit.
74
+
75
+ `ls` follows the shell command's listing conventions. It uses `fd` with `max_depth=1`, includes directories, disables ignore rules, and sorts by name. Set `hidden=True` for `ls -a` behaviour. All `fd` filters remain available.
76
+
77
+ `fd_iter` yields `FileEntry` paths as the walk finds them. It accepts every `fd` filter. Stopping iteration ends the walk. It does not accept `timeout_ms`.
78
+
79
+ `path_re` and `skip_path_re` filter slash-separated relative paths using regexes. They select returned paths or searched files without changing traversal. To skip entire subtrees, use `skip_dir` with a glob or `skip_dir_re` with a regex.
80
+
81
+ ## Text search
82
+
83
+ `rg` and `rg_iter` return structured `SearchLine` rows. They accept the same filters as `fd`: `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system`.
84
+
85
+ Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behaviour. Use `case_sensitive=False` to force case-insensitive matching.
86
+
87
+ Each `SearchLine` has these fields:
88
+
89
+ ```text
90
+ kind 'match', 'before', 'after', or 'context'
91
+ path path relative to root
92
+ line_number 1-based line number
93
+ lnhash exhash-style `lineno|hash|` address for the line
94
+ line line text without the trailing newline
95
+ matches list of (start, end) byte offsets for match rows
96
+ ```
97
+
98
+ `rg`, `search_text`, and `search_path` return `SearchResults` by default. This list subclass displays as rg-style multiline text in `str()` and notebook pretty output. `rg_iter` yields rows lazily.
99
+
100
+ `SearchLine` has a structured `repr` and an rg-style `str`. The string display truncates `line` to 180 characters with a trailing `…`. `repr` and `asdict()` retain the full line. `SearchLine.asdict()` returns the fields as a plain Python dictionary.
101
+
102
+ Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of line numbers. The `line_number` field remains available.
103
+
104
+ For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
105
+
106
+ ### Path results
107
+
108
+ `fd`, `walk`, and `ls` return `PathResults`. So do `rg` and `nbrg` with `paths=True`. This is a list of `FileEntry` rows.
109
+
110
+ A `FileEntry` is a `str` subclass containing a relative path. Its `stat` property calls `os.lstat` on first access and caches the result. It returns `None` if the path has vanished. `size`, `mtime`, and `is_dir` use the cached stat result. The object also supports ordinary string operations.
111
+
112
+ `PathResults` displays as an `ls -l`-style listing of at most `rgapi.MAX_REPR` rows. A final `… N more` line reports omitted rows. Only displayed rows need stat calls. `str()` returns one plain path per line. `list(res)` also displays plain paths.
113
+
114
+ ### Limits and timeouts
115
+
116
+ Set `timeout_ms` on `rg`, `fd`, `walk`, or `ls` to stop at a deadline and return the results collected so far. Their async versions accept it too. Results report why the operation stopped:
117
+
118
+ - `stop_reason=None` means the result is complete.
119
+ - `stop_reason="max_results"` means `max_results` truncated the result.
120
+ - `stop_reason="timeout"` means the deadline was reached.
121
+
122
+ `complete` is true exactly when `stop_reason` is `None`. `count=True` returns a plain integer without a completion flag. It cannot be combined with `timeout_ms`.
123
+
124
+ ### Context lines
125
+
126
+ `before_context`, `after_context`, and `context` correspond to `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
127
+
128
+ ### Block summaries
129
+
130
+ `rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
131
+
132
+ ```python
133
+ rg("TODO", ".", summary=True, context=1, maxlen=120)
134
+ ```
135
+
136
+ The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`.
137
+
138
+ Matches display as `path:start-end:source`. Context displays as `path:start-end-source`. With `lnhashs=True`, the range uses copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Newline runs display as `¶`. `maxlen` limits the displayed text without changing `source` or `asdict()`.
139
+
140
+ In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their context. `summary=True` cannot be combined with `paths` or `count`. It can be combined with `lnhash` for copyable block boundaries.
141
+
142
+ ## Notebooks
143
+
144
+ `nbrg` searches cell source in Jupyter `.ipynb` files and returns matching cells. Each result identifies the cell by its nbformat cell/message id, which stays stable across edits.
145
+
146
+ Plain `rg` searches escaped notebook JSON, including outputs and metadata. Its line numbers refer to that JSON file. `nbrg` searches the reconstructed cell source and identifies the cell you would edit.
147
+
148
+ ```python
149
+ from rgapi import nbrg
150
+
151
+ nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
152
+ nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
153
+ ```
154
+
155
+ Notebook discovery, parsing, and matching run together in one parallel Rust pass. Matching uses `rg`'s regex engine with the same `case_sensitive` and `smart_case` behaviour. `nbrg` accepts the discovery filters from `fd` and `rg`, including `include`, `exclude`, `glob`, `hidden`, `max_depth`, and `skip_dir`.
156
+
157
+ `nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
158
+
159
+ ```text
160
+ path notebook path relative to root
161
+ cell_index 0-based position of the cell in the notebook
162
+ cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
163
+ cell_type 'code', 'markdown', or 'raw'
164
+ kind 'match' or 'context'
165
+ source full cell source
166
+ matches list of SearchLine rows for the matched lines within the cell
167
+ ```
168
+
169
+ `NbCell.asdict()` returns these fields as a plain dictionary. Its `matches` field contains `SearchLine` dictionaries.
170
+
171
+ `str()` and pretty display show one line per cell. Matches use `path:cell_id:source`. Context uses `path:cell_id-source`. Newline runs display as `¶`.
172
+
173
+ A matching cell's display starts at its first matched line. Earlier lines are replaced by `…[Ln]`, where `n` is the matched line's one-based number within the cell. A leading `#|` directive is retained, as in `#| export…[L4]needle here`.
174
+
175
+ `maxlen` limits displayed source and defaults to 120. The full source remains in `source` and `asdict()`. Each cell appears once even when it has multiple matches. All hits remain in `matches`.
176
+
177
+ `cell_context=N` includes the `N` cells before and after each match as `kind="context"` rows. Context cells are deduplicated within each notebook.
178
+
179
+ `nbrg_iter` yields `NbCell` rows as notebooks are parsed. For collected results, `nbrg` accepts these limits and result options:
180
+
181
+ - `max_results` returns at most that many cells after sorting by path and cell index.
182
+ - `count=True` returns the number of matching cells.
183
+ - `timeout_ms` applies a deadline with the same `stop_reason` values as `rg`.
184
+
185
+ The parser reads only each cell's `id`, `cell_type`, and `source`. It skips outputs and metadata without loading embedded images or plots. `search_nb(pattern, path, ...)` searches a single notebook file in the same way.
186
+
187
+ `rgapi-nbrg` exposes notebook search without requiring a Python kernel:
188
+
189
+ ```bash
190
+ rgapi-nbrg 'read_csv' .
191
+ rgapi-nbrg 'read_csv' . --cell-context 1
192
+ rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
193
+ ```
194
+
195
+ Run `rgapi-nbrg --help` for its discovery, matching, and output options.
196
+
197
+ ## Async
198
+
199
+ `fda`, `rga`, and `nbrga` are async versions of `fd`, `rg`, and `nbrg`. The async generators `fda_iter`, `rga_iter`, and `nbrga_iter` yield rows as the search finds them. Async functions accept the same arguments and return the same types as their synchronous equivalents.
200
+
201
+ ```python
202
+ from rgapi import fda, rga, rga_iter
203
+
204
+ await fda(".", ext="py")
205
+ res = await rga("TODO", ".", timeout_ms=200)
206
+ if not res.complete: print(f"partial results: {res.stop_reason}")
207
+ async for row in rga_iter("TODO", "."): ...
208
+ ```
209
+
210
+ Walking and searching use Rust threads, without `asyncio.to_thread` or the event loop's executor. A callback uses `loop.call_soon_threadsafe` to complete the awaited future or supply results to the generator's queue. The event loop remains unblocked, with normal `contextvars` behaviour.
211
+
212
+ Cancellation stops the Rust workers within about one row. This includes timeouts from `asyncio.wait_for` or `asyncio.timeout`. It also includes task cancellation, such as starlette cancelling a disconnected client's request.
213
+
214
+ Wrap an async iterator in `contextlib.aclosing` when leaving its loop early. `break` alone delays generator finalization until garbage collection. The context manager provides prompt cleanup:
215
+
216
+ ```python
217
+ async with aclosing(rga_iter("TODO", ".")) as it:
218
+ async for row in it:
219
+ if enough(row): break
220
+ ```
221
+
222
+ Use streaming results for incremental display, such as sending batches to a browser as they arrive. Collected results with `timeout_ms` return what was found before the deadline. Use `asyncio.wait_for` when a timeout should raise instead.
223
+
224
+
225
+ ## Benchmarks
226
+
227
+ `tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
228
+
229
+ | fixture | rg | rgapi |
230
+ | --- | ---: | ---: |
231
+ | 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
232
+ | 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
233
+ | tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
@@ -9,7 +9,7 @@ mod walk;
9
9
  mod python;
10
10
 
11
11
  pub use block::{BlockIter, SearchBlock, block_iter};
12
- pub use nb::{NbCell, NbIter, NbOptions, nb_iter, nb_search, nb_search_file};
12
+ pub use nb::{NbCell, NbIter, NbOptions, ancestor_indices, heading_level, nb_iter, nb_search, nb_search_file, section_range};
13
13
  pub use search::{MatchSpan, RgIter, RgOptions, SearchKind, SearchLine, compile_regex, rg, rg_iter, search_path, search_text};
14
14
  pub use walk::{FindIter, FindOptions, StreamIter, find, find_iter};
15
15
 
@@ -1,8 +1,9 @@
1
1
  use std::collections::{BTreeMap, HashMap};
2
2
  use std::path::{Path, PathBuf};
3
- use std::sync::Arc;
3
+ use std::sync::{Arc, LazyLock};
4
4
  use std::sync::atomic::Ordering;
5
5
 
6
+ use grep_matcher::Matcher;
6
7
  use grep_regex::RegexMatcher;
7
8
  use ignore::{DirEntry, WalkState};
8
9
  use serde::Deserialize;
@@ -11,6 +12,36 @@ use crate::RgApiError;
11
12
  use crate::search::{SearchLine, compile_regex, search_text};
12
13
  use crate::walk::{PathFilters, StreamIter, entry_err, file_root_flags, normalize_root, rel_path, spawn_walk};
13
14
 
15
+ static HEADING_RE: LazyLock<RegexMatcher> = LazyLock::new(|| RegexMatcher::new(r"^#{1,6} \w").unwrap());
16
+
17
+ /// Heading level of the first nonblank line after skipping `#|` directives; zero unless it matches `^#{1,6} \w`.
18
+ pub fn heading_level(source: &str) -> usize {
19
+ let line = source.lines().find(|l| !l.trim().is_empty() && !l.starts_with("#|")).unwrap_or("");
20
+ if !HEADING_RE.is_match(line.as_bytes()).unwrap_or(false) { return 0; }
21
+ line.bytes().take_while(|&b| b == b'#').count()
22
+ }
23
+
24
+ /// A heading and its descendants; a non-heading selects itself. Zero levels represent non-heading cells.
25
+ pub fn section_range(levels: &[usize], idx: usize) -> std::ops::Range<usize> {
26
+ if levels[idx] == 0 { return idx..idx + 1; }
27
+ let end = (idx + 1..levels.len()).find(|&i| levels[i] > 0 && levels[i] <= levels[idx]).unwrap_or(levels.len());
28
+ idx..end
29
+ }
30
+
31
+ /// Enclosing heading indices, outermost first, excluding the addressed cell.
32
+ pub fn ancestor_indices(levels: &[usize], idx: usize) -> Vec<usize> {
33
+ let mut level = if levels[idx] == 0 { 7 } else { levels[idx] };
34
+ let mut parents = Vec::new();
35
+ for i in (0..idx).rev() {
36
+ if levels[i] > 0 && levels[i] < level {
37
+ parents.push(i);
38
+ level = levels[i];
39
+ }
40
+ }
41
+ parents.reverse();
42
+ parents
43
+ }
44
+
14
45
  #[derive(Debug, Clone)]
15
46
  pub struct NbOptions {
16
47
  pub root: PathBuf,
@@ -104,13 +104,23 @@ pub fn find_iter(opts: &FindOptions) -> Result<FindIter, RgApiError> {
104
104
  ))
105
105
  }
106
106
 
107
- pub struct StreamIter<T> { rx: mpsc::Receiver<Result<T, RgApiError>>, cancel: Arc<AtomicBool>, _worker: std::thread::JoinHandle<()> }
107
+ pub struct StreamIter<T> { rx: mpsc::Receiver<Result<T, RgApiError>>, cancel: Arc<AtomicBool>, worker: Option<std::thread::JoinHandle<()>> }
108
108
 
109
109
  impl<T> StreamIter<T> {
110
110
  pub fn cancel(&self) { self.cancel.store(true, Ordering::Relaxed); }
111
111
 
112
112
  pub fn cancel_flag(&self) -> Arc<AtomicBool> { self.cancel.clone() }
113
113
 
114
+ /// Cancel and wait for the walk's workers to finish. Drain queued sends before
115
+ /// joining so a full result channel cannot deadlock shutdown. Unlike Drop,
116
+ /// this guarantees no background walk remains when it returns. Filesystem
117
+ /// calls already in progress must return first; run this off an async executor.
118
+ pub fn cancel_and_join(mut self) -> Result<(), RgApiError> {
119
+ self.cancel();
120
+ while self.rx.recv().is_ok() {}
121
+ self.worker.take().expect("stream owns its worker").join().map_err(|_| RgApiError::new("search worker panicked"))
122
+ }
123
+
114
124
  pub fn next_timeout(&mut self, timeout: std::time::Duration) -> Result<Result<T, RgApiError>, mpsc::RecvTimeoutError> { self.rx.recv_timeout(timeout) }
115
125
 
116
126
  /// Collect all items, stopping at `timeout_ms`; the bool is true when the deadline stopped it.
@@ -138,6 +148,33 @@ impl<T> Iterator for StreamIter<T> {
138
148
 
139
149
  impl<T> Drop for StreamIter<T> { fn drop(&mut self) { self.cancel(); } }
140
150
 
151
+ #[cfg(test)]
152
+ mod close_tests {
153
+ use super::*;
154
+
155
+ #[test]
156
+ fn cancel_and_join_drains_full_channel_and_waits_for_worker() {
157
+ let (tx, rx) = mpsc::sync_channel(1);
158
+ let (started, ready) = mpsc::channel();
159
+ let cancel = Arc::new(AtomicBool::new(false));
160
+ let worker_cancel = cancel.clone();
161
+ let done = Arc::new(AtomicBool::new(false));
162
+ let worker_done = done.clone();
163
+ let worker = std::thread::spawn(move || {
164
+ tx.send(Ok(1)).unwrap();
165
+ started.send(()).unwrap();
166
+ // This blocks while the channel is full; close must drain, not just join.
167
+ tx.send(Ok(2)).unwrap();
168
+ assert!(worker_cancel.load(Ordering::Acquire));
169
+ worker_done.store(true, Ordering::Release);
170
+ });
171
+ ready.recv().unwrap();
172
+ StreamIter { rx, cancel: cancel.clone(), worker: Some(worker) }.cancel_and_join().unwrap();
173
+ assert!(cancel.load(Ordering::Acquire));
174
+ assert!(done.load(Ordering::Acquire));
175
+ }
176
+ }
177
+
141
178
  #[allow(clippy::too_many_arguments)]
142
179
  pub fn spawn_walk<T, F>(
143
180
  root: PathBuf,
@@ -181,7 +218,7 @@ where
181
218
  })
182
219
  });
183
220
  });
184
- StreamIter { rx, cancel, _worker: worker }
221
+ StreamIter { rx, cancel, worker: Some(worker) }
185
222
  }
186
223
 
187
224
  fn find_entry(
@@ -365,7 +402,7 @@ mod tests {
365
402
  if tx.send(Ok(i)).is_err() { return; }
366
403
  }
367
404
  });
368
- StreamIter { rx, cancel, _worker: worker }
405
+ StreamIter { rx, cancel, worker: Some(worker) }
369
406
  }
370
407
 
371
408
  #[test]
rgapi-0.1.24/PKG-INFO DELETED
@@ -1,201 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: rgapi
3
- Version: 0.1.24
4
- Classifier: Programming Language :: Rust
5
- Classifier: Programming Language :: Python :: Implementation :: CPython
6
- Requires-Dist: fastcore>=1.14.6
7
- Requires-Dist: fastship>=0.0.12 ; extra == 'dev'
8
- Requires-Dist: maturin>=1.0,<2.0 ; extra == 'dev'
9
- Requires-Dist: pytest ; extra == 'dev'
10
- Provides-Extra: dev
11
- License-File: LICENSE
12
- Summary: Python API for ripgrep-style file walking and searching
13
- Home-Page: https://github.com/AnswerDotAI/rgapi
14
- Author-email: Jeremy Howard <j@fast.ai>
15
- License: Apache-2.0
16
- Requires-Python: >=3.10
17
- Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
18
- Project-URL: Homepage, https://github.com/AnswerDotAI/rgapi
19
- Project-URL: Issues, https://github.com/AnswerDotAI/rgapi/issues
20
- Project-URL: Repository, https://github.com/AnswerDotAI/rgapi
21
-
22
- # rgapi
23
-
24
- `rgapi` is a Python API for ripgrep-style walking and search. It is meant for Python code that wants `fd`-style file discovery or `rg`-style searching without shelling out.
25
-
26
- It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
27
-
28
- ## Overview
29
-
30
- For common file discovery and search:
31
-
32
- ```python
33
- from rgapi import fd, ls, rg, rg_iter
34
-
35
- fd(".", ext="py", exclude="test_*.py")
36
- ls("src")
37
- for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
38
- rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
39
- ```
40
-
41
- For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
42
-
43
- ```python
44
- from rgapi import nbrg
45
-
46
- nbrg("read_csv", ".", cell_context=1)
47
- ```
48
-
49
- Every walk and search has an async twin, plus streaming forms that yield results as they are found (see [Async](#async)):
50
-
51
- ```python
52
- from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
53
-
54
- await rga("TODO", ".", ext="py", timeout_ms=200)
55
- async for row in rga_iter("TODO", "."): print(row)
56
- ```
57
-
58
- For direct access to the regex, search, and walk pieces:
59
-
60
- ```python
61
- from rgapi import compile, search_path, search_text, walk
62
-
63
- matcher = compile("TODO")
64
- matcher.is_match("TODO")
65
- matcher.finditer("TODO TODO")
66
-
67
- walk(".")
68
- search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
69
- search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
70
- ```
71
-
72
- ## Install
73
-
74
- ```bash
75
- pip install rgapi
76
- ```
77
-
78
- ## Semantics
79
-
80
- `fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
81
-
82
- `fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
83
-
84
- `ls` lists like the shell command: it is `fd` with defaults flipped to one level (`max_depth=1`), directories included, ignore rules off, and results sorted by name. `hidden=True` is `ls -a`, and every `fd` filter still applies.
85
-
86
- `fd_iter` is the lazy form of `fd`, yielding `FileEntry` paths as the walk finds them, and takes every `fd` filter. It has no `timeout_ms`, since a consumer that stops asking for paths ends the walk itself.
87
-
88
- `path_re` and `skip_path_re` are regex filters on slash-separated relative paths. They filter returned paths or searched files, but do not control traversal. `skip_dir` uses glob syntax to prune matching directory subtrees, and `skip_dir_re` does the same with regex.
89
-
90
- `rg` and `rg_iter` return structured rows rather than raw CLI text. They accept the same `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system` filters as `fd`. Each row is a `SearchLine` with:
91
-
92
- ```text
93
- kind 'match', 'before', 'after', or 'context'
94
- path path relative to root
95
- line_number 1-based line number
96
- lnhash exhash-style `lineno|hash|` address for the line
97
- line line text without the trailing newline
98
- matches list of (start, end) byte offsets for match rows
99
- ```
100
-
101
- `rg`, `search_text`, and `search_path` return `SearchResults` by default, a list subclass whose `str()` and notebook pretty display are rg-style multiline text. `rg_iter` yields rows lazily.
102
-
103
- `SearchLine` has a structured `repr`, an rg-style `str` (the `line` is truncated to 180 chars with a trailing `…` for display; `repr` and `asdict()` keep the full line), and `SearchLine.asdict()` returns row fields as a plain Python dict. Pass `rg(..., lnhashs=True)` or `rg_iter(..., lnhashs=True)` to show `lnhash` addresses instead of line numbers in row display while keeping `line_number` available. `rg(..., paths=True)` returns unique matched paths, and `rg(..., count=True)` returns the total number of match spans. `paths` and `count` cannot both be set.
104
-
105
- `fd`, `walk`, `ls`, and `rg`/`nbrg` with `paths=True` return `PathResults`, a list of `FileEntry` rows. A `FileEntry` is a `str` subclass holding the relative path, so all string uses keep working, and it stats itself lazily on first access: `stat` is a cached `os.lstat` result (`None` if the path has vanished), with `size`, `mtime`, and `is_dir` derived from it. A `PathResults` displays as an `ls -l`-style long listing, capped at `rgapi.MAX_REPR` rows with a final `… N more` line, so stats are read only for displayed rows; `str()` is still one plain path per line, and `list(res)` shows plain paths. `rg(..., timeout_ms=200)` and `fd(..., timeout_ms=200)` stop at the deadline and return whatever was collected by then; `walk`, `ls`, and the async forms take it too. Results record how they ended: `stop_reason` is `None` for a complete result, `"max_results"` when truncated by `max_results`, or `"timeout"` when a deadline hit, and `complete` is true when `stop_reason` is `None`. `count=True` returns a plain int, which cannot carry the flag, so it rejects `timeout_ms`.
106
-
107
- `before_context`, `after_context`, and `context` are like `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
108
-
109
- ### Block summaries
110
-
111
- `rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
112
-
113
- ```python
114
- rg("TODO", ".", summary=True, context=1, maxlen=120)
115
- ```
116
-
117
- The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`. Its display is `path:start-end:source` for matches and `path:start-end-source` for context. With `lnhashs=True`, the numeric range becomes copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Embedded newline runs are shown as `¶`; `maxlen` limits displayed source without changing `source` or `asdict()`.
118
-
119
- In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their block context. `summary=True` cannot be combined with `paths` or `count`; it can be combined with `lnhash` when copyable block boundaries are useful.
120
-
121
- Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behavior, or `case_sensitive=False` to force case-insensitive matching.
122
-
123
- ## Notebooks
124
-
125
- `nbrg` searches Jupyter `.ipynb` files cell-by-cell, so results are *cells* rather than raw JSON lines, and each match is identified by its **cell id** (the nbformat cell/message id) rather than a line number. Searching a notebook with plain `rg` matches the escaped JSON text (including outputs and metadata) and reports meaningless JSON line numbers; `nbrg` instead searches each cell's reconstructed **source** and reports the cell id, which is stable across edits and points at the actual unit you work with.
126
-
127
- ```python
128
- from rgapi import nbrg
129
-
130
- nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
131
- nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
132
- ```
133
-
134
- Notebooks are walked, parsed, and matched together in one parallel Rust pass, using the same regex engine as `rg`, so regex behaviour and the `case_sensitive`/`smart_case` flags match `rg`. Only cell `source` is searched, not outputs or metadata. `nbrg` accepts the same discovery filters as `fd`/`rg` (`include`, `exclude`, `glob`, `hidden`, `max_depth`, `skip_dir`, …).
135
-
136
- `nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
137
-
138
- ```text
139
- path notebook path relative to root
140
- cell_index 0-based position of the cell in the notebook
141
- cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
142
- cell_type 'code', 'markdown', or 'raw'
143
- kind 'match' or 'context'
144
- source full cell source
145
- matches list of SearchLine rows for the matched lines within the cell
146
- ```
147
-
148
- `NbCell.asdict()` returns those fields as a plain dict (with `matches` as `SearchLine` dicts). `str()` and pretty display show one line per cell, newline runs shown as `¶`, keyed by `cell_id`: `path:cell_id:source` for matches and `path:cell_id-source` for context. A match row starts at its first matched line: earlier lines display as `…[Ln]` (`n` the matched line's 1-based number in the cell), except that a leading `#|` directive line is kept, e.g. `#| export…[L4]needle here`. `maxlen` controls the displayed source length and defaults to 120; the full source remains in `source` and `asdict()`. A cell with several matches appears once, with every hit collected in `matches`.
149
-
150
- `cell_context=N` includes the `N` cells before and after each matching cell as `kind="context"` rows (deduplicated per notebook).
151
-
152
- `nbrg_iter` yields `NbCell` rows lazily as notebooks are parsed. `nbrg` also accepts `max_results` (at most that many cells, after sorting by path and cell index), `count=True` (number of matching cells), and `timeout_ms=` with the same `stop_reason` semantics as `rg`.
153
-
154
- Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
155
-
156
- `rgapi-nbrg` exposes notebook search without requiring a Python kernel:
157
-
158
- ```bash
159
- rgapi-nbrg 'read_csv' .
160
- rgapi-nbrg 'read_csv' . --cell-context 1
161
- rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
162
- ```
163
-
164
- Run `rgapi-nbrg --help` for its discovery, matching, and output options.
165
-
166
- ## Async
167
-
168
- `fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
169
-
170
- ```python
171
- from rgapi import fda, rga, rga_iter
172
-
173
- await fda(".", ext="py")
174
- res = await rga("TODO", ".", timeout_ms=200)
175
- if not res.complete: print(f"partial results: {res.stop_reason}")
176
- async for row in rga_iter("TODO", "."): ...
177
- ```
178
-
179
- None of this uses `asyncio.to_thread` or the loop's executor. The walk and search run on Rust threads, and a single callback settles the awaited future (or feeds the generator's queue) through `loop.call_soon_threadsafe`, so the event loop never blocks and contextvars behave normally.
180
-
181
- Cancellation cleans up the Rust workers automatically. Wrapping a call in `asyncio.wait_for` or `asyncio.timeout`, cancelling the task (as starlette does when a client disconnects), or leaving an `async for` early all stop the search within about one row. One caveat comes from the language rather than the library: `break` inside `async for` only finalizes the generator at GC time, so for prompt cleanup wrap the iterator in `contextlib.aclosing`:
182
-
183
- ```python
184
- async with aclosing(rga_iter("TODO", ".")) as it:
185
- async for row in it:
186
- if enough(row): break
187
- ```
188
-
189
- The streaming forms suit incremental display, such as pushing each batch of results to a browser as it arrives. The collected forms with `timeout_ms` give the best results available within a budget, and `asyncio.wait_for` gives timeout-as-failure. Pick per call site.
190
-
191
-
192
- ## Benchmarks
193
-
194
- `tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
195
-
196
- | fixture | rg | rgapi |
197
- | --- | ---: | ---: |
198
- | 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
199
- | 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
200
- | tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
201
-
rgapi-0.1.24/README.md DELETED
@@ -1,179 +0,0 @@
1
- # rgapi
2
-
3
- `rgapi` is a Python API for ripgrep-style walking and search. It is meant for Python code that wants `fd`-style file discovery or `rg`-style searching without shelling out.
4
-
5
- It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
6
-
7
- ## Overview
8
-
9
- For common file discovery and search:
10
-
11
- ```python
12
- from rgapi import fd, ls, rg, rg_iter
13
-
14
- fd(".", ext="py", exclude="test_*.py")
15
- ls("src")
16
- for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
17
- rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
18
- ```
19
-
20
- For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
21
-
22
- ```python
23
- from rgapi import nbrg
24
-
25
- nbrg("read_csv", ".", cell_context=1)
26
- ```
27
-
28
- Every walk and search has an async twin, plus streaming forms that yield results as they are found (see [Async](#async)):
29
-
30
- ```python
31
- from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
32
-
33
- await rga("TODO", ".", ext="py", timeout_ms=200)
34
- async for row in rga_iter("TODO", "."): print(row)
35
- ```
36
-
37
- For direct access to the regex, search, and walk pieces:
38
-
39
- ```python
40
- from rgapi import compile, search_path, search_text, walk
41
-
42
- matcher = compile("TODO")
43
- matcher.is_match("TODO")
44
- matcher.finditer("TODO TODO")
45
-
46
- walk(".")
47
- search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
48
- search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
49
- ```
50
-
51
- ## Install
52
-
53
- ```bash
54
- pip install rgapi
55
- ```
56
-
57
- ## Semantics
58
-
59
- `fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
60
-
61
- `fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
62
-
63
- `ls` lists like the shell command: it is `fd` with defaults flipped to one level (`max_depth=1`), directories included, ignore rules off, and results sorted by name. `hidden=True` is `ls -a`, and every `fd` filter still applies.
64
-
65
- `fd_iter` is the lazy form of `fd`, yielding `FileEntry` paths as the walk finds them, and takes every `fd` filter. It has no `timeout_ms`, since a consumer that stops asking for paths ends the walk itself.
66
-
67
- `path_re` and `skip_path_re` are regex filters on slash-separated relative paths. They filter returned paths or searched files, but do not control traversal. `skip_dir` uses glob syntax to prune matching directory subtrees, and `skip_dir_re` does the same with regex.
68
-
69
- `rg` and `rg_iter` return structured rows rather than raw CLI text. They accept the same `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system` filters as `fd`. Each row is a `SearchLine` with:
70
-
71
- ```text
72
- kind 'match', 'before', 'after', or 'context'
73
- path path relative to root
74
- line_number 1-based line number
75
- lnhash exhash-style `lineno|hash|` address for the line
76
- line line text without the trailing newline
77
- matches list of (start, end) byte offsets for match rows
78
- ```
79
-
80
- `rg`, `search_text`, and `search_path` return `SearchResults` by default, a list subclass whose `str()` and notebook pretty display are rg-style multiline text. `rg_iter` yields rows lazily.
81
-
82
- `SearchLine` has a structured `repr`, an rg-style `str` (the `line` is truncated to 180 chars with a trailing `…` for display; `repr` and `asdict()` keep the full line), and `SearchLine.asdict()` returns row fields as a plain Python dict. Pass `rg(..., lnhashs=True)` or `rg_iter(..., lnhashs=True)` to show `lnhash` addresses instead of line numbers in row display while keeping `line_number` available. `rg(..., paths=True)` returns unique matched paths, and `rg(..., count=True)` returns the total number of match spans. `paths` and `count` cannot both be set.
83
-
84
- `fd`, `walk`, `ls`, and `rg`/`nbrg` with `paths=True` return `PathResults`, a list of `FileEntry` rows. A `FileEntry` is a `str` subclass holding the relative path, so all string uses keep working, and it stats itself lazily on first access: `stat` is a cached `os.lstat` result (`None` if the path has vanished), with `size`, `mtime`, and `is_dir` derived from it. A `PathResults` displays as an `ls -l`-style long listing, capped at `rgapi.MAX_REPR` rows with a final `… N more` line, so stats are read only for displayed rows; `str()` is still one plain path per line, and `list(res)` shows plain paths. `rg(..., timeout_ms=200)` and `fd(..., timeout_ms=200)` stop at the deadline and return whatever was collected by then; `walk`, `ls`, and the async forms take it too. Results record how they ended: `stop_reason` is `None` for a complete result, `"max_results"` when truncated by `max_results`, or `"timeout"` when a deadline hit, and `complete` is true when `stop_reason` is `None`. `count=True` returns a plain int, which cannot carry the flag, so it rejects `timeout_ms`.
85
-
86
- `before_context`, `after_context`, and `context` are like `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
87
-
88
- ### Block summaries
89
-
90
- `rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
91
-
92
- ```python
93
- rg("TODO", ".", summary=True, context=1, maxlen=120)
94
- ```
95
-
96
- The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`. Its display is `path:start-end:source` for matches and `path:start-end-source` for context. With `lnhashs=True`, the numeric range becomes copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Embedded newline runs are shown as `¶`; `maxlen` limits displayed source without changing `source` or `asdict()`.
97
-
98
- In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their block context. `summary=True` cannot be combined with `paths` or `count`; it can be combined with `lnhash` when copyable block boundaries are useful.
99
-
100
- Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behavior, or `case_sensitive=False` to force case-insensitive matching.
101
-
102
- ## Notebooks
103
-
104
- `nbrg` searches Jupyter `.ipynb` files cell-by-cell, so results are *cells* rather than raw JSON lines, and each match is identified by its **cell id** (the nbformat cell/message id) rather than a line number. Searching a notebook with plain `rg` matches the escaped JSON text (including outputs and metadata) and reports meaningless JSON line numbers; `nbrg` instead searches each cell's reconstructed **source** and reports the cell id, which is stable across edits and points at the actual unit you work with.
105
-
106
- ```python
107
- from rgapi import nbrg
108
-
109
- nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
110
- nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
111
- ```
112
-
113
- Notebooks are walked, parsed, and matched together in one parallel Rust pass, using the same regex engine as `rg`, so regex behaviour and the `case_sensitive`/`smart_case` flags match `rg`. Only cell `source` is searched, not outputs or metadata. `nbrg` accepts the same discovery filters as `fd`/`rg` (`include`, `exclude`, `glob`, `hidden`, `max_depth`, `skip_dir`, …).
114
-
115
- `nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
116
-
117
- ```text
118
- path notebook path relative to root
119
- cell_index 0-based position of the cell in the notebook
120
- cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
121
- cell_type 'code', 'markdown', or 'raw'
122
- kind 'match' or 'context'
123
- source full cell source
124
- matches list of SearchLine rows for the matched lines within the cell
125
- ```
126
-
127
- `NbCell.asdict()` returns those fields as a plain dict (with `matches` as `SearchLine` dicts). `str()` and pretty display show one line per cell, newline runs shown as `¶`, keyed by `cell_id`: `path:cell_id:source` for matches and `path:cell_id-source` for context. A match row starts at its first matched line: earlier lines display as `…[Ln]` (`n` the matched line's 1-based number in the cell), except that a leading `#|` directive line is kept, e.g. `#| export…[L4]needle here`. `maxlen` controls the displayed source length and defaults to 120; the full source remains in `source` and `asdict()`. A cell with several matches appears once, with every hit collected in `matches`.
128
-
129
- `cell_context=N` includes the `N` cells before and after each matching cell as `kind="context"` rows (deduplicated per notebook).
130
-
131
- `nbrg_iter` yields `NbCell` rows lazily as notebooks are parsed. `nbrg` also accepts `max_results` (at most that many cells, after sorting by path and cell index), `count=True` (number of matching cells), and `timeout_ms=` with the same `stop_reason` semantics as `rg`.
132
-
133
- Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
134
-
135
- `rgapi-nbrg` exposes notebook search without requiring a Python kernel:
136
-
137
- ```bash
138
- rgapi-nbrg 'read_csv' .
139
- rgapi-nbrg 'read_csv' . --cell-context 1
140
- rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
141
- ```
142
-
143
- Run `rgapi-nbrg --help` for its discovery, matching, and output options.
144
-
145
- ## Async
146
-
147
- `fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
148
-
149
- ```python
150
- from rgapi import fda, rga, rga_iter
151
-
152
- await fda(".", ext="py")
153
- res = await rga("TODO", ".", timeout_ms=200)
154
- if not res.complete: print(f"partial results: {res.stop_reason}")
155
- async for row in rga_iter("TODO", "."): ...
156
- ```
157
-
158
- None of this uses `asyncio.to_thread` or the loop's executor. The walk and search run on Rust threads, and a single callback settles the awaited future (or feeds the generator's queue) through `loop.call_soon_threadsafe`, so the event loop never blocks and contextvars behave normally.
159
-
160
- Cancellation cleans up the Rust workers automatically. Wrapping a call in `asyncio.wait_for` or `asyncio.timeout`, cancelling the task (as starlette does when a client disconnects), or leaving an `async for` early all stop the search within about one row. One caveat comes from the language rather than the library: `break` inside `async for` only finalizes the generator at GC time, so for prompt cleanup wrap the iterator in `contextlib.aclosing`:
161
-
162
- ```python
163
- async with aclosing(rga_iter("TODO", ".")) as it:
164
- async for row in it:
165
- if enough(row): break
166
- ```
167
-
168
- The streaming forms suit incremental display, such as pushing each batch of results to a browser as it arrives. The collected forms with `timeout_ms` give the best results available within a budget, and `asyncio.wait_for` gives timeout-as-failure. Pick per call site.
169
-
170
-
171
- ## Benchmarks
172
-
173
- `tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
174
-
175
- | fixture | rg | rgapi |
176
- | --- | ---: | ---: |
177
- | 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
178
- | 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
179
- | tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes