rgapi 0.1.24__tar.gz → 0.1.26__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rgapi-0.1.24 → rgapi-0.1.26}/Cargo.lock +62 -5
- {rgapi-0.1.24 → rgapi-0.1.26}/Cargo.toml +1 -1
- {rgapi-0.1.24 → rgapi-0.1.26}/DEV.md +4 -0
- rgapi-0.1.26/PKG-INFO +255 -0
- rgapi-0.1.26/README.md +233 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/src/lib.rs +1 -1
- {rgapi-0.1.24 → rgapi-0.1.26}/src/nb.rs +32 -1
- {rgapi-0.1.24 → rgapi-0.1.26}/src/walk.rs +40 -3
- rgapi-0.1.24/PKG-INFO +0 -201
- rgapi-0.1.24/README.md +0 -179
- {rgapi-0.1.24 → rgapi-0.1.26}/.github/workflows/ci.yml +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/.gitignore +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/LICENSE +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/_config.yml +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/_layouts/default.html +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/pyproject.toml +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/python/rgapi/__init__.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/python/rgapi/_cli.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/python/rgapi/block.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/python/rgapi/nb.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/python/rgapi/skill.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/rustfmt.toml +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/src/block.rs +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/src/python.rs +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/src/search.rs +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/tests/test_async.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/tests/test_rgapi.py +0 -0
- {rgapi-0.1.24 → rgapi-0.1.26}/tools/bench.py +0 -0
|
@@ -28,11 +28,17 @@ version = "1.0.4"
|
|
|
28
28
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
29
29
|
checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
|
|
30
30
|
|
|
31
|
+
[[package]]
|
|
32
|
+
name = "core_detect"
|
|
33
|
+
version = "1.0.0"
|
|
34
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
35
|
+
checksum = "7f8f80099a98041a3d1622845c271458a2d73e688351bf3cb999266764b81d48"
|
|
36
|
+
|
|
31
37
|
[[package]]
|
|
32
38
|
name = "crc32fast"
|
|
33
|
-
version = "1.5.
|
|
39
|
+
version = "1.5.2"
|
|
34
40
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
35
|
-
checksum = "
|
|
41
|
+
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
|
36
42
|
dependencies = [
|
|
37
43
|
"cfg-if",
|
|
38
44
|
]
|
|
@@ -64,11 +70,17 @@ checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
|
|
|
64
70
|
|
|
65
71
|
[[package]]
|
|
66
72
|
name = "encoding_rs"
|
|
67
|
-
version = "0.8.
|
|
73
|
+
version = "0.8.41"
|
|
68
74
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
69
|
-
checksum = "
|
|
75
|
+
checksum = "7b5ef0006ac9ab233c38522f5ae99cae3625151de8f706cacee1cba4b8e2832a"
|
|
70
76
|
dependencies = [
|
|
71
77
|
"cfg-if",
|
|
78
|
+
"core_detect",
|
|
79
|
+
"multiversion",
|
|
80
|
+
"multiversion_no_op",
|
|
81
|
+
"rustversion",
|
|
82
|
+
"scopeguard",
|
|
83
|
+
"simdutf8",
|
|
72
84
|
]
|
|
73
85
|
|
|
74
86
|
[[package]]
|
|
@@ -185,6 +197,33 @@ dependencies = [
|
|
|
185
197
|
"libc",
|
|
186
198
|
]
|
|
187
199
|
|
|
200
|
+
[[package]]
|
|
201
|
+
name = "multiversion"
|
|
202
|
+
version = "0.9.0"
|
|
203
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
204
|
+
checksum = "b4ca4bea16ffc3f443cf7d866912118196bfef4c6a1556ca00f9f9b00bb43f7c"
|
|
205
|
+
dependencies = [
|
|
206
|
+
"multiversion-macros",
|
|
207
|
+
]
|
|
208
|
+
|
|
209
|
+
[[package]]
|
|
210
|
+
name = "multiversion-macros"
|
|
211
|
+
version = "0.9.0"
|
|
212
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
213
|
+
checksum = "0d416831a7317ef4b08bee00b69cbbb9c8763da7959a7026244d6266869f9c83"
|
|
214
|
+
dependencies = [
|
|
215
|
+
"proc-macro2",
|
|
216
|
+
"quote",
|
|
217
|
+
"rustversion",
|
|
218
|
+
"syn 3.0.5",
|
|
219
|
+
]
|
|
220
|
+
|
|
221
|
+
[[package]]
|
|
222
|
+
name = "multiversion_no_op"
|
|
223
|
+
version = "1.0.0"
|
|
224
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
225
|
+
checksum = "743fb55ba31b18fb1ecef6bdc9aa2743314978ac084044301a7eee33fb99a20d"
|
|
226
|
+
|
|
188
227
|
[[package]]
|
|
189
228
|
name = "once_cell"
|
|
190
229
|
version = "1.21.4"
|
|
@@ -291,7 +330,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
|
|
|
291
330
|
|
|
292
331
|
[[package]]
|
|
293
332
|
name = "rgapi"
|
|
294
|
-
version = "0.1.
|
|
333
|
+
version = "0.1.26"
|
|
295
334
|
dependencies = [
|
|
296
335
|
"crc32fast",
|
|
297
336
|
"globset",
|
|
@@ -304,6 +343,12 @@ dependencies = [
|
|
|
304
343
|
"serde_json",
|
|
305
344
|
]
|
|
306
345
|
|
|
346
|
+
[[package]]
|
|
347
|
+
name = "rustversion"
|
|
348
|
+
version = "1.0.23"
|
|
349
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
350
|
+
checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f"
|
|
351
|
+
|
|
307
352
|
[[package]]
|
|
308
353
|
name = "same-file"
|
|
309
354
|
version = "1.0.6"
|
|
@@ -313,6 +358,12 @@ dependencies = [
|
|
|
313
358
|
"winapi-util",
|
|
314
359
|
]
|
|
315
360
|
|
|
361
|
+
[[package]]
|
|
362
|
+
name = "scopeguard"
|
|
363
|
+
version = "1.2.0"
|
|
364
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
365
|
+
checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
|
|
366
|
+
|
|
316
367
|
[[package]]
|
|
317
368
|
name = "serde"
|
|
318
369
|
version = "1.0.229"
|
|
@@ -356,6 +407,12 @@ dependencies = [
|
|
|
356
407
|
"zmij",
|
|
357
408
|
]
|
|
358
409
|
|
|
410
|
+
[[package]]
|
|
411
|
+
name = "simdutf8"
|
|
412
|
+
version = "0.1.5"
|
|
413
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
414
|
+
checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
|
|
415
|
+
|
|
359
416
|
[[package]]
|
|
360
417
|
name = "syn"
|
|
361
418
|
version = "2.0.119"
|
|
@@ -48,4 +48,8 @@ Truncation is recorded on collected results: `max_results` sets `stop_reason="ma
|
|
|
48
48
|
|
|
49
49
|
Path results are `FileEntry` rows: a `str` subclass carrying the walk root, so paths stay plain strings for compatibility while stat info loads lazily (one cached `os.lstat` per entry, read only on attribute access). The wrapping happens at result construction on the Python side; Rust still streams plain strings. `PathResults.__repr__` shows an `ls -l`-style listing capped at `MAX_REPR` rows, so a huge result never stats everything, while `str()` stays one plain path per line. `ls` is `fd` with shell-style defaults (one level, dirs, ignore rules off), re-sorted with `stop_reason` preserved.
|
|
50
50
|
|
|
51
|
+
Rust callers can consume a `StreamIter` with `cancel_and_join()` to cancel, drain queued results, and wait for the walk's workers to finish. `Drop` remains nonblocking. Since filesystem calls already in progress must return before joining completes, keep `cancel_and_join()` off async executors.
|
|
52
|
+
|
|
53
|
+
Notebook hierarchy uses `heading_level(source)`, `section_range(levels, idx)` and `ancestor_indices(levels, idx)`. Callers supply zero for non-heading cells. Heading detection skips blank lines and lines starting with `#|`, then checks the first remaining line against `^#{1,6} \w`. It does not search past ordinary text. Section ranges include the addressed cell and end before the next equal-or-higher heading; non-headings select themselves. Ancestors exclude the addressed cell and are returned outermost first. These are calculations over the supplied levels, without retained outline state. Rustygate uses them for its cell selectors.
|
|
54
|
+
|
|
51
55
|
This package intentionally has no CLI. Python is the interface.
|
rgapi-0.1.26/PKG-INFO
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rgapi
|
|
3
|
+
Version: 0.1.26
|
|
4
|
+
Classifier: Programming Language :: Rust
|
|
5
|
+
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
6
|
+
Requires-Dist: fastcore>=1.14.6
|
|
7
|
+
Requires-Dist: fastship>=0.0.12 ; extra == 'dev'
|
|
8
|
+
Requires-Dist: maturin>=1.0,<2.0 ; extra == 'dev'
|
|
9
|
+
Requires-Dist: pytest ; extra == 'dev'
|
|
10
|
+
Provides-Extra: dev
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Summary: Python API for ripgrep-style file walking and searching
|
|
13
|
+
Home-Page: https://github.com/AnswerDotAI/rgapi
|
|
14
|
+
Author-email: Jeremy Howard <j@fast.ai>
|
|
15
|
+
License: Apache-2.0
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
18
|
+
Project-URL: Homepage, https://github.com/AnswerDotAI/rgapi
|
|
19
|
+
Project-URL: Issues, https://github.com/AnswerDotAI/rgapi/issues
|
|
20
|
+
Project-URL: Repository, https://github.com/AnswerDotAI/rgapi
|
|
21
|
+
|
|
22
|
+
# rgapi
|
|
23
|
+
|
|
24
|
+
`rgapi` provides `fd`-style file discovery and `rg`-style text search from Python without starting a shell command.
|
|
25
|
+
|
|
26
|
+
It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
|
|
27
|
+
|
|
28
|
+
## Overview
|
|
29
|
+
|
|
30
|
+
For common file discovery and search:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from rgapi import fd, ls, rg, rg_iter
|
|
34
|
+
|
|
35
|
+
fd(".", ext="py", exclude="test_*.py")
|
|
36
|
+
ls("src")
|
|
37
|
+
for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
|
|
38
|
+
rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from rgapi import nbrg
|
|
45
|
+
|
|
46
|
+
nbrg("read_csv", ".", cell_context=1)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Walk and search functions have async versions. Streaming versions yield results as the search finds them. See [Async](#async):
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
|
|
53
|
+
|
|
54
|
+
await rga("TODO", ".", ext="py", timeout_ms=200)
|
|
55
|
+
async for row in rga_iter("TODO", "."): print(row)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Use the lower-level functions to compile a regex, search text or a single file, or walk a directory:
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from rgapi import compile, search_path, search_text, walk
|
|
62
|
+
|
|
63
|
+
matcher = compile("TODO")
|
|
64
|
+
matcher.is_match("TODO")
|
|
65
|
+
matcher.finditer("TODO TODO")
|
|
66
|
+
|
|
67
|
+
walk(".")
|
|
68
|
+
search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
|
|
69
|
+
search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Install
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
pip install rgapi
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
## File discovery
|
|
79
|
+
|
|
80
|
+
`fd` and `walk` return slash-separated paths relative to `root`. Pass `root` as a `str` or `pathlib.Path`. The sync and async APIs expand `~` and accept `.`, `./`, and paths containing `..`.
|
|
81
|
+
|
|
82
|
+
Discovery uses the `ignore` crate with ripgrep's default filters. It reads `.gitignore`, `.ignore`, and `.rgignore` files. `.rgignore` takes precedence over `.gitignore`. Pass `ignore=False` to disable all ignore-file filtering, including `.rgignore`.
|
|
83
|
+
|
|
84
|
+
Hidden files are skipped unless `hidden=True`. Symlinks are followed only with `follow_links=True`. Use `same_file_system=True` to avoid crossing filesystem boundaries.
|
|
85
|
+
|
|
86
|
+
Traversal runs in parallel without guaranteed result order. Use `sorted(...)` when order matters.
|
|
87
|
+
|
|
88
|
+
`fd` adds filename filters to `walk`. Its `pattern` is a smart-case regex matched against each basename. Lowercase patterns match case-insensitively. A pattern containing uppercase letters is case-sensitive. Use `path_re` to match the slash-separated relative path instead.
|
|
89
|
+
|
|
90
|
+
`include` and `exclude` use glob syntax. `glob=` is an alias for `include=`. A basename glob such as `*.py` also matches nested paths such as `src/app.py`.
|
|
91
|
+
|
|
92
|
+
Filter extensions with `ext="py"` or `ext=["py", "rs"]`. Extension and glob filters must both match. For example, `include="src/*", ext="py"` requires `src/*` and `*.py`, like combining `rg -g` with `-t`.
|
|
93
|
+
|
|
94
|
+
Set `min_depth` and `max_depth` to bound recursion. `max_filesize` skips files above a byte limit.
|
|
95
|
+
|
|
96
|
+
`ls` follows the shell command's listing conventions. It uses `fd` with `max_depth=1`, includes directories, disables ignore rules, and sorts by name. Set `hidden=True` for `ls -a` behaviour. All `fd` filters remain available.
|
|
97
|
+
|
|
98
|
+
`fd_iter` yields `FileEntry` paths as the walk finds them. It accepts every `fd` filter. Stopping iteration ends the walk. It does not accept `timeout_ms`.
|
|
99
|
+
|
|
100
|
+
`path_re` and `skip_path_re` filter slash-separated relative paths using regexes. They select returned paths or searched files without changing traversal. To skip entire subtrees, use `skip_dir` with a glob or `skip_dir_re` with a regex.
|
|
101
|
+
|
|
102
|
+
## Text search
|
|
103
|
+
|
|
104
|
+
`rg` and `rg_iter` return structured `SearchLine` rows. They accept the same filters as `fd`: `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system`.
|
|
105
|
+
|
|
106
|
+
Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behaviour. Use `case_sensitive=False` to force case-insensitive matching.
|
|
107
|
+
|
|
108
|
+
Each `SearchLine` has these fields:
|
|
109
|
+
|
|
110
|
+
```text
|
|
111
|
+
kind 'match', 'before', 'after', or 'context'
|
|
112
|
+
path path relative to root
|
|
113
|
+
line_number 1-based line number
|
|
114
|
+
lnhash exhash-style `lineno|hash|` address for the line
|
|
115
|
+
line line text without the trailing newline
|
|
116
|
+
matches list of (start, end) byte offsets for match rows
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
`rg`, `search_text`, and `search_path` return `SearchResults` by default. This list subclass displays as rg-style multiline text in `str()` and notebook pretty output. `rg_iter` yields rows lazily.
|
|
120
|
+
|
|
121
|
+
`SearchLine` has a structured `repr` and an rg-style `str`. The string display truncates `line` to 180 characters with a trailing `…`. `repr` and `asdict()` retain the full line. `SearchLine.asdict()` returns the fields as a plain Python dictionary.
|
|
122
|
+
|
|
123
|
+
Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of line numbers. The `line_number` field remains available.
|
|
124
|
+
|
|
125
|
+
For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
|
|
126
|
+
|
|
127
|
+
### Path results
|
|
128
|
+
|
|
129
|
+
`fd`, `walk`, and `ls` return `PathResults`. So do `rg` and `nbrg` with `paths=True`. This is a list of `FileEntry` rows.
|
|
130
|
+
|
|
131
|
+
A `FileEntry` is a `str` subclass containing a relative path. Its `stat` property calls `os.lstat` on first access and caches the result. It returns `None` if the path has vanished. `size`, `mtime`, and `is_dir` use the cached stat result. The object also supports ordinary string operations.
|
|
132
|
+
|
|
133
|
+
`PathResults` displays as an `ls -l`-style listing of at most `rgapi.MAX_REPR` rows. A final `… N more` line reports omitted rows. Only displayed rows need stat calls. `str()` returns one plain path per line. `list(res)` also displays plain paths.
|
|
134
|
+
|
|
135
|
+
### Limits and timeouts
|
|
136
|
+
|
|
137
|
+
Set `timeout_ms` on `rg`, `fd`, `walk`, or `ls` to stop at a deadline and return the results collected so far. Their async versions accept it too. Results report why the operation stopped:
|
|
138
|
+
|
|
139
|
+
- `stop_reason=None` means the result is complete.
|
|
140
|
+
- `stop_reason="max_results"` means `max_results` truncated the result.
|
|
141
|
+
- `stop_reason="timeout"` means the deadline was reached.
|
|
142
|
+
|
|
143
|
+
`complete` is true exactly when `stop_reason` is `None`. `count=True` returns a plain integer without a completion flag. It cannot be combined with `timeout_ms`.
|
|
144
|
+
|
|
145
|
+
### Context lines
|
|
146
|
+
|
|
147
|
+
`before_context`, `after_context`, and `context` correspond to `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
|
|
148
|
+
|
|
149
|
+
### Block summaries
|
|
150
|
+
|
|
151
|
+
`rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
rg("TODO", ".", summary=True, context=1, maxlen=120)
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`.
|
|
158
|
+
|
|
159
|
+
Matches display as `path:start-end:source`. Context displays as `path:start-end-source`. With `lnhashs=True`, the range uses copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Newline runs display as `¶`. `maxlen` limits the displayed text without changing `source` or `asdict()`.
|
|
160
|
+
|
|
161
|
+
In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their context. `summary=True` cannot be combined with `paths` or `count`. It can be combined with `lnhash` for copyable block boundaries.
|
|
162
|
+
|
|
163
|
+
## Notebooks
|
|
164
|
+
|
|
165
|
+
`nbrg` searches cell source in Jupyter `.ipynb` files and returns matching cells. Each result identifies the cell by its nbformat cell/message id, which stays stable across edits.
|
|
166
|
+
|
|
167
|
+
Plain `rg` searches escaped notebook JSON, including outputs and metadata. Its line numbers refer to that JSON file. `nbrg` searches the reconstructed cell source and identifies the cell you would edit.
|
|
168
|
+
|
|
169
|
+
```python
|
|
170
|
+
from rgapi import nbrg
|
|
171
|
+
|
|
172
|
+
nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
|
|
173
|
+
nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
Notebook discovery, parsing, and matching run together in one parallel Rust pass. Matching uses `rg`'s regex engine with the same `case_sensitive` and `smart_case` behaviour. `nbrg` accepts the discovery filters from `fd` and `rg`, including `include`, `exclude`, `glob`, `hidden`, `max_depth`, and `skip_dir`.
|
|
177
|
+
|
|
178
|
+
`nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
|
|
179
|
+
|
|
180
|
+
```text
|
|
181
|
+
path notebook path relative to root
|
|
182
|
+
cell_index 0-based position of the cell in the notebook
|
|
183
|
+
cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
|
|
184
|
+
cell_type 'code', 'markdown', or 'raw'
|
|
185
|
+
kind 'match' or 'context'
|
|
186
|
+
source full cell source
|
|
187
|
+
matches list of SearchLine rows for the matched lines within the cell
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
`NbCell.asdict()` returns these fields as a plain dictionary. Its `matches` field contains `SearchLine` dictionaries.
|
|
191
|
+
|
|
192
|
+
`str()` and pretty display show one line per cell. Matches use `path:cell_id:source`. Context uses `path:cell_id-source`. Newline runs display as `¶`.
|
|
193
|
+
|
|
194
|
+
A matching cell's display starts at its first matched line. Earlier lines are replaced by `…[Ln]`, where `n` is the matched line's one-based number within the cell. A leading `#|` directive is retained, as in `#| export…[L4]needle here`.
|
|
195
|
+
|
|
196
|
+
`maxlen` limits displayed source and defaults to 120. The full source remains in `source` and `asdict()`. Each cell appears once even when it has multiple matches. All hits remain in `matches`.
|
|
197
|
+
|
|
198
|
+
`cell_context=N` includes the `N` cells before and after each match as `kind="context"` rows. Context cells are deduplicated within each notebook.
|
|
199
|
+
|
|
200
|
+
`nbrg_iter` yields `NbCell` rows as notebooks are parsed. For collected results, `nbrg` accepts these limits and result options:
|
|
201
|
+
|
|
202
|
+
- `max_results` returns at most that many cells after sorting by path and cell index.
|
|
203
|
+
- `count=True` returns the number of matching cells.
|
|
204
|
+
- `timeout_ms` applies a deadline with the same `stop_reason` values as `rg`.
|
|
205
|
+
|
|
206
|
+
The parser reads only each cell's `id`, `cell_type`, and `source`. It skips outputs and metadata without loading embedded images or plots. `search_nb(pattern, path, ...)` searches a single notebook file in the same way.
|
|
207
|
+
|
|
208
|
+
`rgapi-nbrg` exposes notebook search without requiring a Python kernel:
|
|
209
|
+
|
|
210
|
+
```bash
|
|
211
|
+
rgapi-nbrg 'read_csv' .
|
|
212
|
+
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
213
|
+
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Run `rgapi-nbrg --help` for its discovery, matching, and output options.
|
|
217
|
+
|
|
218
|
+
## Async
|
|
219
|
+
|
|
220
|
+
`fda`, `rga`, and `nbrga` are async versions of `fd`, `rg`, and `nbrg`. The async generators `fda_iter`, `rga_iter`, and `nbrga_iter` yield rows as the search finds them. Async functions accept the same arguments and return the same types as their synchronous equivalents.
|
|
221
|
+
|
|
222
|
+
```python
|
|
223
|
+
from rgapi import fda, rga, rga_iter
|
|
224
|
+
|
|
225
|
+
await fda(".", ext="py")
|
|
226
|
+
res = await rga("TODO", ".", timeout_ms=200)
|
|
227
|
+
if not res.complete: print(f"partial results: {res.stop_reason}")
|
|
228
|
+
async for row in rga_iter("TODO", "."): ...
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
Walking and searching use Rust threads, without `asyncio.to_thread` or the event loop's executor. A callback uses `loop.call_soon_threadsafe` to complete the awaited future or supply results to the generator's queue. The event loop remains unblocked, with normal `contextvars` behaviour.
|
|
232
|
+
|
|
233
|
+
Cancellation stops the Rust workers within about one row. This includes timeouts from `asyncio.wait_for` or `asyncio.timeout`. It also includes task cancellation, such as starlette cancelling a disconnected client's request.
|
|
234
|
+
|
|
235
|
+
Wrap an async iterator in `contextlib.aclosing` when leaving its loop early. `break` alone delays generator finalization until garbage collection. The context manager provides prompt cleanup:
|
|
236
|
+
|
|
237
|
+
```python
|
|
238
|
+
async with aclosing(rga_iter("TODO", ".")) as it:
|
|
239
|
+
async for row in it:
|
|
240
|
+
if enough(row): break
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
Use streaming results for incremental display, such as sending batches to a browser as they arrive. Collected results with `timeout_ms` return what was found before the deadline. Use `asyncio.wait_for` when a timeout should raise instead.
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
## Benchmarks
|
|
247
|
+
|
|
248
|
+
`tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
|
|
249
|
+
|
|
250
|
+
| fixture | rg | rgapi |
|
|
251
|
+
| --- | ---: | ---: |
|
|
252
|
+
| 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
|
|
253
|
+
| 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
|
|
254
|
+
| tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
|
|
255
|
+
|
rgapi-0.1.26/README.md
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
# rgapi
|
|
2
|
+
|
|
3
|
+
`rgapi` provides `fd`-style file discovery and `rg`-style text search from Python without starting a shell command.
|
|
4
|
+
|
|
5
|
+
It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
|
|
6
|
+
|
|
7
|
+
## Overview
|
|
8
|
+
|
|
9
|
+
For common file discovery and search:
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from rgapi import fd, ls, rg, rg_iter
|
|
13
|
+
|
|
14
|
+
fd(".", ext="py", exclude="test_*.py")
|
|
15
|
+
ls("src")
|
|
16
|
+
for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
|
|
17
|
+
rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
from rgapi import nbrg
|
|
24
|
+
|
|
25
|
+
nbrg("read_csv", ".", cell_context=1)
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Walk and search functions have async versions. Streaming versions yield results as the search finds them. See [Async](#async):
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
|
|
32
|
+
|
|
33
|
+
await rga("TODO", ".", ext="py", timeout_ms=200)
|
|
34
|
+
async for row in rga_iter("TODO", "."): print(row)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Use the lower-level functions to compile a regex, search text or a single file, or walk a directory:
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from rgapi import compile, search_path, search_text, walk
|
|
41
|
+
|
|
42
|
+
matcher = compile("TODO")
|
|
43
|
+
matcher.is_match("TODO")
|
|
44
|
+
matcher.finditer("TODO TODO")
|
|
45
|
+
|
|
46
|
+
walk(".")
|
|
47
|
+
search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
|
|
48
|
+
search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Install
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
pip install rgapi
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## File discovery
|
|
58
|
+
|
|
59
|
+
`fd` and `walk` return slash-separated paths relative to `root`. Pass `root` as a `str` or `pathlib.Path`. The sync and async APIs expand `~` and accept `.`, `./`, and paths containing `..`.
|
|
60
|
+
|
|
61
|
+
Discovery uses the `ignore` crate with ripgrep's default filters. It reads `.gitignore`, `.ignore`, and `.rgignore` files. `.rgignore` takes precedence over `.gitignore`. Pass `ignore=False` to disable all ignore-file filtering, including `.rgignore`.
|
|
62
|
+
|
|
63
|
+
Hidden files are skipped unless `hidden=True`. Symlinks are followed only with `follow_links=True`. Use `same_file_system=True` to avoid crossing filesystem boundaries.
|
|
64
|
+
|
|
65
|
+
Traversal runs in parallel without guaranteed result order. Use `sorted(...)` when order matters.
|
|
66
|
+
|
|
67
|
+
`fd` adds filename filters to `walk`. Its `pattern` is a smart-case regex matched against each basename. Lowercase patterns match case-insensitively. A pattern containing uppercase letters is case-sensitive. Use `path_re` to match the slash-separated relative path instead.
|
|
68
|
+
|
|
69
|
+
`include` and `exclude` use glob syntax. `glob=` is an alias for `include=`. A basename glob such as `*.py` also matches nested paths such as `src/app.py`.
|
|
70
|
+
|
|
71
|
+
Filter extensions with `ext="py"` or `ext=["py", "rs"]`. Extension and glob filters must both match. For example, `include="src/*", ext="py"` requires `src/*` and `*.py`, like combining `rg -g` with `-t`.
|
|
72
|
+
|
|
73
|
+
Set `min_depth` and `max_depth` to bound recursion. `max_filesize` skips files above a byte limit.
|
|
74
|
+
|
|
75
|
+
`ls` follows the shell command's listing conventions. It uses `fd` with `max_depth=1`, includes directories, disables ignore rules, and sorts by name. Set `hidden=True` for `ls -a` behaviour. All `fd` filters remain available.
|
|
76
|
+
|
|
77
|
+
`fd_iter` yields `FileEntry` paths as the walk finds them. It accepts every `fd` filter. Stopping iteration ends the walk. It does not accept `timeout_ms`.
|
|
78
|
+
|
|
79
|
+
`path_re` and `skip_path_re` filter slash-separated relative paths using regexes. They select returned paths or searched files without changing traversal. To skip entire subtrees, use `skip_dir` with a glob or `skip_dir_re` with a regex.
|
|
80
|
+
|
|
81
|
+
## Text search
|
|
82
|
+
|
|
83
|
+
`rg` and `rg_iter` return structured `SearchLine` rows. They accept the same filters as `fd`: `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system`.
|
|
84
|
+
|
|
85
|
+
Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behaviour. Use `case_sensitive=False` to force case-insensitive matching.
|
|
86
|
+
|
|
87
|
+
Each `SearchLine` has these fields:
|
|
88
|
+
|
|
89
|
+
```text
|
|
90
|
+
kind 'match', 'before', 'after', or 'context'
|
|
91
|
+
path path relative to root
|
|
92
|
+
line_number 1-based line number
|
|
93
|
+
lnhash exhash-style `lineno|hash|` address for the line
|
|
94
|
+
line line text without the trailing newline
|
|
95
|
+
matches list of (start, end) byte offsets for match rows
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
`rg`, `search_text`, and `search_path` return `SearchResults` by default. This list subclass displays as rg-style multiline text in `str()` and notebook pretty output. `rg_iter` yields rows lazily.
|
|
99
|
+
|
|
100
|
+
`SearchLine` has a structured `repr` and an rg-style `str`. The string display truncates `line` to 180 characters with a trailing `…`. `repr` and `asdict()` retain the full line. `SearchLine.asdict()` returns the fields as a plain Python dictionary.
|
|
101
|
+
|
|
102
|
+
Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of line numbers. The `line_number` field remains available.
|
|
103
|
+
|
|
104
|
+
For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
|
|
105
|
+
|
|
106
|
+
### Path results
|
|
107
|
+
|
|
108
|
+
`fd`, `walk`, and `ls` return `PathResults`. So do `rg` and `nbrg` with `paths=True`. This is a list of `FileEntry` rows.
|
|
109
|
+
|
|
110
|
+
A `FileEntry` is a `str` subclass containing a relative path. Its `stat` property calls `os.lstat` on first access and caches the result. It returns `None` if the path has vanished. `size`, `mtime`, and `is_dir` use the cached stat result. The object also supports ordinary string operations.
|
|
111
|
+
|
|
112
|
+
`PathResults` displays as an `ls -l`-style listing of at most `rgapi.MAX_REPR` rows. A final `… N more` line reports omitted rows. Only displayed rows need stat calls. `str()` returns one plain path per line. `list(res)` also displays plain paths.
|
|
113
|
+
|
|
114
|
+
### Limits and timeouts
|
|
115
|
+
|
|
116
|
+
Set `timeout_ms` on `rg`, `fd`, `walk`, or `ls` to stop at a deadline and return the results collected so far. Their async versions accept it too. Results report why the operation stopped:
|
|
117
|
+
|
|
118
|
+
- `stop_reason=None` means the result is complete.
|
|
119
|
+
- `stop_reason="max_results"` means `max_results` truncated the result.
|
|
120
|
+
- `stop_reason="timeout"` means the deadline was reached.
|
|
121
|
+
|
|
122
|
+
`complete` is true exactly when `stop_reason` is `None`. `count=True` returns a plain integer without a completion flag. It cannot be combined with `timeout_ms`.
|
|
123
|
+
|
|
124
|
+
### Context lines
|
|
125
|
+
|
|
126
|
+
`before_context`, `after_context`, and `context` correspond to `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
|
|
127
|
+
|
|
128
|
+
### Block summaries
|
|
129
|
+
|
|
130
|
+
`rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
rg("TODO", ".", summary=True, context=1, maxlen=120)
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`.
|
|
137
|
+
|
|
138
|
+
Matches display as `path:start-end:source`. Context displays as `path:start-end-source`. With `lnhashs=True`, the range uses copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Newline runs display as `¶`. `maxlen` limits the displayed text without changing `source` or `asdict()`.
|
|
139
|
+
|
|
140
|
+
In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their context. `summary=True` cannot be combined with `paths` or `count`. It can be combined with `lnhash` for copyable block boundaries.
|
|
141
|
+
|
|
142
|
+
## Notebooks
|
|
143
|
+
|
|
144
|
+
`nbrg` searches cell source in Jupyter `.ipynb` files and returns matching cells. Each result identifies the cell by its nbformat cell/message id, which stays stable across edits.
|
|
145
|
+
|
|
146
|
+
Plain `rg` searches escaped notebook JSON, including outputs and metadata. Its line numbers refer to that JSON file. `nbrg` searches the reconstructed cell source and identifies the cell you would edit.
|
|
147
|
+
|
|
148
|
+
```python
|
|
149
|
+
from rgapi import nbrg
|
|
150
|
+
|
|
151
|
+
nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
|
|
152
|
+
nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Notebook discovery, parsing, and matching run together in one parallel Rust pass. Matching uses `rg`'s regex engine with the same `case_sensitive` and `smart_case` behaviour. `nbrg` accepts the discovery filters from `fd` and `rg`, including `include`, `exclude`, `glob`, `hidden`, `max_depth`, and `skip_dir`.
|
|
156
|
+
|
|
157
|
+
`nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
|
|
158
|
+
|
|
159
|
+
```text
|
|
160
|
+
path notebook path relative to root
|
|
161
|
+
cell_index 0-based position of the cell in the notebook
|
|
162
|
+
cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
|
|
163
|
+
cell_type 'code', 'markdown', or 'raw'
|
|
164
|
+
kind 'match' or 'context'
|
|
165
|
+
source full cell source
|
|
166
|
+
matches list of SearchLine rows for the matched lines within the cell
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
`NbCell.asdict()` returns these fields as a plain dictionary. Its `matches` field contains `SearchLine` dictionaries.
|
|
170
|
+
|
|
171
|
+
`str()` and pretty display show one line per cell. Matches use `path:cell_id:source`. Context uses `path:cell_id-source`. Newline runs display as `¶`.
|
|
172
|
+
|
|
173
|
+
A matching cell's display starts at its first matched line. Earlier lines are replaced by `…[Ln]`, where `n` is the matched line's one-based number within the cell. A leading `#|` directive is retained, as in `#| export…[L4]needle here`.
|
|
174
|
+
|
|
175
|
+
`maxlen` limits displayed source and defaults to 120. The full source remains in `source` and `asdict()`. Each cell appears once even when it has multiple matches. All hits remain in `matches`.
|
|
176
|
+
|
|
177
|
+
`cell_context=N` includes the `N` cells before and after each match as `kind="context"` rows. Context cells are deduplicated within each notebook.
|
|
178
|
+
|
|
179
|
+
`nbrg_iter` yields `NbCell` rows as notebooks are parsed. For collected results, `nbrg` accepts these limits and result options:
|
|
180
|
+
|
|
181
|
+
- `max_results` returns at most that many cells after sorting by path and cell index.
|
|
182
|
+
- `count=True` returns the number of matching cells.
|
|
183
|
+
- `timeout_ms` applies a deadline with the same `stop_reason` values as `rg`.
|
|
184
|
+
|
|
185
|
+
The parser reads only each cell's `id`, `cell_type`, and `source`. It skips outputs and metadata without loading embedded images or plots. `search_nb(pattern, path, ...)` searches a single notebook file in the same way.
|
|
186
|
+
|
|
187
|
+
`rgapi-nbrg` exposes notebook search without requiring a Python kernel:
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
rgapi-nbrg 'read_csv' .
|
|
191
|
+
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
192
|
+
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Run `rgapi-nbrg --help` for its discovery, matching, and output options.
|
|
196
|
+
|
|
197
|
+
## Async
|
|
198
|
+
|
|
199
|
+
`fda`, `rga`, and `nbrga` are async versions of `fd`, `rg`, and `nbrg`. The async generators `fda_iter`, `rga_iter`, and `nbrga_iter` yield rows as the search finds them. Async functions accept the same arguments and return the same types as their synchronous equivalents.
|
|
200
|
+
|
|
201
|
+
```python
|
|
202
|
+
from rgapi import fda, rga, rga_iter
|
|
203
|
+
|
|
204
|
+
await fda(".", ext="py")
|
|
205
|
+
res = await rga("TODO", ".", timeout_ms=200)
|
|
206
|
+
if not res.complete: print(f"partial results: {res.stop_reason}")
|
|
207
|
+
async for row in rga_iter("TODO", "."): ...
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
Walking and searching use Rust threads, without `asyncio.to_thread` or the event loop's executor. A callback uses `loop.call_soon_threadsafe` to complete the awaited future or supply results to the generator's queue. The event loop remains unblocked, with normal `contextvars` behaviour.
|
|
211
|
+
|
|
212
|
+
Cancellation stops the Rust workers within about one row. This includes timeouts from `asyncio.wait_for` or `asyncio.timeout`. It also includes task cancellation, such as starlette cancelling a disconnected client's request.
|
|
213
|
+
|
|
214
|
+
Wrap an async iterator in `contextlib.aclosing` when leaving its loop early. `break` alone delays generator finalization until garbage collection. The context manager provides prompt cleanup:
|
|
215
|
+
|
|
216
|
+
```python
|
|
217
|
+
async with aclosing(rga_iter("TODO", ".")) as it:
|
|
218
|
+
async for row in it:
|
|
219
|
+
if enough(row): break
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
Use streaming results for incremental display, such as sending batches to a browser as they arrive. Collected results with `timeout_ms` return what was found before the deadline. Use `asyncio.wait_for` when a timeout should raise instead.
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
## Benchmarks
|
|
226
|
+
|
|
227
|
+
`tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
|
|
228
|
+
|
|
229
|
+
| fixture | rg | rgapi |
|
|
230
|
+
| --- | ---: | ---: |
|
|
231
|
+
| 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
|
|
232
|
+
| 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
|
|
233
|
+
| tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
|
|
@@ -9,7 +9,7 @@ mod walk;
|
|
|
9
9
|
mod python;
|
|
10
10
|
|
|
11
11
|
pub use block::{BlockIter, SearchBlock, block_iter};
|
|
12
|
-
pub use nb::{NbCell, NbIter, NbOptions, nb_iter, nb_search, nb_search_file};
|
|
12
|
+
pub use nb::{NbCell, NbIter, NbOptions, ancestor_indices, heading_level, nb_iter, nb_search, nb_search_file, section_range};
|
|
13
13
|
pub use search::{MatchSpan, RgIter, RgOptions, SearchKind, SearchLine, compile_regex, rg, rg_iter, search_path, search_text};
|
|
14
14
|
pub use walk::{FindIter, FindOptions, StreamIter, find, find_iter};
|
|
15
15
|
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
use std::collections::{BTreeMap, HashMap};
|
|
2
2
|
use std::path::{Path, PathBuf};
|
|
3
|
-
use std::sync::Arc;
|
|
3
|
+
use std::sync::{Arc, LazyLock};
|
|
4
4
|
use std::sync::atomic::Ordering;
|
|
5
5
|
|
|
6
|
+
use grep_matcher::Matcher;
|
|
6
7
|
use grep_regex::RegexMatcher;
|
|
7
8
|
use ignore::{DirEntry, WalkState};
|
|
8
9
|
use serde::Deserialize;
|
|
@@ -11,6 +12,36 @@ use crate::RgApiError;
|
|
|
11
12
|
use crate::search::{SearchLine, compile_regex, search_text};
|
|
12
13
|
use crate::walk::{PathFilters, StreamIter, entry_err, file_root_flags, normalize_root, rel_path, spawn_walk};
|
|
13
14
|
|
|
15
|
+
static HEADING_RE: LazyLock<RegexMatcher> = LazyLock::new(|| RegexMatcher::new(r"^#{1,6} \w").unwrap());
|
|
16
|
+
|
|
17
|
+
/// Heading level of the first nonblank line after skipping `#|` directives; zero unless it matches `^#{1,6} \w`.
|
|
18
|
+
pub fn heading_level(source: &str) -> usize {
|
|
19
|
+
let line = source.lines().find(|l| !l.trim().is_empty() && !l.starts_with("#|")).unwrap_or("");
|
|
20
|
+
if !HEADING_RE.is_match(line.as_bytes()).unwrap_or(false) { return 0; }
|
|
21
|
+
line.bytes().take_while(|&b| b == b'#').count()
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/// A heading and its descendants; a non-heading selects itself. Zero levels represent non-heading cells.
|
|
25
|
+
pub fn section_range(levels: &[usize], idx: usize) -> std::ops::Range<usize> {
|
|
26
|
+
if levels[idx] == 0 { return idx..idx + 1; }
|
|
27
|
+
let end = (idx + 1..levels.len()).find(|&i| levels[i] > 0 && levels[i] <= levels[idx]).unwrap_or(levels.len());
|
|
28
|
+
idx..end
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/// Enclosing heading indices, outermost first, excluding the addressed cell.
|
|
32
|
+
pub fn ancestor_indices(levels: &[usize], idx: usize) -> Vec<usize> {
|
|
33
|
+
let mut level = if levels[idx] == 0 { 7 } else { levels[idx] };
|
|
34
|
+
let mut parents = Vec::new();
|
|
35
|
+
for i in (0..idx).rev() {
|
|
36
|
+
if levels[i] > 0 && levels[i] < level {
|
|
37
|
+
parents.push(i);
|
|
38
|
+
level = levels[i];
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
parents.reverse();
|
|
42
|
+
parents
|
|
43
|
+
}
|
|
44
|
+
|
|
14
45
|
#[derive(Debug, Clone)]
|
|
15
46
|
pub struct NbOptions {
|
|
16
47
|
pub root: PathBuf,
|
|
@@ -104,13 +104,23 @@ pub fn find_iter(opts: &FindOptions) -> Result<FindIter, RgApiError> {
|
|
|
104
104
|
))
|
|
105
105
|
}
|
|
106
106
|
|
|
107
|
-
pub struct StreamIter<T> { rx: mpsc::Receiver<Result<T, RgApiError>>, cancel: Arc<AtomicBool>,
|
|
107
|
+
pub struct StreamIter<T> { rx: mpsc::Receiver<Result<T, RgApiError>>, cancel: Arc<AtomicBool>, worker: Option<std::thread::JoinHandle<()>> }
|
|
108
108
|
|
|
109
109
|
impl<T> StreamIter<T> {
|
|
110
110
|
pub fn cancel(&self) { self.cancel.store(true, Ordering::Relaxed); }
|
|
111
111
|
|
|
112
112
|
pub fn cancel_flag(&self) -> Arc<AtomicBool> { self.cancel.clone() }
|
|
113
113
|
|
|
114
|
+
/// Cancel and wait for the walk's workers to finish. Drain queued sends before
|
|
115
|
+
/// joining so a full result channel cannot deadlock shutdown. Unlike Drop,
|
|
116
|
+
/// this guarantees no background walk remains when it returns. Filesystem
|
|
117
|
+
/// calls already in progress must return first; run this off an async executor.
|
|
118
|
+
pub fn cancel_and_join(mut self) -> Result<(), RgApiError> {
|
|
119
|
+
self.cancel();
|
|
120
|
+
while self.rx.recv().is_ok() {}
|
|
121
|
+
self.worker.take().expect("stream owns its worker").join().map_err(|_| RgApiError::new("search worker panicked"))
|
|
122
|
+
}
|
|
123
|
+
|
|
114
124
|
pub fn next_timeout(&mut self, timeout: std::time::Duration) -> Result<Result<T, RgApiError>, mpsc::RecvTimeoutError> { self.rx.recv_timeout(timeout) }
|
|
115
125
|
|
|
116
126
|
/// Collect all items, stopping at `timeout_ms`; the bool is true when the deadline stopped it.
|
|
@@ -138,6 +148,33 @@ impl<T> Iterator for StreamIter<T> {
|
|
|
138
148
|
|
|
139
149
|
impl<T> Drop for StreamIter<T> { fn drop(&mut self) { self.cancel(); } }
|
|
140
150
|
|
|
151
|
+
#[cfg(test)]
|
|
152
|
+
mod close_tests {
|
|
153
|
+
use super::*;
|
|
154
|
+
|
|
155
|
+
#[test]
|
|
156
|
+
fn cancel_and_join_drains_full_channel_and_waits_for_worker() {
|
|
157
|
+
let (tx, rx) = mpsc::sync_channel(1);
|
|
158
|
+
let (started, ready) = mpsc::channel();
|
|
159
|
+
let cancel = Arc::new(AtomicBool::new(false));
|
|
160
|
+
let worker_cancel = cancel.clone();
|
|
161
|
+
let done = Arc::new(AtomicBool::new(false));
|
|
162
|
+
let worker_done = done.clone();
|
|
163
|
+
let worker = std::thread::spawn(move || {
|
|
164
|
+
tx.send(Ok(1)).unwrap();
|
|
165
|
+
started.send(()).unwrap();
|
|
166
|
+
// This blocks while the channel is full; close must drain, not just join.
|
|
167
|
+
tx.send(Ok(2)).unwrap();
|
|
168
|
+
assert!(worker_cancel.load(Ordering::Acquire));
|
|
169
|
+
worker_done.store(true, Ordering::Release);
|
|
170
|
+
});
|
|
171
|
+
ready.recv().unwrap();
|
|
172
|
+
StreamIter { rx, cancel: cancel.clone(), worker: Some(worker) }.cancel_and_join().unwrap();
|
|
173
|
+
assert!(cancel.load(Ordering::Acquire));
|
|
174
|
+
assert!(done.load(Ordering::Acquire));
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
141
178
|
#[allow(clippy::too_many_arguments)]
|
|
142
179
|
pub fn spawn_walk<T, F>(
|
|
143
180
|
root: PathBuf,
|
|
@@ -181,7 +218,7 @@ where
|
|
|
181
218
|
})
|
|
182
219
|
});
|
|
183
220
|
});
|
|
184
|
-
StreamIter { rx, cancel,
|
|
221
|
+
StreamIter { rx, cancel, worker: Some(worker) }
|
|
185
222
|
}
|
|
186
223
|
|
|
187
224
|
fn find_entry(
|
|
@@ -365,7 +402,7 @@ mod tests {
|
|
|
365
402
|
if tx.send(Ok(i)).is_err() { return; }
|
|
366
403
|
}
|
|
367
404
|
});
|
|
368
|
-
StreamIter { rx, cancel,
|
|
405
|
+
StreamIter { rx, cancel, worker: Some(worker) }
|
|
369
406
|
}
|
|
370
407
|
|
|
371
408
|
#[test]
|
rgapi-0.1.24/PKG-INFO
DELETED
|
@@ -1,201 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: rgapi
|
|
3
|
-
Version: 0.1.24
|
|
4
|
-
Classifier: Programming Language :: Rust
|
|
5
|
-
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
6
|
-
Requires-Dist: fastcore>=1.14.6
|
|
7
|
-
Requires-Dist: fastship>=0.0.12 ; extra == 'dev'
|
|
8
|
-
Requires-Dist: maturin>=1.0,<2.0 ; extra == 'dev'
|
|
9
|
-
Requires-Dist: pytest ; extra == 'dev'
|
|
10
|
-
Provides-Extra: dev
|
|
11
|
-
License-File: LICENSE
|
|
12
|
-
Summary: Python API for ripgrep-style file walking and searching
|
|
13
|
-
Home-Page: https://github.com/AnswerDotAI/rgapi
|
|
14
|
-
Author-email: Jeremy Howard <j@fast.ai>
|
|
15
|
-
License: Apache-2.0
|
|
16
|
-
Requires-Python: >=3.10
|
|
17
|
-
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
18
|
-
Project-URL: Homepage, https://github.com/AnswerDotAI/rgapi
|
|
19
|
-
Project-URL: Issues, https://github.com/AnswerDotAI/rgapi/issues
|
|
20
|
-
Project-URL: Repository, https://github.com/AnswerDotAI/rgapi
|
|
21
|
-
|
|
22
|
-
# rgapi
|
|
23
|
-
|
|
24
|
-
`rgapi` is a Python API for ripgrep-style walking and search. It is meant for Python code that wants `fd`-style file discovery or `rg`-style searching without shelling out.
|
|
25
|
-
|
|
26
|
-
It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
|
|
27
|
-
|
|
28
|
-
## Overview
|
|
29
|
-
|
|
30
|
-
For common file discovery and search:
|
|
31
|
-
|
|
32
|
-
```python
|
|
33
|
-
from rgapi import fd, ls, rg, rg_iter
|
|
34
|
-
|
|
35
|
-
fd(".", ext="py", exclude="test_*.py")
|
|
36
|
-
ls("src")
|
|
37
|
-
for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
|
|
38
|
-
rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
|
|
39
|
-
```
|
|
40
|
-
|
|
41
|
-
For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
|
|
42
|
-
|
|
43
|
-
```python
|
|
44
|
-
from rgapi import nbrg
|
|
45
|
-
|
|
46
|
-
nbrg("read_csv", ".", cell_context=1)
|
|
47
|
-
```
|
|
48
|
-
|
|
49
|
-
Every walk and search has an async twin, plus streaming forms that yield results as they are found (see [Async](#async)):
|
|
50
|
-
|
|
51
|
-
```python
|
|
52
|
-
from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
|
|
53
|
-
|
|
54
|
-
await rga("TODO", ".", ext="py", timeout_ms=200)
|
|
55
|
-
async for row in rga_iter("TODO", "."): print(row)
|
|
56
|
-
```
|
|
57
|
-
|
|
58
|
-
For direct access to the regex, search, and walk pieces:
|
|
59
|
-
|
|
60
|
-
```python
|
|
61
|
-
from rgapi import compile, search_path, search_text, walk
|
|
62
|
-
|
|
63
|
-
matcher = compile("TODO")
|
|
64
|
-
matcher.is_match("TODO")
|
|
65
|
-
matcher.finditer("TODO TODO")
|
|
66
|
-
|
|
67
|
-
walk(".")
|
|
68
|
-
search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
|
|
69
|
-
search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
|
|
70
|
-
```
|
|
71
|
-
|
|
72
|
-
## Install
|
|
73
|
-
|
|
74
|
-
```bash
|
|
75
|
-
pip install rgapi
|
|
76
|
-
```
|
|
77
|
-
|
|
78
|
-
## Semantics
|
|
79
|
-
|
|
80
|
-
`fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
|
|
81
|
-
|
|
82
|
-
`fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
|
|
83
|
-
|
|
84
|
-
`ls` lists like the shell command: it is `fd` with defaults flipped to one level (`max_depth=1`), directories included, ignore rules off, and results sorted by name. `hidden=True` is `ls -a`, and every `fd` filter still applies.
|
|
85
|
-
|
|
86
|
-
`fd_iter` is the lazy form of `fd`, yielding `FileEntry` paths as the walk finds them, and takes every `fd` filter. It has no `timeout_ms`, since a consumer that stops asking for paths ends the walk itself.
|
|
87
|
-
|
|
88
|
-
`path_re` and `skip_path_re` are regex filters on slash-separated relative paths. They filter returned paths or searched files, but do not control traversal. `skip_dir` uses glob syntax to prune matching directory subtrees, and `skip_dir_re` does the same with regex.
|
|
89
|
-
|
|
90
|
-
`rg` and `rg_iter` return structured rows rather than raw CLI text. They accept the same `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system` filters as `fd`. Each row is a `SearchLine` with:
|
|
91
|
-
|
|
92
|
-
```text
|
|
93
|
-
kind 'match', 'before', 'after', or 'context'
|
|
94
|
-
path path relative to root
|
|
95
|
-
line_number 1-based line number
|
|
96
|
-
lnhash exhash-style `lineno|hash|` address for the line
|
|
97
|
-
line line text without the trailing newline
|
|
98
|
-
matches list of (start, end) byte offsets for match rows
|
|
99
|
-
```
|
|
100
|
-
|
|
101
|
-
`rg`, `search_text`, and `search_path` return `SearchResults` by default, a list subclass whose `str()` and notebook pretty display are rg-style multiline text. `rg_iter` yields rows lazily.
|
|
102
|
-
|
|
103
|
-
`SearchLine` has a structured `repr`, an rg-style `str` (the `line` is truncated to 180 chars with a trailing `…` for display; `repr` and `asdict()` keep the full line), and `SearchLine.asdict()` returns row fields as a plain Python dict. Pass `rg(..., lnhashs=True)` or `rg_iter(..., lnhashs=True)` to show `lnhash` addresses instead of line numbers in row display while keeping `line_number` available. `rg(..., paths=True)` returns unique matched paths, and `rg(..., count=True)` returns the total number of match spans. `paths` and `count` cannot both be set.
|
|
104
|
-
|
|
105
|
-
`fd`, `walk`, `ls`, and `rg`/`nbrg` with `paths=True` return `PathResults`, a list of `FileEntry` rows. A `FileEntry` is a `str` subclass holding the relative path, so all string uses keep working, and it stats itself lazily on first access: `stat` is a cached `os.lstat` result (`None` if the path has vanished), with `size`, `mtime`, and `is_dir` derived from it. A `PathResults` displays as an `ls -l`-style long listing, capped at `rgapi.MAX_REPR` rows with a final `… N more` line, so stats are read only for displayed rows; `str()` is still one plain path per line, and `list(res)` shows plain paths. `rg(..., timeout_ms=200)` and `fd(..., timeout_ms=200)` stop at the deadline and return whatever was collected by then; `walk`, `ls`, and the async forms take it too. Results record how they ended: `stop_reason` is `None` for a complete result, `"max_results"` when truncated by `max_results`, or `"timeout"` when a deadline hit, and `complete` is true when `stop_reason` is `None`. `count=True` returns a plain int, which cannot carry the flag, so it rejects `timeout_ms`.
|
|
106
|
-
|
|
107
|
-
`before_context`, `after_context`, and `context` are like `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
|
|
108
|
-
|
|
109
|
-
### Block summaries
|
|
110
|
-
|
|
111
|
-
`rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
|
|
112
|
-
|
|
113
|
-
```python
|
|
114
|
-
rg("TODO", ".", summary=True, context=1, maxlen=120)
|
|
115
|
-
```
|
|
116
|
-
|
|
117
|
-
The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`. Its display is `path:start-end:source` for matches and `path:start-end-source` for context. With `lnhashs=True`, the numeric range becomes copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Embedded newline runs are shown as `¶`; `maxlen` limits displayed source without changing `source` or `asdict()`.
|
|
118
|
-
|
|
119
|
-
In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their block context. `summary=True` cannot be combined with `paths` or `count`; it can be combined with `lnhash` when copyable block boundaries are useful.
|
|
120
|
-
|
|
121
|
-
Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behavior, or `case_sensitive=False` to force case-insensitive matching.
|
|
122
|
-
|
|
123
|
-
## Notebooks
|
|
124
|
-
|
|
125
|
-
`nbrg` searches Jupyter `.ipynb` files cell-by-cell, so results are *cells* rather than raw JSON lines, and each match is identified by its **cell id** (the nbformat cell/message id) rather than a line number. Searching a notebook with plain `rg` matches the escaped JSON text (including outputs and metadata) and reports meaningless JSON line numbers; `nbrg` instead searches each cell's reconstructed **source** and reports the cell id, which is stable across edits and points at the actual unit you work with.
|
|
126
|
-
|
|
127
|
-
```python
|
|
128
|
-
from rgapi import nbrg
|
|
129
|
-
|
|
130
|
-
nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
|
|
131
|
-
nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
|
|
132
|
-
```
|
|
133
|
-
|
|
134
|
-
Notebooks are walked, parsed, and matched together in one parallel Rust pass, using the same regex engine as `rg`, so regex behaviour and the `case_sensitive`/`smart_case` flags match `rg`. Only cell `source` is searched, not outputs or metadata. `nbrg` accepts the same discovery filters as `fd`/`rg` (`include`, `exclude`, `glob`, `hidden`, `max_depth`, `skip_dir`, …).
|
|
135
|
-
|
|
136
|
-
`nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
|
|
137
|
-
|
|
138
|
-
```text
|
|
139
|
-
path notebook path relative to root
|
|
140
|
-
cell_index 0-based position of the cell in the notebook
|
|
141
|
-
cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
|
|
142
|
-
cell_type 'code', 'markdown', or 'raw'
|
|
143
|
-
kind 'match' or 'context'
|
|
144
|
-
source full cell source
|
|
145
|
-
matches list of SearchLine rows for the matched lines within the cell
|
|
146
|
-
```
|
|
147
|
-
|
|
148
|
-
`NbCell.asdict()` returns those fields as a plain dict (with `matches` as `SearchLine` dicts). `str()` and pretty display show one line per cell, newline runs shown as `¶`, keyed by `cell_id`: `path:cell_id:source` for matches and `path:cell_id-source` for context. A match row starts at its first matched line: earlier lines display as `…[Ln]` (`n` the matched line's 1-based number in the cell), except that a leading `#|` directive line is kept, e.g. `#| export…[L4]needle here`. `maxlen` controls the displayed source length and defaults to 120; the full source remains in `source` and `asdict()`. A cell with several matches appears once, with every hit collected in `matches`.
|
|
149
|
-
|
|
150
|
-
`cell_context=N` includes the `N` cells before and after each matching cell as `kind="context"` rows (deduplicated per notebook).
|
|
151
|
-
|
|
152
|
-
`nbrg_iter` yields `NbCell` rows lazily as notebooks are parsed. `nbrg` also accepts `max_results` (at most that many cells, after sorting by path and cell index), `count=True` (number of matching cells), and `timeout_ms=` with the same `stop_reason` semantics as `rg`.
|
|
153
|
-
|
|
154
|
-
Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
|
|
155
|
-
|
|
156
|
-
`rgapi-nbrg` exposes notebook search without requiring a Python kernel:
|
|
157
|
-
|
|
158
|
-
```bash
|
|
159
|
-
rgapi-nbrg 'read_csv' .
|
|
160
|
-
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
161
|
-
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
162
|
-
```
|
|
163
|
-
|
|
164
|
-
Run `rgapi-nbrg --help` for its discovery, matching, and output options.
|
|
165
|
-
|
|
166
|
-
## Async
|
|
167
|
-
|
|
168
|
-
`fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
|
|
169
|
-
|
|
170
|
-
```python
|
|
171
|
-
from rgapi import fda, rga, rga_iter
|
|
172
|
-
|
|
173
|
-
await fda(".", ext="py")
|
|
174
|
-
res = await rga("TODO", ".", timeout_ms=200)
|
|
175
|
-
if not res.complete: print(f"partial results: {res.stop_reason}")
|
|
176
|
-
async for row in rga_iter("TODO", "."): ...
|
|
177
|
-
```
|
|
178
|
-
|
|
179
|
-
None of this uses `asyncio.to_thread` or the loop's executor. The walk and search run on Rust threads, and a single callback settles the awaited future (or feeds the generator's queue) through `loop.call_soon_threadsafe`, so the event loop never blocks and contextvars behave normally.
|
|
180
|
-
|
|
181
|
-
Cancellation cleans up the Rust workers automatically. Wrapping a call in `asyncio.wait_for` or `asyncio.timeout`, cancelling the task (as starlette does when a client disconnects), or leaving an `async for` early all stop the search within about one row. One caveat comes from the language rather than the library: `break` inside `async for` only finalizes the generator at GC time, so for prompt cleanup wrap the iterator in `contextlib.aclosing`:
|
|
182
|
-
|
|
183
|
-
```python
|
|
184
|
-
async with aclosing(rga_iter("TODO", ".")) as it:
|
|
185
|
-
async for row in it:
|
|
186
|
-
if enough(row): break
|
|
187
|
-
```
|
|
188
|
-
|
|
189
|
-
The streaming forms suit incremental display, such as pushing each batch of results to a browser as it arrives. The collected forms with `timeout_ms` give the best results available within a budget, and `asyncio.wait_for` gives timeout-as-failure. Pick per call site.
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
## Benchmarks
|
|
193
|
-
|
|
194
|
-
`tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
|
|
195
|
-
|
|
196
|
-
| fixture | rg | rgapi |
|
|
197
|
-
| --- | ---: | ---: |
|
|
198
|
-
| 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
|
|
199
|
-
| 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
|
|
200
|
-
| tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
|
|
201
|
-
|
rgapi-0.1.24/README.md
DELETED
|
@@ -1,179 +0,0 @@
|
|
|
1
|
-
# rgapi
|
|
2
|
-
|
|
3
|
-
`rgapi` is a Python API for ripgrep-style walking and search. It is meant for Python code that wants `fd`-style file discovery or `rg`-style searching without shelling out.
|
|
4
|
-
|
|
5
|
-
It uses the same `ignore`, `grep-regex`, and `grep-searcher` crates that ripgrep uses for walking, regex matching, and file scanning. Walking and searching run in parallel by default. Most expensive work stays in Rust.
|
|
6
|
-
|
|
7
|
-
## Overview
|
|
8
|
-
|
|
9
|
-
For common file discovery and search:
|
|
10
|
-
|
|
11
|
-
```python
|
|
12
|
-
from rgapi import fd, ls, rg, rg_iter
|
|
13
|
-
|
|
14
|
-
fd(".", ext="py", exclude="test_*.py")
|
|
15
|
-
ls("src")
|
|
16
|
-
for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
|
|
17
|
-
rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
|
|
18
|
-
```
|
|
19
|
-
|
|
20
|
-
For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
|
|
21
|
-
|
|
22
|
-
```python
|
|
23
|
-
from rgapi import nbrg
|
|
24
|
-
|
|
25
|
-
nbrg("read_csv", ".", cell_context=1)
|
|
26
|
-
```
|
|
27
|
-
|
|
28
|
-
Every walk and search has an async twin, plus streaming forms that yield results as they are found (see [Async](#async)):
|
|
29
|
-
|
|
30
|
-
```python
|
|
31
|
-
from rgapi import fda, rga, rga_iter, nbrga, nbrga_iter
|
|
32
|
-
|
|
33
|
-
await rga("TODO", ".", ext="py", timeout_ms=200)
|
|
34
|
-
async for row in rga_iter("TODO", "."): print(row)
|
|
35
|
-
```
|
|
36
|
-
|
|
37
|
-
For direct access to the regex, search, and walk pieces:
|
|
38
|
-
|
|
39
|
-
```python
|
|
40
|
-
from rgapi import compile, search_path, search_text, walk
|
|
41
|
-
|
|
42
|
-
matcher = compile("TODO")
|
|
43
|
-
matcher.is_match("TODO")
|
|
44
|
-
matcher.finditer("TODO TODO")
|
|
45
|
-
|
|
46
|
-
walk(".")
|
|
47
|
-
search_text(matcher, "alpha\nTODO\nomega\n", path="memory.txt", context=1)
|
|
48
|
-
search_path(matcher, "src/lib.rs", display_path="src/lib.rs")
|
|
49
|
-
```
|
|
50
|
-
|
|
51
|
-
## Install
|
|
52
|
-
|
|
53
|
-
```bash
|
|
54
|
-
pip install rgapi
|
|
55
|
-
```
|
|
56
|
-
|
|
57
|
-
## Semantics
|
|
58
|
-
|
|
59
|
-
`fd` and `walk` return slash-separated paths relative to `root`. They use the `ignore` crate, so `.gitignore`, `.ignore`, and the usual ripgrep filters apply by default. `.rgignore` files are also honored and take precedence over `.gitignore`. Hidden files are skipped unless `hidden=True`. Pass `ignore=False` to disable all ignore filtering (including `.rgignore`). Symlinks are not followed unless `follow_links=True`; `same_file_system=True` avoids crossing filesystem boundaries. Traversal is parallel, and result order is not guaranteed; use `sorted(...)` if order matters. `root` arguments accept `str` or `pathlib.Path` and expand `~`; `.`, `./`, and paths containing `..` work across the sync and async APIs.
|
|
60
|
-
|
|
61
|
-
`fd` adds fd-like filtering on top of `walk`: `pattern` is a smart-case regex matched against each basename, and `include`/`exclude` use glob syntax. Lowercase patterns match case-insensitively; a pattern containing uppercase letters is case-sensitive. Use `path_re` when matching the slash-separated relative path instead. `glob=` is accepted as an alias for `include=`. A basename glob such as `*.py` also matches recursively, so it finds `src/app.py`. Use `ext="py"` or `ext=["py", "rs"]` for extension filters, which compose as AND with `include`/`glob` (so `include="src/*", ext="py"` means `src/*` *and* `*.py`, like combining `rg -g` with `-t`); use `min_depth=`/`max_depth=` to bound recursion, and `max_filesize=` to skip files above a byte limit.
|
|
62
|
-
|
|
63
|
-
`ls` lists like the shell command: it is `fd` with defaults flipped to one level (`max_depth=1`), directories included, ignore rules off, and results sorted by name. `hidden=True` is `ls -a`, and every `fd` filter still applies.
|
|
64
|
-
|
|
65
|
-
`fd_iter` is the lazy form of `fd`, yielding `FileEntry` paths as the walk finds them, and takes every `fd` filter. It has no `timeout_ms`, since a consumer that stops asking for paths ends the walk itself.
|
|
66
|
-
|
|
67
|
-
`path_re` and `skip_path_re` are regex filters on slash-separated relative paths. They filter returned paths or searched files, but do not control traversal. `skip_dir` uses glob syntax to prune matching directory subtrees, and `skip_dir_re` does the same with regex.
|
|
68
|
-
|
|
69
|
-
`rg` and `rg_iter` return structured rows rather than raw CLI text. They accept the same `include`, `exclude`, `glob`, `ext`, `path_re`, `skip_path_re`, `skip_dir`, `skip_dir_re`, `min_depth`, `max_depth`, `max_filesize`, `follow_links`, and `same_file_system` filters as `fd`. Each row is a `SearchLine` with:
|
|
70
|
-
|
|
71
|
-
```text
|
|
72
|
-
kind 'match', 'before', 'after', or 'context'
|
|
73
|
-
path path relative to root
|
|
74
|
-
line_number 1-based line number
|
|
75
|
-
lnhash exhash-style `lineno|hash|` address for the line
|
|
76
|
-
line line text without the trailing newline
|
|
77
|
-
matches list of (start, end) byte offsets for match rows
|
|
78
|
-
```
|
|
79
|
-
|
|
80
|
-
`rg`, `search_text`, and `search_path` return `SearchResults` by default, a list subclass whose `str()` and notebook pretty display are rg-style multiline text. `rg_iter` yields rows lazily.
|
|
81
|
-
|
|
82
|
-
`SearchLine` has a structured `repr`, an rg-style `str` (the `line` is truncated to 180 chars with a trailing `…` for display; `repr` and `asdict()` keep the full line), and `SearchLine.asdict()` returns row fields as a plain Python dict. Pass `rg(..., lnhashs=True)` or `rg_iter(..., lnhashs=True)` to show `lnhash` addresses instead of line numbers in row display while keeping `line_number` available. `rg(..., paths=True)` returns unique matched paths, and `rg(..., count=True)` returns the total number of match spans. `paths` and `count` cannot both be set.
|
|
83
|
-
|
|
84
|
-
`fd`, `walk`, `ls`, and `rg`/`nbrg` with `paths=True` return `PathResults`, a list of `FileEntry` rows. A `FileEntry` is a `str` subclass holding the relative path, so all string uses keep working, and it stats itself lazily on first access: `stat` is a cached `os.lstat` result (`None` if the path has vanished), with `size`, `mtime`, and `is_dir` derived from it. A `PathResults` displays as an `ls -l`-style long listing, capped at `rgapi.MAX_REPR` rows with a final `… N more` line, so stats are read only for displayed rows; `str()` is still one plain path per line, and `list(res)` shows plain paths. `rg(..., timeout_ms=200)` and `fd(..., timeout_ms=200)` stop at the deadline and return whatever was collected by then; `walk`, `ls`, and the async forms take it too. Results record how they ended: `stop_reason` is `None` for a complete result, `"max_results"` when truncated by `max_results`, or `"timeout"` when a deadline hit, and `complete` is true when `stop_reason` is `None`. `count=True` returns a plain int, which cannot carry the flag, so it rejects `timeout_ms`.
|
|
85
|
-
|
|
86
|
-
`before_context`, `after_context`, and `context` are like `rg -B`, `rg -A`, and `rg -C`. Files containing NUL bytes or invalid UTF-8 are skipped.
|
|
87
|
-
|
|
88
|
-
### Block summaries
|
|
89
|
-
|
|
90
|
-
`rg(..., summary=True)` returns one row per blank-line-delimited block instead of one row per matching line. Empty and whitespace-only lines delimit blocks. A block containing several matching lines appears once and keeps every matching `SearchLine` in `matches`.
|
|
91
|
-
|
|
92
|
-
```python
|
|
93
|
-
rg("TODO", ".", summary=True, context=1, maxlen=120)
|
|
94
|
-
```
|
|
95
|
-
|
|
96
|
-
The result is `BlockResults`, a list of `SearchBlock` objects. Each block has `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind`, full `source`, and `matches`. Its display is `path:start-end:source` for matches and `path:start-end-source` for context. With `lnhashs=True`, the numeric range becomes copyable boundary addresses such as `path:4|a3f2|,6|b1c3|:source`. Embedded newline runs are shown as `¶`; `maxlen` limits displayed source without changing `source` or `asdict()`.
|
|
97
|
-
|
|
98
|
-
In summary mode, `before_context`, `after_context`, and `context` count neighbouring blocks. `max_results` counts matching blocks and retains their block context. `summary=True` cannot be combined with `paths` or `count`; it can be combined with `lnhash` when copyable block boundaries are useful.
|
|
99
|
-
|
|
100
|
-
Search is case-sensitive by default, matching `rg`. Use `smart_case=True` for `rg --smart-case` behavior, or `case_sensitive=False` to force case-insensitive matching.
|
|
101
|
-
|
|
102
|
-
## Notebooks
|
|
103
|
-
|
|
104
|
-
`nbrg` searches Jupyter `.ipynb` files cell-by-cell, so results are *cells* rather than raw JSON lines, and each match is identified by its **cell id** (the nbformat cell/message id) rather than a line number. Searching a notebook with plain `rg` matches the escaped JSON text (including outputs and metadata) and reports meaningless JSON line numbers; `nbrg` instead searches each cell's reconstructed **source** and reports the cell id, which is stable across edits and points at the actual unit you work with.
|
|
105
|
-
|
|
106
|
-
```python
|
|
107
|
-
from rgapi import nbrg
|
|
108
|
-
|
|
109
|
-
nbrg("read_csv", ".") # cells whose source matches, across all notebooks under "."
|
|
110
|
-
nbrg("read_csv", ".", cell_context=1) # also include neighbouring cells as context
|
|
111
|
-
```
|
|
112
|
-
|
|
113
|
-
Notebooks are walked, parsed, and matched together in one parallel Rust pass, using the same regex engine as `rg`, so regex behaviour and the `case_sensitive`/`smart_case` flags match `rg`. Only cell `source` is searched, not outputs or metadata. `nbrg` accepts the same discovery filters as `fd`/`rg` (`include`, `exclude`, `glob`, `hidden`, `max_depth`, `skip_dir`, …).
|
|
114
|
-
|
|
115
|
-
`nbrg` returns `NbResults`, a list of `NbCell`. Each `NbCell` has:
|
|
116
|
-
|
|
117
|
-
```text
|
|
118
|
-
path notebook path relative to root
|
|
119
|
-
cell_index 0-based position of the cell in the notebook
|
|
120
|
-
cell_id nbformat cell id (falls back to the cell index for notebooks without ids)
|
|
121
|
-
cell_type 'code', 'markdown', or 'raw'
|
|
122
|
-
kind 'match' or 'context'
|
|
123
|
-
source full cell source
|
|
124
|
-
matches list of SearchLine rows for the matched lines within the cell
|
|
125
|
-
```
|
|
126
|
-
|
|
127
|
-
`NbCell.asdict()` returns those fields as a plain dict (with `matches` as `SearchLine` dicts). `str()` and pretty display show one line per cell, newline runs shown as `¶`, keyed by `cell_id`: `path:cell_id:source` for matches and `path:cell_id-source` for context. A match row starts at its first matched line: earlier lines display as `…[Ln]` (`n` the matched line's 1-based number in the cell), except that a leading `#|` directive line is kept, e.g. `#| export…[L4]needle here`. `maxlen` controls the displayed source length and defaults to 120; the full source remains in `source` and `asdict()`. A cell with several matches appears once, with every hit collected in `matches`.
|
|
128
|
-
|
|
129
|
-
`cell_context=N` includes the `N` cells before and after each matching cell as `kind="context"` rows (deduplicated per notebook).
|
|
130
|
-
|
|
131
|
-
`nbrg_iter` yields `NbCell` rows lazily as notebooks are parsed. `nbrg` also accepts `max_results` (at most that many cells, after sorting by path and cell index), `count=True` (number of matching cells), and `timeout_ms=` with the same `stop_reason` semantics as `rg`.
|
|
132
|
-
|
|
133
|
-
Notebook walking, parsing, and matching all happen in parallel in Rust, in the same pass as the file walk. Parsing uses a lean model that reads only each cell's `id`, `cell_type`, and `source` and skips outputs and metadata, so large embedded outputs (images, plots) are never materialized. `search_nb(pattern, path, ...)` searches a single notebook file the same way.
|
|
134
|
-
|
|
135
|
-
`rgapi-nbrg` exposes notebook search without requiring a Python kernel:
|
|
136
|
-
|
|
137
|
-
```bash
|
|
138
|
-
rgapi-nbrg 'read_csv' .
|
|
139
|
-
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
140
|
-
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
141
|
-
```
|
|
142
|
-
|
|
143
|
-
Run `rgapi-nbrg --help` for its discovery, matching, and output options.
|
|
144
|
-
|
|
145
|
-
## Async
|
|
146
|
-
|
|
147
|
-
`fda`, `rga`, and `nbrga` are awaitable twins of `fd`, `rg`, and `nbrg`, and `fda_iter`, `rga_iter` and `nbrga_iter` are async generators that yield rows as the search finds them. All take the same arguments and return the same types as their sync counterparts.
|
|
148
|
-
|
|
149
|
-
```python
|
|
150
|
-
from rgapi import fda, rga, rga_iter
|
|
151
|
-
|
|
152
|
-
await fda(".", ext="py")
|
|
153
|
-
res = await rga("TODO", ".", timeout_ms=200)
|
|
154
|
-
if not res.complete: print(f"partial results: {res.stop_reason}")
|
|
155
|
-
async for row in rga_iter("TODO", "."): ...
|
|
156
|
-
```
|
|
157
|
-
|
|
158
|
-
None of this uses `asyncio.to_thread` or the loop's executor. The walk and search run on Rust threads, and a single callback settles the awaited future (or feeds the generator's queue) through `loop.call_soon_threadsafe`, so the event loop never blocks and contextvars behave normally.
|
|
159
|
-
|
|
160
|
-
Cancellation cleans up the Rust workers automatically. Wrapping a call in `asyncio.wait_for` or `asyncio.timeout`, cancelling the task (as starlette does when a client disconnects), or leaving an `async for` early all stop the search within about one row. One caveat comes from the language rather than the library: `break` inside `async for` only finalizes the generator at GC time, so for prompt cleanup wrap the iterator in `contextlib.aclosing`:
|
|
161
|
-
|
|
162
|
-
```python
|
|
163
|
-
async with aclosing(rga_iter("TODO", ".")) as it:
|
|
164
|
-
async for row in it:
|
|
165
|
-
if enough(row): break
|
|
166
|
-
```
|
|
167
|
-
|
|
168
|
-
The streaming forms suit incremental display, such as pushing each batch of results to a browser as it arrives. The collected forms with `timeout_ms` give the best results available within a budget, and `asyncio.wait_for` gives timeout-as-failure. Pick per call site.
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
## Benchmarks
|
|
172
|
-
|
|
173
|
-
`tools/bench.py` compares the `rg` CLI with in-process `rgapi`. Run it against a release build. One run on this machine, using best time from seven repeats:
|
|
174
|
-
|
|
175
|
-
| fixture | rg | rgapi |
|
|
176
|
-
| --- | ---: | ---: |
|
|
177
|
-
| 6 x 2 MB files, 2 matches | 6.54 ms | 1.44 ms |
|
|
178
|
-
| 800 x 1.5 KB files, 2 matches | 13.90 ms | 10.94 ms |
|
|
179
|
-
| tiny dir, repeated 30x | 5.92 ms | 2.14 ms |
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|