rgapi 0.1.30__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rgapi-0.1.30 → rgapi-0.2.0}/Cargo.lock +3 -25
- {rgapi-0.1.30 → rgapi-0.2.0}/Cargo.toml +1 -1
- {rgapi-0.1.30 → rgapi-0.2.0}/DEV.md +5 -5
- {rgapi-0.1.30 → rgapi-0.2.0}/PKG-INFO +7 -3
- {rgapi-0.1.30 → rgapi-0.2.0}/README.md +6 -2
- {rgapi-0.1.30 → rgapi-0.2.0}/python/rgapi/__init__.py +66 -53
- {rgapi-0.1.30 → rgapi-0.2.0}/python/rgapi/_cli.py +3 -3
- {rgapi-0.1.30 → rgapi-0.2.0}/python/rgapi/block.py +6 -1
- {rgapi-0.1.30 → rgapi-0.2.0}/python/rgapi/nb.py +26 -20
- rgapi-0.2.0/python/rgapi/skill.py +19 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/src/block.rs +11 -24
- {rgapi-0.1.30 → rgapi-0.2.0}/src/lib.rs +1 -1
- {rgapi-0.1.30 → rgapi-0.2.0}/src/nb.rs +14 -43
- {rgapi-0.1.30 → rgapi-0.2.0}/src/python.rs +40 -739
- {rgapi-0.1.30 → rgapi-0.2.0}/src/search.rs +13 -75
- {rgapi-0.1.30 → rgapi-0.2.0}/src/walk.rs +128 -121
- {rgapi-0.1.30 → rgapi-0.2.0}/tests/test_rgapi.py +25 -0
- rgapi-0.1.30/python/rgapi/skill.py +0 -37
- {rgapi-0.1.30 → rgapi-0.2.0}/.github/workflows/ci.yml +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/.gitignore +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/LICENSE +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/_config.yml +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/_layouts/default.html +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/pyproject.toml +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/rustfmt.toml +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/tests/test_async.py +0 -0
- {rgapi-0.1.30 → rgapi-0.2.0}/tools/bench.py +0 -0
|
@@ -70,13 +70,12 @@ checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
|
|
|
70
70
|
|
|
71
71
|
[[package]]
|
|
72
72
|
name = "encoding_rs"
|
|
73
|
-
version = "0.8.
|
|
73
|
+
version = "0.8.42"
|
|
74
74
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
75
|
-
checksum = "
|
|
75
|
+
checksum = "8e985e0451871ad22fb8d2b6b076e2028a502a0d3950998c2c5c0a4f9b5d9679"
|
|
76
76
|
dependencies = [
|
|
77
77
|
"cfg-if",
|
|
78
78
|
"core_detect",
|
|
79
|
-
"multiversion",
|
|
80
79
|
"multiversion_no_op",
|
|
81
80
|
"rustversion",
|
|
82
81
|
"scopeguard",
|
|
@@ -197,27 +196,6 @@ dependencies = [
|
|
|
197
196
|
"libc",
|
|
198
197
|
]
|
|
199
198
|
|
|
200
|
-
[[package]]
|
|
201
|
-
name = "multiversion"
|
|
202
|
-
version = "0.9.0"
|
|
203
|
-
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
204
|
-
checksum = "b4ca4bea16ffc3f443cf7d866912118196bfef4c6a1556ca00f9f9b00bb43f7c"
|
|
205
|
-
dependencies = [
|
|
206
|
-
"multiversion-macros",
|
|
207
|
-
]
|
|
208
|
-
|
|
209
|
-
[[package]]
|
|
210
|
-
name = "multiversion-macros"
|
|
211
|
-
version = "0.9.0"
|
|
212
|
-
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
213
|
-
checksum = "0d416831a7317ef4b08bee00b69cbbb9c8763da7959a7026244d6266869f9c83"
|
|
214
|
-
dependencies = [
|
|
215
|
-
"proc-macro2",
|
|
216
|
-
"quote",
|
|
217
|
-
"rustversion",
|
|
218
|
-
"syn 3.0.6",
|
|
219
|
-
]
|
|
220
|
-
|
|
221
199
|
[[package]]
|
|
222
200
|
name = "multiversion_no_op"
|
|
223
201
|
version = "1.0.0"
|
|
@@ -330,7 +308,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
|
|
|
330
308
|
|
|
331
309
|
[[package]]
|
|
332
310
|
name = "rgapi"
|
|
333
|
-
version = "0.
|
|
311
|
+
version = "0.2.0"
|
|
334
312
|
dependencies = [
|
|
335
313
|
"crc32fast",
|
|
336
314
|
"globset",
|
|
@@ -13,7 +13,7 @@ python/rgapi/ public Python wrappers over `rgapi._core`, plus the `rgapi-nbr
|
|
|
13
13
|
tests/ pytest coverage for the Python API
|
|
14
14
|
```
|
|
15
15
|
|
|
16
|
-
The public Python API lives in `python/rgapi/__init__.py`. The extension module is private as `rgapi._core`; keep crate-like functions there and put Python-facing argument policy in the wrapper when that stays concise. For example, `glob=` and `ext=` are Python wrapper conveniences over the core include glob list.
|
|
16
|
+
The public Python API lives in `python/rgapi/__init__.py`. The extension module is private as `rgapi._core`; keep crate-like functions there and put Python-facing argument policy in the wrapper when that stays concise. For example, `glob=` and `ext=` are Python wrapper conveniences over the core include glob list. Every walking `_core` function takes the walk options as one dict, which PyO3 reads into `WalkOptions`. `_walk(root, **kwargs)` builds that dict. A new walk parameter adds a field to `WalkOptions` and a parameter to `_walk_args`. No `_core` signature changes.
|
|
17
17
|
|
|
18
18
|
## Commands
|
|
19
19
|
|
|
@@ -38,20 +38,20 @@ The GitHub workflow builds wheels for Python 3.10-3.13 on Linux and macOS and pu
|
|
|
38
38
|
|
|
39
39
|
## Design notes
|
|
40
40
|
|
|
41
|
-
Python discovery and `paths=True` results contain absolute `pathlib.Path` objects. Structured search rows
|
|
41
|
+
Python discovery and `paths=True` results contain absolute `pathlib.Path` objects. Structured search rows hold base-relative string labels with `/` separators. Traversal uses `ignore::WalkParallel`, so result order is not part of the API contract. Search results are structured rows; collected result lists use rg-style `str()` and notebook display. `SearchLine.lnhash` is computed with the same CRC-32-based line-content hash format as exhash (`lineno|hash|`, low 12 bits of CRC-32 over the line's UTF-8 bytes, encoded as two Base64url characters); `lnhashs=True` only changes row display, not `line_number` or matching behavior. Path regexes filter returned/searched paths; `skip_dir` and `skip_dir_re` prune traversal through `ignore::WalkBuilder::filter_entry`. Depth, size, filesystem, hidden, and ignore options use `ignore::WalkBuilder` settings. The `ignore` walker follows every root it is given. Discovery checks each root with `symlink_metadata`. The worker sends each unfollowed link root as a result before walking the other roots. This also supports dangling roots, which the underlying walker would reject. `ls` sets `walk_root_links`, which gives a root link to a directory to the walker. The walker then reports paths under the link. Other discovery roots use absolute paths without canonicalizing. Content searches retain canonical root resolution. `rg_iter` exposes the same parallel search stream that `rg` collects by default; `paths=True` and `count=True` consume that stream with different reducers. Text search skips binary files and invalid UTF-8 content.
|
|
42
42
|
|
|
43
43
|
Streaming engine: `walk.rs` owns the generic machinery. `StreamIter<T>` is the worker-thread-plus-bounded-channel iterator (`sync_channel(8192)`, so producers block rather than buffer without limit when a consumer lags), and `spawn_walk` owns the shared scaffold: walker config, panic catching, cancel flag, and worker thread. `rg_iter` (`T = SearchLine`), `block_iter` (`T = SearchBlock`), `nb_iter` (`T = NbCell`), and `find_iter` (`T = PathBuf`, the path walk) plug entry closures into that engine. Block search reads each file once, searches it once, groups nonblank lines into blocks, maps matching lines to their blocks, and expands context by block index. Each `SearchBlock` carries numeric boundaries plus hashes for its first and last source lines. Python keeps both and chooses the displayed address without another file read.
|
|
44
44
|
|
|
45
|
+
`resolve_roots` makes the roots absolute, or canonical for content searches, and drops repeats. It also computes the base that result paths and filters are relative to. A directory root is its own base. A file root, or a link root that is not followed, has its parent as its base. Several roots use the common ancestor of their bases. `spawn_walk` walks the roots one after another on its worker thread, with one `ignore` walker per root. Each walker applies the file-root rules to its own root. Entry closures receive the base, not the root. When one root is inside another, a shared set of visited paths lets each path reach the entry closure once. Python gets the same base from `_core.walk_base`.
|
|
46
|
+
|
|
45
47
|
Async API: `fda`, `fda_iter`, `rga`, `rga_iter`, `nbrga`, and `nbrga_iter` wrap the corresponding private core operations. `rga(summary=True)` uses `_core.block_search_async`; ordinary `rga` uses `_core.rg_async`. Each collected core function takes a Python callback, runs on Rust threads through the generic `stream_async` helper, and delivers with one GIL attach at the end. Iterator forms use `stream_iter_async` and attach once per batch. The Python side settles an `asyncio.Future` or feeds an `asyncio.Queue` via `loop.call_soon_threadsafe`; no Python thread blocks and `asyncio.to_thread` is not involved. `AsyncHandle.cancel()` sets the same atomic flag used by the Rust iterators.
|
|
46
48
|
|
|
47
49
|
Truncation is recorded on collected results: `max_results` sets `stop_reason="max_results"`, and `timeout_ms` on `rg`/`rga`/`nbrg`/`nbrga`/`fd`/`fda`/`walk`/`ls` sets `stop_reason="timeout"`. `SearchResults`, `BlockResults`, `PathResults`, and `NbResults` share this through `_Results`; `complete` means `stop_reason is None`. In block summary mode, `max_results` counts matching blocks and keeps their block context. `count=True` returns a plain int, so it rejects timeouts and block summary mode.
|
|
48
50
|
|
|
49
|
-
Rust discovery returns
|
|
51
|
+
Rust discovery returns base-relative `PathBuf` values. Glob filters operate on native paths; only regex matching uses string labels. PyO3 converts discovery roots and results with its filesystem-path support. Python joins each result to the base from `_core.walk_base` without resolving links. `PathResults` stores that display base and completion status, including across slices. Its `__repr__` uses `Path.lstat()` for an `ls -l`-style listing capped at `MAX_REPR` rows. `str()` returns one base-relative name per line. `ls` is `fd` with shell-style defaults (one level, dirs, ignore rules off) and `walk_root_links`, sorted in place.
|
|
50
52
|
|
|
51
53
|
Rust callers can consume a `StreamIter` with `cancel_and_join()` to cancel, drain queued results, and wait for the walk's workers to finish. `Drop` remains nonblocking. Since filesystem calls already in progress must return before joining completes, keep `cancel_and_join()` off async executors.
|
|
52
54
|
|
|
53
55
|
Notebook hierarchy uses `heading_level(source)`, `section_range(levels, idx)` and `ancestor_indices(levels, idx)`. Callers supply zero for non-heading cells. Heading detection skips blank lines and lines starting with `#|`, then checks the first remaining line against `^#{1,6} \w`. It does not search past ordinary text. Section ranges include the addressed cell and end before the next equal-or-higher heading; non-headings select themselves. Ancestors exclude the addressed cell and are returned outermost first. These are calculations over the supplied levels, without retained outline state. Rustygate uses them for its cell selectors.
|
|
54
56
|
|
|
55
57
|
`cell_refs(cell)` takes an nbformat cell as a `serde_json::Value` and returns its sigil references as `CellRefs { vars, cmds, tools }`, in order of appearance, with duplicates. A prompt cell has `solveit_ai: true` in its metadata. `vars` holds each `expr` written as `` $`expr` `` in a prompt cell's source. `cmds` holds each `cmd` written as `` !`cmd` `` in a prompt cell's source. `tools` holds each name written as `` &`name` `` or `` &`[a, b]` ``. A name holds word characters and dots. `cell_refs` reads `tools` from the source of a prompt or Markdown cell. For every other cell it reads the `text/markdown` data of `display_data` and `execute_result` outputs. It never reads a prompt cell's outputs. This function is Rust-only. Rustygate uses it for the cells API's `refs=true`.
|
|
56
|
-
|
|
57
|
-
This package intentionally has no CLI. Python is the interface.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rgapi
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
6
6
|
Requires-Dist: fastcore>=2.2.29
|
|
@@ -36,6 +36,7 @@ fd(".", ext="py", exclude="test_*.py")
|
|
|
36
36
|
ls("src")
|
|
37
37
|
for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
|
|
38
38
|
rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
|
|
39
|
+
rg("TODO", ["src", "tests"], ext="py")
|
|
39
40
|
```
|
|
40
41
|
|
|
41
42
|
For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
|
|
@@ -77,7 +78,9 @@ pip install rgapi
|
|
|
77
78
|
|
|
78
79
|
## File discovery
|
|
79
80
|
|
|
80
|
-
`fd` and `walk` return absolute `pathlib.Path` objects. Use them directly with `.read_text()`, `.open()`, `.stat()`, or other filesystem operations. Their collected results display names relative to `root`. Pass `root` as a `str` or `Path
|
|
81
|
+
`fd` and `walk` return absolute `pathlib.Path` objects. Use them directly with `.read_text()`, `.open()`, `.stat()`, or other filesystem operations. Their collected results display names relative to `root`. Pass `root` as a `str` or `Path`, or as a list of them. The sync and async APIs expand `~` and accept `.`, `./`, and paths containing `..`.
|
|
82
|
+
|
|
83
|
+
A list of roots, such as `rg("TODO", ["src", "tests"])`, works with every walk and search function. The roots are searched as one walk that returns one result list. `timeout_ms` and `max_results` apply to the whole walk. Result paths are relative to the common ancestor of the roots. For example, rows read `src/app.py` and `tests/test_app.py`. `PathResults` displays use the same relative paths. Filters such as `include`, `path_re` and `skip_dir` match these relative paths. A file under more than one root, such as `src/app.py` with roots `[".", "src"]`, appears once.
|
|
81
84
|
|
|
82
85
|
Discovery uses the `ignore` crate with ripgrep's default filters. It reads `.gitignore`, `.ignore`, and `.rgignore` files. `.rgignore` takes precedence over `.gitignore`. Pass `ignore=False` to disable all ignore-file filtering, including `.rgignore`.
|
|
83
86
|
|
|
@@ -142,7 +145,7 @@ Use `.name`, `.suffix`, and `.relative_to(root)` for path components. `.stat()`
|
|
|
142
145
|
|
|
143
146
|
Structured text and notebook search rows retain root-relative string labels in their `path` fields. Content searches still follow explicitly named root links. This differs from discovery's default of returning the link itself.
|
|
144
147
|
|
|
145
|
-
The Rust `find` and `find_iter` APIs return native `PathBuf` values relative to the root
|
|
148
|
+
The Rust `find` and `find_iter` APIs return native `PathBuf` values relative to the root, or to the common ancestor of several roots in `WalkOptions::roots`. For an explicitly named file or unfollowed root link, the result is its basename.
|
|
146
149
|
|
|
147
150
|
Rust callers can set `FindOptions::special_files` to also discover FIFOs, sockets, and device nodes, for example to reject unsupported entries during archiving. This defaults to false; normal discovery returns regular files, directories when requested, and symlinks.
|
|
148
151
|
|
|
@@ -224,6 +227,7 @@ The parser reads only each cell's `id`, `cell_type`, and `source`. It skips outp
|
|
|
224
227
|
```bash
|
|
225
228
|
rgapi-nbrg 'read_csv' .
|
|
226
229
|
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
230
|
+
rgapi-nbrg 'read_csv' nbs tests
|
|
227
231
|
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
228
232
|
```
|
|
229
233
|
|
|
@@ -15,6 +15,7 @@ fd(".", ext="py", exclude="test_*.py")
|
|
|
15
15
|
ls("src")
|
|
16
16
|
for row in rg_iter("TODO", ".", include="*.py", context=2): print(row.asdict())
|
|
17
17
|
rg("TODO", ".", ext="py", skip_dir=".venv", paths=True)
|
|
18
|
+
rg("TODO", ["src", "tests"], ext="py")
|
|
18
19
|
```
|
|
19
20
|
|
|
20
21
|
For cell-aware search of Jupyter notebooks (see [Notebooks](#notebooks)):
|
|
@@ -56,7 +57,9 @@ pip install rgapi
|
|
|
56
57
|
|
|
57
58
|
## File discovery
|
|
58
59
|
|
|
59
|
-
`fd` and `walk` return absolute `pathlib.Path` objects. Use them directly with `.read_text()`, `.open()`, `.stat()`, or other filesystem operations. Their collected results display names relative to `root`. Pass `root` as a `str` or `Path
|
|
60
|
+
`fd` and `walk` return absolute `pathlib.Path` objects. Use them directly with `.read_text()`, `.open()`, `.stat()`, or other filesystem operations. Their collected results display names relative to `root`. Pass `root` as a `str` or `Path`, or as a list of them. The sync and async APIs expand `~` and accept `.`, `./`, and paths containing `..`.
|
|
61
|
+
|
|
62
|
+
A list of roots, such as `rg("TODO", ["src", "tests"])`, works with every walk and search function. The roots are searched as one walk that returns one result list. `timeout_ms` and `max_results` apply to the whole walk. Result paths are relative to the common ancestor of the roots. For example, rows read `src/app.py` and `tests/test_app.py`. `PathResults` displays use the same relative paths. Filters such as `include`, `path_re` and `skip_dir` match these relative paths. A file under more than one root, such as `src/app.py` with roots `[".", "src"]`, appears once.
|
|
60
63
|
|
|
61
64
|
Discovery uses the `ignore` crate with ripgrep's default filters. It reads `.gitignore`, `.ignore`, and `.rgignore` files. `.rgignore` takes precedence over `.gitignore`. Pass `ignore=False` to disable all ignore-file filtering, including `.rgignore`.
|
|
62
65
|
|
|
@@ -121,7 +124,7 @@ Use `.name`, `.suffix`, and `.relative_to(root)` for path components. `.stat()`
|
|
|
121
124
|
|
|
122
125
|
Structured text and notebook search rows retain root-relative string labels in their `path` fields. Content searches still follow explicitly named root links. This differs from discovery's default of returning the link itself.
|
|
123
126
|
|
|
124
|
-
The Rust `find` and `find_iter` APIs return native `PathBuf` values relative to the root
|
|
127
|
+
The Rust `find` and `find_iter` APIs return native `PathBuf` values relative to the root, or to the common ancestor of several roots in `WalkOptions::roots`. For an explicitly named file or unfollowed root link, the result is its basename.
|
|
125
128
|
|
|
126
129
|
Rust callers can set `FindOptions::special_files` to also discover FIFOs, sockets, and device nodes, for example to reject unsupported entries during archiving. This defaults to false; normal discovery returns regular files, directories when requested, and symlinks.
|
|
127
130
|
|
|
@@ -203,6 +206,7 @@ The parser reads only each cell's `id`, `cell_type`, and `source`. It skips outp
|
|
|
203
206
|
```bash
|
|
204
207
|
rgapi-nbrg 'read_csv' .
|
|
205
208
|
rgapi-nbrg 'read_csv' . --cell-context 1
|
|
209
|
+
rgapi-nbrg 'read_csv' nbs tests
|
|
206
210
|
rgapi-nbrg 'read_csv' nbs --glob '*.ipynb' --max-results 20
|
|
207
211
|
```
|
|
208
212
|
|
|
@@ -3,7 +3,7 @@ from contextlib import aclosing
|
|
|
3
3
|
from datetime import datetime
|
|
4
4
|
from stat import S_ISLNK, filemode
|
|
5
5
|
|
|
6
|
-
from os import fspath
|
|
6
|
+
from os import PathLike, fspath
|
|
7
7
|
from pathlib import Path
|
|
8
8
|
from fastcore.meta import delegates
|
|
9
9
|
|
|
@@ -41,12 +41,11 @@ def _hsize(n):
|
|
|
41
41
|
n /= 1024
|
|
42
42
|
return f"{n:.0f}" if u == "B" else f"{n:.1f}{u}"
|
|
43
43
|
|
|
44
|
-
def _path_base(root, follow_links=True):
|
|
45
|
-
root = Path(root).absolute()
|
|
46
|
-
return root if (follow_links or not root.is_symlink()) and root.is_dir() else root.parent
|
|
47
|
-
|
|
48
44
|
class PathResults(_Results):
|
|
49
|
-
"Absolute `Path` objects with root-relative plain and `ls -l`-style displays
|
|
45
|
+
"""Absolute `Path` objects with root-relative plain and `ls -l`-style displays.
|
|
46
|
+
|
|
47
|
+
The display is an ls-style table capped at `MAX_REPR` names. `str(res)` gives one root-relative name per line, and `list(res)`
|
|
48
|
+
the absolute Paths. Slices keep the display root and completion status."""
|
|
50
49
|
def __init__(self, paths=(), root=".", show_target=False):
|
|
51
50
|
self.root,self.show_target = Path(root).absolute(),show_target
|
|
52
51
|
super().__init__(self.root/p for p in paths)
|
|
@@ -94,7 +93,7 @@ def _context(context, before_context, after_context):
|
|
|
94
93
|
|
|
95
94
|
|
|
96
95
|
def walk(
|
|
97
|
-
root:str|Path=".", # Directory or file to walk (expands `~`)
|
|
96
|
+
root:str|Path|list=".", # Directory or file to walk, or a list of them (expands `~`)
|
|
98
97
|
hidden:bool=False, # Include hidden files and directories
|
|
99
98
|
ignore:bool=True, # Respect `.gitignore` and other ignore files
|
|
100
99
|
max_depth:int|None=None, # Maximum directory depth to descend
|
|
@@ -111,10 +110,9 @@ def walk(
|
|
|
111
110
|
timeout_ms:int|None=None, # Cancel the walk after this long and return partial results
|
|
112
111
|
) -> PathResults:
|
|
113
112
|
"Walk a directory and return absolute file and/or directory Paths."
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
return _mk_results(PathResults, paths, False, timed_out, root=_path_base(rt, follow_links))
|
|
113
|
+
return _find(root, None, files, dirs, timeout_ms, hidden=hidden, ignore=ignore, max_depth=max_depth, min_depth=min_depth,
|
|
114
|
+
max_filesize=max_filesize, follow_links=follow_links, same_file_system=same_file_system, path_re=path_re, skip_path_re=skip_path_re,
|
|
115
|
+
skip_dir=skip_dir, skip_dir_re=skip_dir_re)
|
|
118
116
|
|
|
119
117
|
|
|
120
118
|
def _walk_args(
|
|
@@ -134,14 +132,26 @@ def _walk_args(
|
|
|
134
132
|
skip_dir:str|list|None=None, # Directory glob or globs to prune
|
|
135
133
|
skip_dir_re:str|None=None, # Directory regex used to prune traversal
|
|
136
134
|
):
|
|
137
|
-
"
|
|
135
|
+
"`_core` walk options from the shared walk parameters. Delegators pass their `**kwargs` here whole."
|
|
138
136
|
include, exclude, exts = _filters(glob, include, exclude, ext)
|
|
139
|
-
return (include, exclude, exts, hidden, ignore, max_depth, min_depth,
|
|
140
|
-
follow_links, same_file_system, path_re, skip_path_re,
|
|
137
|
+
return dict(includes=include, excludes=exclude, exts=exts, hidden=hidden, ignore=ignore, max_depth=max_depth, min_depth=min_depth,
|
|
138
|
+
max_filesize=max_filesize, follow_links=follow_links, same_file_system=same_file_system, path_re=path_re, skip_path_re=skip_path_re,
|
|
139
|
+
skip_dirs=_listify(skip_dir), skip_dir_re=skip_dir_re)
|
|
140
|
+
|
|
141
|
+
def _walk(root, **kwargs):
|
|
142
|
+
"`_core` walk options for `root`, which is one path or an iterable of paths, and the shared walk parameters"
|
|
143
|
+
roots = [root] if isinstance(root, (str, PathLike)) else root
|
|
144
|
+
return dict(roots=[Path(r).expanduser() for r in roots], **_walk_args(**kwargs))
|
|
145
|
+
|
|
146
|
+
def _find(root, pattern=None, files=True, dirs=False, timeout_ms=None, show_target=False, root_links=False, **kwargs):
|
|
147
|
+
"Collected `_core.find` results as `PathResults`. `root_links` walks a root link to a directory, as `ls` does."
|
|
148
|
+
w = _walk(root, **kwargs)
|
|
149
|
+
paths, timed_out = _core.find(w, pattern, files, dirs, timeout_ms, root_links)
|
|
150
|
+
return _mk_results(PathResults, paths, False, timed_out, root=_core.walk_base(w, False, root_links), show_target=show_target)
|
|
141
151
|
|
|
142
152
|
@delegates(_walk_args)
|
|
143
153
|
def fd(
|
|
144
|
-
root:str|Path=".", # Directory or file to walk (expands `~`)
|
|
154
|
+
root:str|Path|list=".", # Directory or file to walk, or a list of them (expands `~`)
|
|
145
155
|
pattern:str|None=None, # Smart-case regex matched against each basename
|
|
146
156
|
files:bool=True, # Include files in results
|
|
147
157
|
dirs:bool=False, # Include directories in results
|
|
@@ -150,28 +160,26 @@ def fd(
|
|
|
150
160
|
**kwargs
|
|
151
161
|
) -> PathResults:
|
|
152
162
|
"Find absolute Paths with fd-style filters and gitignore support."
|
|
153
|
-
|
|
154
|
-
paths, timed_out = _core.find(rt, pattern, *_walk_args(**kwargs), files, dirs, timeout_ms)
|
|
155
|
-
return _mk_results(PathResults, paths, False, timed_out, root=_path_base(rt, kwargs.get('follow_links', False)), show_target=show_target)
|
|
163
|
+
return _find(root, pattern, files, dirs, timeout_ms, show_target, **kwargs)
|
|
156
164
|
|
|
157
165
|
|
|
158
166
|
@delegates(_walk_args)
|
|
159
167
|
def fd_iter(
|
|
160
|
-
root:str|Path=".", # Directory or file to walk (expands `~`)
|
|
168
|
+
root:str|Path|list=".", # Directory or file to walk, or a list of them (expands `~`)
|
|
161
169
|
pattern:str|None=None, # Smart-case regex matched against each basename
|
|
162
170
|
files:bool=True, # Include files in results
|
|
163
171
|
dirs:bool=False, # Include directories in results
|
|
164
172
|
**kwargs
|
|
165
173
|
):
|
|
166
174
|
"Walk lazily, yielding absolute Paths; early exit stops the walk."
|
|
167
|
-
|
|
168
|
-
base =
|
|
169
|
-
return (base/p for p in _core.find_iter(
|
|
175
|
+
w = _walk(root, **kwargs)
|
|
176
|
+
base = _core.walk_base(w, False, False)
|
|
177
|
+
return (base/p for p in _core.find_iter(w, pattern, files, dirs))
|
|
170
178
|
|
|
171
179
|
|
|
172
180
|
@delegates(fd)
|
|
173
181
|
def ls(
|
|
174
|
-
root:str|Path=".", # Directory or file to list (expands `~`)
|
|
182
|
+
root:str|Path|list=".", # Directory or file to list, or a list of them (expands `~`)
|
|
175
183
|
pattern:str|None=None, # Smart-case regex matched against each basename
|
|
176
184
|
hidden:bool=False, # Include hidden files and directories, like `ls -a`
|
|
177
185
|
dirs:bool=True, # Include directories in results
|
|
@@ -179,8 +187,8 @@ def ls(
|
|
|
179
187
|
ignore:bool=False, # Respect `.gitignore` and other ignore files
|
|
180
188
|
**kwargs
|
|
181
189
|
) -> PathResults:
|
|
182
|
-
"List a directory like `ls`: one level, directories included, ignore rules off, sorted by name."
|
|
183
|
-
res =
|
|
190
|
+
"List a directory like `ls`: one level, directories included, ignore rules off, sorted by name. A root link to a directory lists that directory, with paths under the link."
|
|
191
|
+
res = _find(root, pattern, dirs=dirs, root_links=True, hidden=hidden, max_depth=max_depth, ignore=ignore, **kwargs)
|
|
184
192
|
res.sort()
|
|
185
193
|
return res
|
|
186
194
|
|
|
@@ -203,22 +211,23 @@ async def _acall(fn, *args):
|
|
|
203
211
|
|
|
204
212
|
@delegates(fd)
|
|
205
213
|
async def fda(
|
|
206
|
-
root:str|Path=".", # Directory or file to walk (expands `~`)
|
|
214
|
+
root:str|Path|list=".", # Directory or file to walk, or a list of them (expands `~`)
|
|
207
215
|
pattern:str|None=None, # Smart-case regex matched against each basename
|
|
208
216
|
files:bool=True, # Include files in results
|
|
209
217
|
dirs:bool=False, # Include directories in results
|
|
210
218
|
timeout_ms:int|None=None, # Cancel the walk after this long and return partial results
|
|
219
|
+
show_target:bool=False, # Append `-> target` to symlink rows in the display
|
|
211
220
|
**kwargs
|
|
212
221
|
) -> PathResults:
|
|
213
222
|
"Async `fd`: find paths on Rust threads without blocking the event loop."
|
|
214
|
-
|
|
215
|
-
paths, timed_out = await _acall(_core.find_async,
|
|
216
|
-
return _mk_results(PathResults, paths, False, timed_out, root=
|
|
223
|
+
w = _walk(root, **kwargs)
|
|
224
|
+
paths, timed_out = await _acall(_core.find_async, w, pattern, files, dirs, timeout_ms)
|
|
225
|
+
return _mk_results(PathResults, paths, False, timed_out, root=_core.walk_base(w, False, False), show_target=show_target)
|
|
217
226
|
|
|
218
227
|
|
|
219
228
|
@delegates(_walk_args)
|
|
220
229
|
async def fda_iter(
|
|
221
|
-
root:str|Path=".", # Directory or file to walk (expands `~`)
|
|
230
|
+
root:str|Path|list=".", # Directory or file to walk, or a list of them (expands `~`)
|
|
222
231
|
pattern:str|None=None, # Smart-case regex matched against each basename
|
|
223
232
|
files:bool=True, # Include files in results
|
|
224
233
|
dirs:bool=False, # Include directories in results
|
|
@@ -226,10 +235,9 @@ async def fda_iter(
|
|
|
226
235
|
**kwargs
|
|
227
236
|
):
|
|
228
237
|
"Async `fd_iter`: yield absolute Paths; early exit stops the walk."
|
|
229
|
-
|
|
230
|
-
base =
|
|
231
|
-
async with aclosing(_abatches(_core.find_iter_async, batch_max,
|
|
232
|
-
*_walk_args(**kwargs), files, dirs)) as batches:
|
|
238
|
+
w = _walk(root, **kwargs)
|
|
239
|
+
base = _core.walk_base(w, False, False)
|
|
240
|
+
async with aclosing(_abatches(_core.find_iter_async, batch_max, w, pattern, files, dirs)) as batches:
|
|
233
241
|
async for paths in batches:
|
|
234
242
|
for p in paths: yield base/p
|
|
235
243
|
|
|
@@ -258,7 +266,7 @@ def _mk_results(cls, items, capped, timed_out, **kwargs):
|
|
|
258
266
|
return res
|
|
259
267
|
|
|
260
268
|
|
|
261
|
-
def _paths_reduce(rows,
|
|
269
|
+
def _paths_reduce(rows, w, max_results, timed_out=False):
|
|
262
270
|
"Unique matched paths as `PathResults` from rows with `kind`/`path`, capped at `max_results`"
|
|
263
271
|
seen,res,capped = set(),[],False
|
|
264
272
|
for row in rows:
|
|
@@ -268,20 +276,20 @@ def _paths_reduce(rows, root, max_results, timed_out=False):
|
|
|
268
276
|
break
|
|
269
277
|
seen.add(row.path)
|
|
270
278
|
res.append(row.path)
|
|
271
|
-
return _mk_results(PathResults, res, capped, timed_out, root=
|
|
279
|
+
return _mk_results(PathResults, res, capped, timed_out, root=_core.walk_base(w, True, False))
|
|
272
280
|
|
|
273
|
-
def _rg_post(rows, paths, count, max_results, timed_out,
|
|
281
|
+
def _rg_post(rows, paths, count, max_results, timed_out, w):
|
|
274
282
|
"Reduce collected rows to the requested `rg`/`rga` result form"
|
|
275
283
|
if count: return sum(len(r.matches) for r in rows if r.kind == "match")
|
|
276
284
|
if not paths: return _mk_results(SearchResults, *_cap_rows(rows, max_results), timed_out)
|
|
277
|
-
return _paths_reduce(rows,
|
|
285
|
+
return _paths_reduce(rows, w, max_results, timed_out)
|
|
278
286
|
|
|
279
287
|
|
|
280
288
|
|
|
281
289
|
@delegates(_walk_args)
|
|
282
290
|
def rg(
|
|
283
291
|
pattern:str, # Regex pattern to search for
|
|
284
|
-
root:str|Path=".", # Directory or file to search (expands `~`)
|
|
292
|
+
root:str|Path|list=".", # Directory or file to search, or a list of them (expands `~`)
|
|
285
293
|
case_sensitive:bool|None=None, # True/False forces case; None allows `smart_case`
|
|
286
294
|
smart_case:bool=False, # Match `rg --smart-case` behavior
|
|
287
295
|
before_context:int=0, # Lines of context before each match, like `rg -B`
|
|
@@ -296,28 +304,34 @@ def rg(
|
|
|
296
304
|
maxlen:int=MAXLEN, # Maximum source characters per displayed block
|
|
297
305
|
**kwargs
|
|
298
306
|
):
|
|
299
|
-
"Search files and return `SearchResults`, matched paths, or a count; `lnhashs=True` shows exhash-style addresses.
|
|
307
|
+
"""Search files and return `SearchResults`, matched paths, or a count; `lnhashs=True` shows exhash-style addresses.
|
|
308
|
+
|
|
309
|
+
`summary=True` returns one `SearchBlock` per block of lines separated by blank or whitespace-only lines, however many matches
|
|
310
|
+
it holds, and `context` then counts blocks. Summary mode can't combine with `paths` or `count`, and with `lnhashs=True` shows
|
|
311
|
+
each block's boundary addresses. `^` and `$` anchor each line, and a line ends at LF or CRLF. Patterns containing a newline or
|
|
312
|
+
carriage return are rejected. Row `path` fields are labels relative to the root, or to the common ancestor of several roots.
|
|
313
|
+
Each row has `asdict()`. A symlink named as `root` is followed."""
|
|
300
314
|
assert not (paths and count), "paths and count are mutually exclusive"
|
|
301
315
|
assert not (count and max_results), "count and max_results are mutually exclusive"
|
|
302
316
|
assert not (count and timeout_ms is not None), "count and timeout_ms are mutually exclusive"
|
|
303
317
|
assert not (summary and count), "summary and count are mutually exclusive"
|
|
304
318
|
assert not (summary and paths), "summary and paths are mutually exclusive"
|
|
305
319
|
before_context, after_context = _context(context, before_context, after_context)
|
|
306
|
-
|
|
307
|
-
args = (
|
|
320
|
+
w = _walk(root, **kwargs)
|
|
321
|
+
args = (w, pattern, case_sensitive, smart_case, before_context, after_context)
|
|
308
322
|
if summary:
|
|
309
323
|
rows,timed_out = _core.block_search(*args, timeout_ms)
|
|
310
324
|
return _block_post(rows, max_results, before_context, after_context, timed_out, maxlen, lnhashs)
|
|
311
325
|
if count: return sum(len(row.matches) for row in _core.rg_iter(*args, False) if row.kind == "match")
|
|
312
|
-
if paths and timeout_ms is None: return _paths_reduce(_core.rg_iter(*args, False),
|
|
326
|
+
if paths and timeout_ms is None: return _paths_reduce(_core.rg_iter(*args, False), w, max_results)
|
|
313
327
|
rows, timed_out = _core.rg(*args, lnhashs, timeout_ms)
|
|
314
|
-
return _rg_post(rows, paths, False, max_results, timed_out,
|
|
328
|
+
return _rg_post(rows, paths, False, max_results, timed_out, w)
|
|
315
329
|
|
|
316
330
|
|
|
317
331
|
@delegates(_walk_args)
|
|
318
332
|
def rg_iter(
|
|
319
333
|
pattern:str, # Regex pattern to search for
|
|
320
|
-
root:str|Path=".", # Directory or file to search (expands `~`)
|
|
334
|
+
root:str|Path|list=".", # Directory or file to search, or a list of them (expands `~`)
|
|
321
335
|
case_sensitive:bool|None=None, # True/False forces case; None allows `smart_case`
|
|
322
336
|
smart_case:bool=False, # Match `rg --smart-case` behavior
|
|
323
337
|
before_context:int=0, # Lines of context before each match, like `rg -B`
|
|
@@ -328,14 +342,13 @@ def rg_iter(
|
|
|
328
342
|
) -> RgIter:
|
|
329
343
|
"Search files lazily, yielding `SearchLine` rows; `lnhashs=True` shows exhash-style addresses."
|
|
330
344
|
before_context, after_context = _context(context, before_context, after_context)
|
|
331
|
-
return _core.rg_iter(
|
|
332
|
-
case_sensitive, smart_case, before_context, after_context, lnhashs)
|
|
345
|
+
return _core.rg_iter(_walk(root, **kwargs), pattern, case_sensitive, smart_case, before_context, after_context, lnhashs)
|
|
333
346
|
|
|
334
347
|
|
|
335
348
|
@delegates(_walk_args)
|
|
336
349
|
async def rga(
|
|
337
350
|
pattern:str, # Regex pattern to search for
|
|
338
|
-
root:str|Path=".", # Directory or file to search (expands `~`)
|
|
351
|
+
root:str|Path|list=".", # Directory or file to search, or a list of them (expands `~`)
|
|
339
352
|
case_sensitive:bool|None=None, # True/False forces case; None allows `smart_case`
|
|
340
353
|
smart_case:bool=False, # Match `rg --smart-case` behavior
|
|
341
354
|
before_context:int=0, # Lines of context before each match, like `rg -B`
|
|
@@ -357,13 +370,13 @@ async def rga(
|
|
|
357
370
|
assert not (summary and count), "summary and count are mutually exclusive"
|
|
358
371
|
assert not (summary and paths), "summary and paths are mutually exclusive"
|
|
359
372
|
before_context, after_context = _context(context, before_context, after_context)
|
|
360
|
-
|
|
361
|
-
args = (
|
|
373
|
+
w = _walk(root, **kwargs)
|
|
374
|
+
args = (w, pattern, case_sensitive, smart_case, before_context, after_context)
|
|
362
375
|
if summary:
|
|
363
376
|
rows,timed_out = await _acall(_core.block_search_async, *args, timeout_ms)
|
|
364
377
|
return _block_post(rows, max_results, before_context, after_context, timed_out, maxlen, lnhashs)
|
|
365
378
|
rows, timed_out = await _acall(_core.rg_async, *args, lnhashs, timeout_ms)
|
|
366
|
-
return _rg_post(rows, paths, count, max_results, timed_out,
|
|
379
|
+
return _rg_post(rows, paths, count, max_results, timed_out, w)
|
|
367
380
|
|
|
368
381
|
|
|
369
382
|
|
|
@@ -387,7 +400,7 @@ async def _abatches(fn, *args):
|
|
|
387
400
|
@delegates(_walk_args)
|
|
388
401
|
async def rga_iter(
|
|
389
402
|
pattern:str, # Regex pattern to search for
|
|
390
|
-
root:str|Path=".", # Directory or file to search (expands `~`)
|
|
403
|
+
root:str|Path|list=".", # Directory or file to search, or a list of them (expands `~`)
|
|
391
404
|
case_sensitive:bool|None=None, # True/False forces case; None allows `smart_case`
|
|
392
405
|
smart_case:bool=False, # Match `rg --smart-case` behavior
|
|
393
406
|
before_context:int=0, # Lines of context before each match, like `rg -B`
|
|
@@ -399,7 +412,7 @@ async def rga_iter(
|
|
|
399
412
|
):
|
|
400
413
|
"Async `rg_iter`: yield `SearchLine` rows as they are found; early exit cancels the search."
|
|
401
414
|
before_context, after_context = _context(context, before_context, after_context)
|
|
402
|
-
async with aclosing(_abatches(_core.rg_iter_async, batch_max,
|
|
415
|
+
async with aclosing(_abatches(_core.rg_iter_async, batch_max, _walk(root, **kwargs), pattern,
|
|
403
416
|
case_sensitive, smart_case, before_context, after_context, lnhashs)) as batches:
|
|
404
417
|
async for rows in batches:
|
|
405
418
|
for row in rows: yield row
|
|
@@ -4,10 +4,10 @@ from fastcore.script import call_parse
|
|
|
4
4
|
from . import nbrg
|
|
5
5
|
|
|
6
6
|
|
|
7
|
-
@call_parse
|
|
7
|
+
@call_parse
|
|
8
8
|
def nbrg_cli(
|
|
9
9
|
pattern:str, # Regex pattern to search for
|
|
10
|
-
|
|
10
|
+
*roots:str, # Files or directories to search (default `.`)
|
|
11
11
|
cell_context:int=0, # Neighbouring cells to include before and after matches
|
|
12
12
|
multiline:bool=False, # Allow matches across lines within a cell?
|
|
13
13
|
smart_case:bool=False, # Use case-sensitive matching when the pattern contains uppercase?
|
|
@@ -23,6 +23,6 @@ def nbrg_cli(
|
|
|
23
23
|
timeout_ms:int=None, # Stop searching after this many milliseconds
|
|
24
24
|
):
|
|
25
25
|
"Search notebook cell sources and report stable cell IDs."
|
|
26
|
-
print(nbrg(pattern,
|
|
26
|
+
print(nbrg(pattern, roots or '.', cell_context=cell_context, multiline=multiline, smart_case=smart_case,
|
|
27
27
|
case_sensitive=True if case else None, paths=paths, count=count, max_results=max_results, maxlen=maxlen,
|
|
28
28
|
glob=glob, exclude=exclude, hidden=hidden, max_depth=max_depth, timeout_ms=timeout_ms))
|
|
@@ -4,7 +4,12 @@ from . import MAXLEN, _Results, _mk_results, _preview
|
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
class SearchBlock:
|
|
7
|
-
"A blank-line-delimited source block containing a match, or context for one.
|
|
7
|
+
"""A blank-line-delimited source block containing a match, or context for one.
|
|
8
|
+
|
|
9
|
+
Fields: `path`, `block_index`, `start_line`, `end_line`, `start_lnhash`, `end_lnhash`, `kind` (match or context), the full
|
|
10
|
+
`source`, and `matches` as matching `SearchLine`s. The display is `path:start-end:source`, with `-` in place of the last colon
|
|
11
|
+
on context rows. Hashed locations are `start_lnhash,end_lnhash`, or one hash for a single line. Newline runs display as ¶,
|
|
12
|
+
keeping indentation. `maxlen` limits the displayed source, not the stored `source`."""
|
|
8
13
|
def __init__(self, path, block_index, start_line, end_line, start_lnhash, end_lnhash, kind, source, matches, maxlen=MAXLEN, display_lnhash=False):
|
|
9
14
|
self.path,self.block_index,self.start_line,self.end_line = path,block_index,start_line,end_line
|
|
10
15
|
self.start_lnhash,self.end_lnhash,self.display_lnhash = start_lnhash,end_lnhash,display_lnhash
|