rgapi 0.1.28__tar.gz → 0.1.30__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rgapi-0.1.28 → rgapi-0.1.30}/Cargo.lock +1 -1
- {rgapi-0.1.28 → rgapi-0.1.30}/Cargo.toml +1 -1
- {rgapi-0.1.28 → rgapi-0.1.30}/DEV.md +2 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/PKG-INFO +4 -2
- {rgapi-0.1.28 → rgapi-0.1.30}/README.md +2 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/pyproject.toml +2 -2
- {rgapi-0.1.28 → rgapi-0.1.30}/python/rgapi/skill.py +2 -2
- {rgapi-0.1.28 → rgapi-0.1.30}/src/lib.rs +1 -1
- {rgapi-0.1.28 → rgapi-0.1.30}/src/nb.rs +51 -1
- {rgapi-0.1.28 → rgapi-0.1.30}/src/search.rs +5 -5
- {rgapi-0.1.28 → rgapi-0.1.30}/tests/test_rgapi.py +11 -2
- {rgapi-0.1.28 → rgapi-0.1.30}/.github/workflows/ci.yml +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/.gitignore +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/LICENSE +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/_config.yml +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/_layouts/default.html +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/python/rgapi/__init__.py +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/python/rgapi/_cli.py +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/python/rgapi/block.py +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/python/rgapi/nb.py +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/rustfmt.toml +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/src/block.rs +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/src/python.rs +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/src/walk.rs +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/tests/test_async.py +0 -0
- {rgapi-0.1.28 → rgapi-0.1.30}/tools/bench.py +0 -0
|
@@ -52,4 +52,6 @@ Rust callers can consume a `StreamIter` with `cancel_and_join()` to cancel, drai
|
|
|
52
52
|
|
|
53
53
|
Notebook hierarchy uses `heading_level(source)`, `section_range(levels, idx)` and `ancestor_indices(levels, idx)`. Callers supply zero for non-heading cells. Heading detection skips blank lines and lines starting with `#|`, then checks the first remaining line against `^#{1,6} \w`. It does not search past ordinary text. Section ranges include the addressed cell and end before the next equal-or-higher heading; non-headings select themselves. Ancestors exclude the addressed cell and are returned outermost first. These are calculations over the supplied levels, without retained outline state. Rustygate uses them for its cell selectors.
|
|
54
54
|
|
|
55
|
+
`cell_refs(cell)` takes an nbformat cell as a `serde_json::Value` and returns its sigil references as `CellRefs { vars, cmds, tools }`, in order of appearance, with duplicates. A prompt cell has `solveit_ai: true` in its metadata. `vars` holds each `expr` written as `` $`expr` `` in a prompt cell's source. `cmds` holds each `cmd` written as `` !`cmd` `` in a prompt cell's source. `tools` holds each name written as `` &`name` `` or `` &`[a, b]` ``. A name holds word characters and dots. `cell_refs` reads `tools` from the source of a prompt or Markdown cell. For every other cell it reads the `text/markdown` data of `display_data` and `execute_result` outputs. It never reads a prompt cell's outputs. This function is Rust-only. Rustygate uses it for the cells API's `refs=true`.
|
|
56
|
+
|
|
55
57
|
This package intentionally has no CLI. Python is the interface.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rgapi
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.30
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
6
6
|
Requires-Dist: fastcore>=2.2.29
|
|
@@ -12,7 +12,7 @@ License-File: LICENSE
|
|
|
12
12
|
Summary: Python API for ripgrep-style file walking and searching
|
|
13
13
|
Home-Page: https://github.com/AnswerDotAI/rgapi
|
|
14
14
|
Author-email: Jeremy Howard <j@fast.ai>
|
|
15
|
-
License: Apache-2.0
|
|
15
|
+
License-Expression: Apache-2.0
|
|
16
16
|
Requires-Python: >=3.10
|
|
17
17
|
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
18
18
|
Project-URL: Homepage, https://github.com/AnswerDotAI/rgapi
|
|
@@ -124,6 +124,8 @@ Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of li
|
|
|
124
124
|
|
|
125
125
|
For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
|
|
126
126
|
|
|
127
|
+
`^` and `$` match at the start and end of each line. A line ends at `\n` or `\r\n`. `$` also matches before a `\r` that has no `\n` after it. `rg`, `rgstr`, and `nbrg` without `multiline=True` raise `ValueError` for a pattern that contains a literal `\n` or `\r`.
|
|
128
|
+
|
|
127
129
|
### Path results
|
|
128
130
|
|
|
129
131
|
`fd`, `walk`, and `ls` return `PathResults`, a list of absolute `Path` objects. So do `rg` and `nbrg` with `paths=True`, and their async equivalents. Indexing or iterating returns ordinary Paths:
|
|
@@ -103,6 +103,8 @@ Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of li
|
|
|
103
103
|
|
|
104
104
|
For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
|
|
105
105
|
|
|
106
|
+
`^` and `$` match at the start and end of each line. A line ends at `\n` or `\r\n`. `$` also matches before a `\r` that has no `\n` after it. `rg`, `rgstr`, and `nbrg` without `multiline=True` raise `ValueError` for a pattern that contains a literal `\n` or `\r`.
|
|
107
|
+
|
|
106
108
|
### Path results
|
|
107
109
|
|
|
108
110
|
`fd`, `walk`, and `ls` return `PathResults`, a list of absolute `Path` objects. So do `rg` and `nbrg` with `paths=True`, and their async equivalents. Indexing or iterating returns ordinary Paths:
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
[build-system]
|
|
2
|
-
requires = ["maturin>=1.
|
|
2
|
+
requires = ["maturin>=1.9.2,<2.0"]
|
|
3
3
|
build-backend = "maturin"
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "rgapi"
|
|
7
7
|
dynamic = ["version"]
|
|
8
8
|
description = "Python API for ripgrep-style file walking and searching"
|
|
9
|
-
license =
|
|
9
|
+
license = "Apache-2.0"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
readme = "README.md"
|
|
12
12
|
authors = [{name = "Jeremy Howard", email = "j@fast.ai"}]
|
|
@@ -6,8 +6,8 @@ For orientation, start with `rg(summary=True)`; use line-level results where nee
|
|
|
6
6
|
|
|
7
7
|
## Search units
|
|
8
8
|
|
|
9
|
-
- `rg`: lines, or `SearchBlock` rows with `summary=True`. Blank/whitespace-only lines separate blocks; multiple matches in one block yield one row. Context counts the selected unit. Summary mode cannot combine with `paths` or `count`, but supports hashed block boundaries.
|
|
10
|
-
- `nbrg`: `NbResults` of `NbCell` rows from source only, never metadata/outputs. `multiline=True` matches across cell lines while `^`/`$` remain line anchors; ordinary line-oriented `rg` rejects newline
|
|
9
|
+
- `rg`: lines, or `SearchBlock` rows with `summary=True`. Blank/whitespace-only lines separate blocks; multiple matches in one block yield one row. Context counts the selected unit. Summary mode cannot combine with `paths` or `count`, but supports hashed block boundaries. `^`/`$` anchor each line. A line ends at LF or CRLF.
|
|
10
|
+
- `nbrg`: `NbResults` of `NbCell` rows from source only, never metadata/outputs. `multiline=True` matches across cell lines while `^`/`$` remain line anchors; ordinary line-oriented `rg` rejects patterns containing a newline or carriage return.
|
|
11
11
|
- Traversal/search run in parallel in Rust; sort when stable order is required. `path_re`/`skip_path_re` filter paths without pruning; `skip_dir`/`skip_dir_re` prune subtrees.
|
|
12
12
|
|
|
13
13
|
## Result fields and display
|
|
@@ -9,7 +9,7 @@ mod walk;
|
|
|
9
9
|
mod python;
|
|
10
10
|
|
|
11
11
|
pub use block::{BlockIter, SearchBlock, block_iter};
|
|
12
|
-
pub use nb::{NbCell, NbIter, NbOptions, ancestor_indices, heading_level, nb_iter, nb_search, nb_search_file, section_range};
|
|
12
|
+
pub use nb::{CellRefs, NbCell, NbIter, NbOptions, ancestor_indices, cell_refs, heading_level, nb_iter, nb_search, nb_search_file, section_range};
|
|
13
13
|
pub use search::{MatchSpan, RgIter, RgOptions, SearchKind, SearchLine, compile_regex, rg, rg_iter, search_path, search_text};
|
|
14
14
|
pub use walk::{FindIter, FindOptions, StreamIter, find, find_iter};
|
|
15
15
|
|
|
@@ -3,7 +3,7 @@ use std::path::{Path, PathBuf};
|
|
|
3
3
|
use std::sync::{Arc, LazyLock};
|
|
4
4
|
use std::sync::atomic::Ordering;
|
|
5
5
|
|
|
6
|
-
use grep_matcher::Matcher;
|
|
6
|
+
use grep_matcher::{Captures, Matcher};
|
|
7
7
|
use grep_regex::RegexMatcher;
|
|
8
8
|
use ignore::{DirEntry, WalkState};
|
|
9
9
|
use serde::Deserialize;
|
|
@@ -42,6 +42,56 @@ pub fn ancestor_indices(levels: &[usize], idx: usize) -> Vec<usize> {
|
|
|
42
42
|
parents
|
|
43
43
|
}
|
|
44
44
|
|
|
45
|
+
/// Group 1 of every `` sigil`body` `` match in `text`, in order of appearance. `sigil` and `body` are regex fragments.
|
|
46
|
+
fn sigil_caps(text: &str, sigil: &str, body: &str) -> Vec<String> {
|
|
47
|
+
let re = RegexMatcher::new(&format!("{sigil}`({body})`")).expect("sigil pattern compiles");
|
|
48
|
+
let mut caps = re.new_captures().expect("captures allocate");
|
|
49
|
+
let mut res = Vec::new();
|
|
50
|
+
re.captures_iter(text.as_bytes(), &mut caps, |c| {
|
|
51
|
+
if let Some(m) = c.get(1) { res.push(text[m.start()..m.end()].to_string()); }
|
|
52
|
+
true
|
|
53
|
+
}).expect("regex search is infallible");
|
|
54
|
+
res
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/// Names written as `` &`name` `` or `` &`[a, b]` `` in `text`. A name holds word characters and dots.
|
|
58
|
+
fn tool_names(text: &str) -> Vec<String> {
|
|
59
|
+
let groups = sigil_caps(text, "&", r"[\w.]+|\[[\w.,\s]+\]");
|
|
60
|
+
groups.iter().flat_map(|g| g.split(['[', ']', ','])).map(str::trim).filter(|s| !s.is_empty()).map(String::from).collect()
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/// nbformat multiline text, a string or a list of strings, as one string.
|
|
64
|
+
fn nb_text(v: &serde_json::Value) -> String {
|
|
65
|
+
match v { serde_json::Value::String(s) => s.clone(), serde_json::Value::Array(a) => a.iter().filter_map(|o| o.as_str()).collect(), _ => String::new() }
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/// The sigil references in one notebook cell, in order of appearance.
|
|
69
|
+
#[derive(Debug, Default, Clone, PartialEq)]
|
|
70
|
+
pub struct CellRefs {
|
|
71
|
+
/// Each `expr` written as `` $`expr` ``
|
|
72
|
+
pub vars: Vec<String>,
|
|
73
|
+
/// Each `cmd` written as `` !`cmd` ``
|
|
74
|
+
pub cmds: Vec<String>,
|
|
75
|
+
/// Each name written as `` &`name` `` or `` &`[a, b]` ``
|
|
76
|
+
pub tools: Vec<String>,
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/// The sigil references in `cell`, an nbformat cell. A prompt cell has `solveit_ai: true` in its metadata.
|
|
80
|
+
/// `vars` and `cmds` come from a prompt cell's source. `tools` comes from the source of a prompt or Markdown cell.
|
|
81
|
+
/// For every other cell, `tools` comes from the `text/markdown` data of its `display_data` and `execute_result` outputs.
|
|
82
|
+
/// A prompt cell's outputs are never read.
|
|
83
|
+
pub fn cell_refs(cell: &serde_json::Value) -> CellRefs {
|
|
84
|
+
let src = nb_text(&cell["source"]);
|
|
85
|
+
let prompt = cell["metadata"]["solveit_ai"] == true;
|
|
86
|
+
let mut res = CellRefs::default();
|
|
87
|
+
if prompt { (res.vars, res.cmds) = (sigil_caps(&src, r"\$", "[^`]+"), sigil_caps(&src, "!", "[^`]+")); }
|
|
88
|
+
if prompt || cell["cell_type"] == "markdown" { res.tools = tool_names(&src); return res; }
|
|
89
|
+
for o in cell["outputs"].as_array().into_iter().flatten() {
|
|
90
|
+
if matches!(o["output_type"].as_str(), Some("display_data" | "execute_result")) { res.tools.extend(tool_names(&nb_text(&o["data"]["text/markdown"]))); }
|
|
91
|
+
}
|
|
92
|
+
res
|
|
93
|
+
}
|
|
94
|
+
|
|
45
95
|
#[derive(Debug, Clone)]
|
|
46
96
|
pub struct NbOptions {
|
|
47
97
|
pub root: PathBuf,
|
|
@@ -5,7 +5,7 @@ use std::sync::{
|
|
|
5
5
|
mpsc::SyncSender,
|
|
6
6
|
};
|
|
7
7
|
|
|
8
|
-
use grep_matcher::Matcher;
|
|
8
|
+
use grep_matcher::{LineTerminator, Matcher};
|
|
9
9
|
use grep_regex::{RegexMatcher, RegexMatcherBuilder};
|
|
10
10
|
use grep_searcher::{BinaryDetection, SearcherBuilder, Sink, SinkContext, SinkContextKind, SinkError, SinkMatch};
|
|
11
11
|
use ignore::{DirEntry, WalkState};
|
|
@@ -167,8 +167,8 @@ fn send_search_error(tx: &SyncSender<Result<SearchLine, RgApiError>>, err: RgApi
|
|
|
167
167
|
pub fn compile_regex(pattern: &str, case_sensitive: Option<bool>, smart_case: bool, multiline: bool) -> Result<RegexMatcher, RgApiError> {
|
|
168
168
|
if pattern.is_empty() { return Err(RgApiError::new("pattern may not be empty")); }
|
|
169
169
|
let mut builder = RegexMatcherBuilder::new();
|
|
170
|
-
|
|
171
|
-
|
|
170
|
+
builder.multi_line(true).crlf(true);
|
|
171
|
+
if multiline { builder.line_terminator(None); }
|
|
172
172
|
match case_sensitive {
|
|
173
173
|
Some(true) => {
|
|
174
174
|
builder.case_insensitive(false);
|
|
@@ -208,7 +208,7 @@ fn search_path_cancelable(
|
|
|
208
208
|
cancel: Option<Arc<AtomicBool>>,
|
|
209
209
|
) -> Result<Vec<SearchLine>, RgApiError> {
|
|
210
210
|
let mut builder = SearcherBuilder::new();
|
|
211
|
-
builder.line_number(true).before_context(before_context).after_context(after_context).binary_detection(BinaryDetection::quit(0));
|
|
211
|
+
builder.line_number(true).line_terminator(LineTerminator::crlf()).before_context(before_context).after_context(after_context).binary_detection(BinaryDetection::quit(0));
|
|
212
212
|
let mut searcher = builder.build();
|
|
213
213
|
let mut out = Vec::new();
|
|
214
214
|
let search_matcher = matcher.clone();
|
|
@@ -236,7 +236,7 @@ fn search_bytes(
|
|
|
236
236
|
multiline: bool,
|
|
237
237
|
) -> Result<Vec<SearchLine>, RgApiError> {
|
|
238
238
|
let mut builder = SearcherBuilder::new();
|
|
239
|
-
builder.line_number(true).before_context(before_context).after_context(after_context).multi_line(multiline);
|
|
239
|
+
builder.line_number(true).line_terminator(LineTerminator::crlf()).before_context(before_context).after_context(after_context).multi_line(multiline);
|
|
240
240
|
let mut searcher = builder.build();
|
|
241
241
|
let mut out = Vec::new();
|
|
242
242
|
let search_matcher = matcher.clone();
|
|
@@ -188,8 +188,7 @@ def test_depth_size_and_filesystem_options(tmp_path):
|
|
|
188
188
|
def test_lnhash_matches_fastcore(tmp_path):
|
|
189
189
|
from fastcore.tools import lnhash as py_hash
|
|
190
190
|
make_tree(tmp_path)
|
|
191
|
-
for row in rg(".", str(tmp_path)):
|
|
192
|
-
assert row.lnhash == py_hash(row.line_number, row.line)
|
|
191
|
+
for row in rg(".", str(tmp_path)): assert row.lnhash == py_hash(row.line_number, row.line)
|
|
193
192
|
|
|
194
193
|
|
|
195
194
|
def test_rg_returns_structured_matches_context_and_relative_paths(tmp_path):
|
|
@@ -662,3 +661,13 @@ def test_nbrg_multiline(tmp_path):
|
|
|
662
661
|
m, = res[0].matches
|
|
663
662
|
assert m.line_number == 1 and "export" in m.line and "import" in m.line
|
|
664
663
|
assert nbrg(r"^import", str(tmp_path), multiline=True, count=True) == 2 # ^ still means line start
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def test_line_anchor_spans(tmp_path):
|
|
667
|
+
"`$` anchors at line end in match spans, for LF and CRLF lines. `count=True` sums those spans."
|
|
668
|
+
(tmp_path/"a.txt").write_text("foo\nbar foo\nfoo bar\nlast foo")
|
|
669
|
+
assert sorted((r.line_number, r.matches) for r in rg(r"foo$", tmp_path)) == [(1, [(0, 3)]), (2, [(4, 7)]), (4, [(5, 8)])]
|
|
670
|
+
assert rg(r"foo$", tmp_path, count=True) == 3
|
|
671
|
+
assert [r.matches for r in rgstr(r"foo$", "foo\nbar foo")] == [[(0, 3)], [(4, 7)]]
|
|
672
|
+
(tmp_path/"crlf.txt").write_bytes(b"foo\r\nbar foo\r\n")
|
|
673
|
+
assert sorted((r.line_number, r.line, r.matches) for r in rg(r"foo$", tmp_path/"crlf.txt")) == [(1, "foo", [(0, 3)]), (2, "bar foo", [(4, 7)])]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|