rgapi 0.1.28__tar.gz → 0.1.30__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -330,7 +330,7 @@ checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4"
330
330
 
331
331
  [[package]]
332
332
  name = "rgapi"
333
- version = "0.1.28"
333
+ version = "0.1.30"
334
334
  dependencies = [
335
335
  "crc32fast",
336
336
  "globset",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "rgapi"
3
- version = "0.1.28"
3
+ version = "0.1.30"
4
4
  edition = "2024"
5
5
  rust-version = "1.91"
6
6
  license = "Apache-2.0"
@@ -52,4 +52,6 @@ Rust callers can consume a `StreamIter` with `cancel_and_join()` to cancel, drai
52
52
 
53
53
  Notebook hierarchy uses `heading_level(source)`, `section_range(levels, idx)` and `ancestor_indices(levels, idx)`. Callers supply zero for non-heading cells. Heading detection skips blank lines and lines starting with `#|`, then checks the first remaining line against `^#{1,6} \w`. It does not search past ordinary text. Section ranges include the addressed cell and end before the next equal-or-higher heading; non-headings select themselves. Ancestors exclude the addressed cell and are returned outermost first. These are calculations over the supplied levels, without retained outline state. Rustygate uses them for its cell selectors.
54
54
 
55
+ `cell_refs(cell)` takes an nbformat cell as a `serde_json::Value` and returns its sigil references as `CellRefs { vars, cmds, tools }`, in order of appearance, with duplicates. A prompt cell has `solveit_ai: true` in its metadata. `vars` holds each `expr` written as `` $`expr` `` in a prompt cell's source. `cmds` holds each `cmd` written as `` !`cmd` `` in a prompt cell's source. `tools` holds each name written as `` &`name` `` or `` &`[a, b]` ``. A name holds word characters and dots. `cell_refs` reads `tools` from the source of a prompt or Markdown cell. For every other cell it reads the `text/markdown` data of `display_data` and `execute_result` outputs. It never reads a prompt cell's outputs. This function is Rust-only. Rustygate uses it for the cells API's `refs=true`.
56
+
55
57
  This package intentionally has no CLI. Python is the interface.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rgapi
3
- Version: 0.1.28
3
+ Version: 0.1.30
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Programming Language :: Python :: Implementation :: CPython
6
6
  Requires-Dist: fastcore>=2.2.29
@@ -12,7 +12,7 @@ License-File: LICENSE
12
12
  Summary: Python API for ripgrep-style file walking and searching
13
13
  Home-Page: https://github.com/AnswerDotAI/rgapi
14
14
  Author-email: Jeremy Howard <j@fast.ai>
15
- License: Apache-2.0
15
+ License-Expression: Apache-2.0
16
16
  Requires-Python: >=3.10
17
17
  Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
18
18
  Project-URL: Homepage, https://github.com/AnswerDotAI/rgapi
@@ -124,6 +124,8 @@ Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of li
124
124
 
125
125
  For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
126
126
 
127
+ `^` and `$` match at the start and end of each line. A line ends at `\n` or `\r\n`. `$` also matches before a `\r` that has no `\n` after it. `rg`, `rgstr`, and `nbrg` without `multiline=True` raise `ValueError` for a pattern that contains a literal `\n` or `\r`.
128
+
127
129
  ### Path results
128
130
 
129
131
  `fd`, `walk`, and `ls` return `PathResults`, a list of absolute `Path` objects. So do `rg` and `nbrg` with `paths=True`, and their async equivalents. Indexing or iterating returns ordinary Paths:
@@ -103,6 +103,8 @@ Pass `lnhashs=True` to `rg` or `rg_iter` to display hash addresses instead of li
103
103
 
104
104
  For other result forms, use `rg(..., paths=True)` to return unique matched paths or `rg(..., count=True)` to count match spans. `paths` and `count` cannot both be set.
105
105
 
106
+ `^` and `$` match at the start and end of each line. A line ends at `\n` or `\r\n`. `$` also matches before a `\r` that has no `\n` after it. `rg`, `rgstr`, and `nbrg` without `multiline=True` raise `ValueError` for a pattern that contains a literal `\n` or `\r`.
107
+
106
108
  ### Path results
107
109
 
108
110
  `fd`, `walk`, and `ls` return `PathResults`, a list of absolute `Path` objects. So do `rg` and `nbrg` with `paths=True`, and their async equivalents. Indexing or iterating returns ordinary Paths:
@@ -1,12 +1,12 @@
1
1
  [build-system]
2
- requires = ["maturin>=1.0,<2.0"]
2
+ requires = ["maturin>=1.9.2,<2.0"]
3
3
  build-backend = "maturin"
4
4
 
5
5
  [project]
6
6
  name = "rgapi"
7
7
  dynamic = ["version"]
8
8
  description = "Python API for ripgrep-style file walking and searching"
9
- license = {text = "Apache-2.0"}
9
+ license = "Apache-2.0"
10
10
  requires-python = ">=3.10"
11
11
  readme = "README.md"
12
12
  authors = [{name = "Jeremy Howard", email = "j@fast.ai"}]
@@ -6,8 +6,8 @@ For orientation, start with `rg(summary=True)`; use line-level results where nee
6
6
 
7
7
  ## Search units
8
8
 
9
- - `rg`: lines, or `SearchBlock` rows with `summary=True`. Blank/whitespace-only lines separate blocks; multiple matches in one block yield one row. Context counts the selected unit. Summary mode cannot combine with `paths` or `count`, but supports hashed block boundaries.
10
- - `nbrg`: `NbResults` of `NbCell` rows from source only, never metadata/outputs. `multiline=True` matches across cell lines while `^`/`$` remain line anchors; ordinary line-oriented `rg` rejects newline patterns.
9
+ - `rg`: lines, or `SearchBlock` rows with `summary=True`. Blank/whitespace-only lines separate blocks; multiple matches in one block yield one row. Context counts the selected unit. Summary mode cannot combine with `paths` or `count`, but supports hashed block boundaries. `^`/`$` anchor each line. A line ends at LF or CRLF.
10
+ - `nbrg`: `NbResults` of `NbCell` rows from source only, never metadata/outputs. `multiline=True` matches across cell lines while `^`/`$` remain line anchors; ordinary line-oriented `rg` rejects patterns containing a newline or carriage return.
11
11
  - Traversal/search run in parallel in Rust; sort when stable order is required. `path_re`/`skip_path_re` filter paths without pruning; `skip_dir`/`skip_dir_re` prune subtrees.
12
12
 
13
13
  ## Result fields and display
@@ -9,7 +9,7 @@ mod walk;
9
9
  mod python;
10
10
 
11
11
  pub use block::{BlockIter, SearchBlock, block_iter};
12
- pub use nb::{NbCell, NbIter, NbOptions, ancestor_indices, heading_level, nb_iter, nb_search, nb_search_file, section_range};
12
+ pub use nb::{CellRefs, NbCell, NbIter, NbOptions, ancestor_indices, cell_refs, heading_level, nb_iter, nb_search, nb_search_file, section_range};
13
13
  pub use search::{MatchSpan, RgIter, RgOptions, SearchKind, SearchLine, compile_regex, rg, rg_iter, search_path, search_text};
14
14
  pub use walk::{FindIter, FindOptions, StreamIter, find, find_iter};
15
15
 
@@ -3,7 +3,7 @@ use std::path::{Path, PathBuf};
3
3
  use std::sync::{Arc, LazyLock};
4
4
  use std::sync::atomic::Ordering;
5
5
 
6
- use grep_matcher::Matcher;
6
+ use grep_matcher::{Captures, Matcher};
7
7
  use grep_regex::RegexMatcher;
8
8
  use ignore::{DirEntry, WalkState};
9
9
  use serde::Deserialize;
@@ -42,6 +42,56 @@ pub fn ancestor_indices(levels: &[usize], idx: usize) -> Vec<usize> {
42
42
  parents
43
43
  }
44
44
 
45
+ /// Group 1 of every `` sigil`body` `` match in `text`, in order of appearance. `sigil` and `body` are regex fragments.
46
+ fn sigil_caps(text: &str, sigil: &str, body: &str) -> Vec<String> {
47
+ let re = RegexMatcher::new(&format!("{sigil}`({body})`")).expect("sigil pattern compiles");
48
+ let mut caps = re.new_captures().expect("captures allocate");
49
+ let mut res = Vec::new();
50
+ re.captures_iter(text.as_bytes(), &mut caps, |c| {
51
+ if let Some(m) = c.get(1) { res.push(text[m.start()..m.end()].to_string()); }
52
+ true
53
+ }).expect("regex search is infallible");
54
+ res
55
+ }
56
+
57
+ /// Names written as `` &`name` `` or `` &`[a, b]` `` in `text`. A name holds word characters and dots.
58
+ fn tool_names(text: &str) -> Vec<String> {
59
+ let groups = sigil_caps(text, "&", r"[\w.]+|\[[\w.,\s]+\]");
60
+ groups.iter().flat_map(|g| g.split(['[', ']', ','])).map(str::trim).filter(|s| !s.is_empty()).map(String::from).collect()
61
+ }
62
+
63
+ /// nbformat multiline text, a string or a list of strings, as one string.
64
+ fn nb_text(v: &serde_json::Value) -> String {
65
+ match v { serde_json::Value::String(s) => s.clone(), serde_json::Value::Array(a) => a.iter().filter_map(|o| o.as_str()).collect(), _ => String::new() }
66
+ }
67
+
68
+ /// The sigil references in one notebook cell, in order of appearance.
69
+ #[derive(Debug, Default, Clone, PartialEq)]
70
+ pub struct CellRefs {
71
+ /// Each `expr` written as `` $`expr` ``
72
+ pub vars: Vec<String>,
73
+ /// Each `cmd` written as `` !`cmd` ``
74
+ pub cmds: Vec<String>,
75
+ /// Each name written as `` &`name` `` or `` &`[a, b]` ``
76
+ pub tools: Vec<String>,
77
+ }
78
+
79
+ /// The sigil references in `cell`, an nbformat cell. A prompt cell has `solveit_ai: true` in its metadata.
80
+ /// `vars` and `cmds` come from a prompt cell's source. `tools` comes from the source of a prompt or Markdown cell.
81
+ /// For every other cell, `tools` comes from the `text/markdown` data of its `display_data` and `execute_result` outputs.
82
+ /// A prompt cell's outputs are never read.
83
+ pub fn cell_refs(cell: &serde_json::Value) -> CellRefs {
84
+ let src = nb_text(&cell["source"]);
85
+ let prompt = cell["metadata"]["solveit_ai"] == true;
86
+ let mut res = CellRefs::default();
87
+ if prompt { (res.vars, res.cmds) = (sigil_caps(&src, r"\$", "[^`]+"), sigil_caps(&src, "!", "[^`]+")); }
88
+ if prompt || cell["cell_type"] == "markdown" { res.tools = tool_names(&src); return res; }
89
+ for o in cell["outputs"].as_array().into_iter().flatten() {
90
+ if matches!(o["output_type"].as_str(), Some("display_data" | "execute_result")) { res.tools.extend(tool_names(&nb_text(&o["data"]["text/markdown"]))); }
91
+ }
92
+ res
93
+ }
94
+
45
95
  #[derive(Debug, Clone)]
46
96
  pub struct NbOptions {
47
97
  pub root: PathBuf,
@@ -5,7 +5,7 @@ use std::sync::{
5
5
  mpsc::SyncSender,
6
6
  };
7
7
 
8
- use grep_matcher::Matcher;
8
+ use grep_matcher::{LineTerminator, Matcher};
9
9
  use grep_regex::{RegexMatcher, RegexMatcherBuilder};
10
10
  use grep_searcher::{BinaryDetection, SearcherBuilder, Sink, SinkContext, SinkContextKind, SinkError, SinkMatch};
11
11
  use ignore::{DirEntry, WalkState};
@@ -167,8 +167,8 @@ fn send_search_error(tx: &SyncSender<Result<SearchLine, RgApiError>>, err: RgApi
167
167
  pub fn compile_regex(pattern: &str, case_sensitive: Option<bool>, smart_case: bool, multiline: bool) -> Result<RegexMatcher, RgApiError> {
168
168
  if pattern.is_empty() { return Err(RgApiError::new("pattern may not be empty")); }
169
169
  let mut builder = RegexMatcherBuilder::new();
170
- if multiline { builder.multi_line(true); }
171
- else { builder.line_terminator(Some(b'\n')); }
170
+ builder.multi_line(true).crlf(true);
171
+ if multiline { builder.line_terminator(None); }
172
172
  match case_sensitive {
173
173
  Some(true) => {
174
174
  builder.case_insensitive(false);
@@ -208,7 +208,7 @@ fn search_path_cancelable(
208
208
  cancel: Option<Arc<AtomicBool>>,
209
209
  ) -> Result<Vec<SearchLine>, RgApiError> {
210
210
  let mut builder = SearcherBuilder::new();
211
- builder.line_number(true).before_context(before_context).after_context(after_context).binary_detection(BinaryDetection::quit(0));
211
+ builder.line_number(true).line_terminator(LineTerminator::crlf()).before_context(before_context).after_context(after_context).binary_detection(BinaryDetection::quit(0));
212
212
  let mut searcher = builder.build();
213
213
  let mut out = Vec::new();
214
214
  let search_matcher = matcher.clone();
@@ -236,7 +236,7 @@ fn search_bytes(
236
236
  multiline: bool,
237
237
  ) -> Result<Vec<SearchLine>, RgApiError> {
238
238
  let mut builder = SearcherBuilder::new();
239
- builder.line_number(true).before_context(before_context).after_context(after_context).multi_line(multiline);
239
+ builder.line_number(true).line_terminator(LineTerminator::crlf()).before_context(before_context).after_context(after_context).multi_line(multiline);
240
240
  let mut searcher = builder.build();
241
241
  let mut out = Vec::new();
242
242
  let search_matcher = matcher.clone();
@@ -188,8 +188,7 @@ def test_depth_size_and_filesystem_options(tmp_path):
188
188
  def test_lnhash_matches_fastcore(tmp_path):
189
189
  from fastcore.tools import lnhash as py_hash
190
190
  make_tree(tmp_path)
191
- for row in rg(".", str(tmp_path)):
192
- assert row.lnhash == py_hash(row.line_number, row.line)
191
+ for row in rg(".", str(tmp_path)): assert row.lnhash == py_hash(row.line_number, row.line)
193
192
 
194
193
 
195
194
  def test_rg_returns_structured_matches_context_and_relative_paths(tmp_path):
@@ -662,3 +661,13 @@ def test_nbrg_multiline(tmp_path):
662
661
  m, = res[0].matches
663
662
  assert m.line_number == 1 and "export" in m.line and "import" in m.line
664
663
  assert nbrg(r"^import", str(tmp_path), multiline=True, count=True) == 2 # ^ still means line start
664
+
665
+
666
+ def test_line_anchor_spans(tmp_path):
667
+ "`$` anchors at line end in match spans, for LF and CRLF lines. `count=True` sums those spans."
668
+ (tmp_path/"a.txt").write_text("foo\nbar foo\nfoo bar\nlast foo")
669
+ assert sorted((r.line_number, r.matches) for r in rg(r"foo$", tmp_path)) == [(1, [(0, 3)]), (2, [(4, 7)]), (4, [(5, 8)])]
670
+ assert rg(r"foo$", tmp_path, count=True) == 3
671
+ assert [r.matches for r in rgstr(r"foo$", "foo\nbar foo")] == [[(0, 3)], [(4, 7)]]
672
+ (tmp_path/"crlf.txt").write_bytes(b"foo\r\nbar foo\r\n")
673
+ assert sorted((r.line_number, r.line, r.matches) for r in rg(r"foo$", tmp_path/"crlf.txt")) == [(1, "foo", [(0, 3)]), (2, "bar foo", [(4, 7)])]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes