SourceIndex 0.1.3__tar.gz → 0.1.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {sourceindex-0.1.3 → sourceindex-0.1.4}/GETTING_STARTED.md +14 -12
  2. {sourceindex-0.1.3 → sourceindex-0.1.4}/PKG-INFO +19 -14
  3. {sourceindex-0.1.3 → sourceindex-0.1.4}/pyproject.toml +8 -1
  4. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/__init__.py +1 -1
  5. sourceindex-0.1.4/sourceindex/build/linerange/anchor.py +220 -0
  6. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/linerange/python_ast.py +21 -33
  7. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/linerange/treesitter.py +24 -73
  8. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/__init__.py +5 -4
  9. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/commands.py +14 -1
  10. sourceindex-0.1.4/sourceindex/cli/consent.py +118 -0
  11. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/install.py +270 -50
  12. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/upgrade.py +51 -18
  13. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/client.py +90 -23
  14. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/lifecycle.py +7 -7
  15. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/protocol.py +5 -0
  16. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/server.py +11 -0
  17. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/env.py +9 -0
  18. {sourceindex-0.1.3 → sourceindex-0.1.4}/.gitignore +0 -0
  19. {sourceindex-0.1.3 → sourceindex-0.1.4}/LICENSE +0 -0
  20. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/__init__.py +0 -0
  21. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/indexer.py +0 -0
  22. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/linerange/__init__.py +0 -0
  23. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/prompts.py +0 -0
  24. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/state.py +0 -0
  25. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/walker.py +0 -0
  26. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/__init__.py +0 -0
  27. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/savings.py +0 -0
  28. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/savings_summary.py +0 -0
  29. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/statusline.py +0 -0
  30. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/__main__.py +0 -0
  31. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/api_key.py +0 -0
  32. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/__init__.py +0 -0
  33. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/crypto.py +0 -0
  34. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/keyring_store.py +0 -0
  35. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/store.py +0 -0
  36. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/__init__.py +0 -0
  37. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/backend.py +0 -0
  38. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/cost.py +0 -0
  39. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/errors.py +0 -0
  40. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/git.py +0 -0
  41. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/languages.py +0 -0
  42. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/llm.py +0 -0
  43. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/log.py +0 -0
  44. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/registry.py +0 -0
  45. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/timing.py +0 -0
  46. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/__init__.py +0 -0
  47. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/experiments.py +0 -0
  48. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/imports.py +0 -0
  49. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/passes.py +0 -0
  50. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/prompts.py +0 -0
  51. {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/roadmap.py +0 -0
@@ -10,8 +10,8 @@ weird" — would be hugely helpful to hear. No feedback is too small.
10
10
  ## What it is
11
11
 
12
12
  When you ask a coding agent to fix or build something, it has to find the
13
- relevant code first. By default Claude Code and opencode do that live, in
14
- your session. SourceIndex does it ahead of time so the agent jumps straight
13
+ relevant code first. By default Claude Code, opencode, and Codex do that
14
+ live, in your session. SourceIndex does it ahead of time so the agent jumps straight
15
15
  to the right files.
16
16
 
17
17
  **Best on an existing repo you didn't fully write yourself.** Not very useful
@@ -22,8 +22,8 @@ on a brand-new empty project.
22
22
  You'll need Python 3.10+ and a SourceIndex API key (starts with `sk-si-`).
23
23
  Apply for access at https://sourceindex.dev/access.
24
24
 
25
- **Easiest path:** open this file in Claude Code or opencode and tell it "set
26
- sourceindex up in this repo." It can run the steps for you.
25
+ **Easiest path:** open this file in Claude Code, opencode, or Codex and tell
26
+ it "set sourceindex up in this repo." It can run the steps for you.
27
27
 
28
28
  If you'd rather do it by hand:
29
29
 
@@ -33,16 +33,18 @@ cd /path/to/your/repo
33
33
  sourceindex init # auto-detects your agent(s)
34
34
  sourceindex init claude # set up Claude Code only
35
35
  sourceindex init opencode # set up opencode only
36
+ sourceindex init codex # set up Codex only
36
37
  ```
37
38
 
38
39
  It'll prompt for the key, set up some git hooks plus a subagent for the
39
40
  coding agent your repo already uses (a `CLAUDE.md` or `.claude/` means
40
- Claude Code; an `AGENTS.md` or `.opencode/` means opencode; neither means
41
- Claude Code), and scan your repo. A few minutes, once per repo. You can
42
- also name both at once (`sourceindex init claude opencode`), and it's safe
43
- to re-run init later to add the other agent.
44
- If you don't have a `CLAUDE.md` (Claude Code) or `AGENTS.md` (opencode)
45
- yet, run `/init` in your agent first.
41
+ Claude Code; a `.opencode/` means opencode; a `.codex/` means Codex; an
42
+ `AGENTS.md` alone means both opencode and Codex; none of those means Claude
43
+ Code), and scan your repo. A few minutes, once per repo. You can also name
44
+ several at once (`sourceindex init claude opencode codex`), and it's safe
45
+ to re-run init later to add another agent.
46
+ If you don't have a `CLAUDE.md` (Claude Code) or `AGENTS.md` (opencode,
47
+ Codex) yet, run `/init` in your agent first.
46
48
 
47
49
  To replace a rotated or expired key later:
48
50
 
@@ -68,8 +70,8 @@ global default.
68
70
 
69
71
  ## Day-to-day
70
72
 
71
- Nothing. Open the repo in Claude Code or opencode as usual — it'll use
72
- SourceIndex on its own.
73
+ Nothing. Open the repo in Claude Code, opencode, or Codex as usual — it'll
74
+ use SourceIndex on its own.
73
75
 
74
76
  ## Feedback I'd love
75
77
 
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: SourceIndex
3
- Version: 0.1.3
3
+ Version: 0.1.4
4
4
  Summary: Codebase index for agentic coding
5
5
  Author: MANTEON PTE. LTD.
6
6
  License-Expression: LicenseRef-Proprietary
@@ -23,6 +23,9 @@ Requires-Dist: tree-sitter-ruby>=0.23
23
23
  Requires-Dist: tree-sitter-rust>=0.23
24
24
  Requires-Dist: tree-sitter-typescript>=0.23
25
25
  Requires-Dist: tree-sitter>=0.23
26
+ Provides-Extra: test
27
+ Requires-Dist: pytest; extra == 'test'
28
+ Requires-Dist: tomli; (python_version < '3.11') and extra == 'test'
26
29
  Description-Content-Type: text/markdown
27
30
 
28
31
  # Getting Started with SourceIndex
@@ -37,8 +40,8 @@ weird" — would be hugely helpful to hear. No feedback is too small.
37
40
  ## What it is
38
41
 
39
42
  When you ask a coding agent to fix or build something, it has to find the
40
- relevant code first. By default Claude Code and opencode do that live, in
41
- your session. SourceIndex does it ahead of time so the agent jumps straight
43
+ relevant code first. By default Claude Code, opencode, and Codex do that
44
+ live, in your session. SourceIndex does it ahead of time so the agent jumps straight
42
45
  to the right files.
43
46
 
44
47
  **Best on an existing repo you didn't fully write yourself.** Not very useful
@@ -49,8 +52,8 @@ on a brand-new empty project.
49
52
  You'll need Python 3.10+ and a SourceIndex API key (starts with `sk-si-`).
50
53
  Apply for access at https://sourceindex.dev/access.
51
54
 
52
- **Easiest path:** open this file in Claude Code or opencode and tell it "set
53
- sourceindex up in this repo." It can run the steps for you.
55
+ **Easiest path:** open this file in Claude Code, opencode, or Codex and tell
56
+ it "set sourceindex up in this repo." It can run the steps for you.
54
57
 
55
58
  If you'd rather do it by hand:
56
59
 
@@ -60,16 +63,18 @@ cd /path/to/your/repo
60
63
  sourceindex init # auto-detects your agent(s)
61
64
  sourceindex init claude # set up Claude Code only
62
65
  sourceindex init opencode # set up opencode only
66
+ sourceindex init codex # set up Codex only
63
67
  ```
64
68
 
65
69
  It'll prompt for the key, set up some git hooks plus a subagent for the
66
70
  coding agent your repo already uses (a `CLAUDE.md` or `.claude/` means
67
- Claude Code; an `AGENTS.md` or `.opencode/` means opencode; neither means
68
- Claude Code), and scan your repo. A few minutes, once per repo. You can
69
- also name both at once (`sourceindex init claude opencode`), and it's safe
70
- to re-run init later to add the other agent.
71
- If you don't have a `CLAUDE.md` (Claude Code) or `AGENTS.md` (opencode)
72
- yet, run `/init` in your agent first.
71
+ Claude Code; a `.opencode/` means opencode; a `.codex/` means Codex; an
72
+ `AGENTS.md` alone means both opencode and Codex; none of those means Claude
73
+ Code), and scan your repo. A few minutes, once per repo. You can also name
74
+ several at once (`sourceindex init claude opencode codex`), and it's safe
75
+ to re-run init later to add another agent.
76
+ If you don't have a `CLAUDE.md` (Claude Code) or `AGENTS.md` (opencode,
77
+ Codex) yet, run `/init` in your agent first.
73
78
 
74
79
  To replace a rotated or expired key later:
75
80
 
@@ -95,8 +100,8 @@ global default.
95
100
 
96
101
  ## Day-to-day
97
102
 
98
- Nothing. Open the repo in Claude Code or opencode as usual — it'll use
99
- SourceIndex on its own.
103
+ Nothing. Open the repo in Claude Code, opencode, or Codex as usual — it'll
104
+ use SourceIndex on its own.
100
105
 
101
106
  ## Feedback I'd love
102
107
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "SourceIndex"
7
- version = "0.1.3"
7
+ version = "0.1.4"
8
8
  description = "Codebase index for agentic coding"
9
9
  readme = "GETTING_STARTED.md"
10
10
  requires-python = ">=3.10"
@@ -47,6 +47,13 @@ dependencies = [
47
47
  "tree-sitter-typescript>=0.23",
48
48
  ]
49
49
 
50
+ [project.optional-dependencies]
51
+ test = [
52
+ "pytest",
53
+ # tomllib backport for the TOML-validity tests on the 3.10 floor.
54
+ "tomli; python_version < '3.11'",
55
+ ]
56
+
50
57
  [project.scripts]
51
58
  sourceindex = "sourceindex.cli:main"
52
59
 
@@ -1,4 +1,4 @@
1
- __version__ = "0.1.3"
1
+ __version__ = "0.1.4"
2
2
 
3
3
  # Surfaced at the top level so the CLI's argparse setup can use it for
4
4
  # --workers help/default without importing the build pipeline (which
@@ -0,0 +1,220 @@
1
+ """Anchor tier-2 entries to parser truth, name-first.
2
+
3
+ Shared by the python-ast and tree-sitter fixers. This is a PERMANENT
4
+ build-pipeline stage (every ``init`` runs it right after the LLM responds),
5
+ not a repair utility — LLM-emitted line numbers are unreliable for every
6
+ builder model, and this stage is the deterministic guarantee that cached
7
+ ranges are parser truth whenever the symbol exists. On perfect LLM output
8
+ it is a no-op.
9
+
10
+ The original fixers were format-first: a strict regex had to recognize the
11
+ whole entry line (canonical ``name (Lstart-Lend): desc``) before any range
12
+ correction happened. Models that drift off-format — Gemma emits ``(23:18)``
13
+ colon ranges ~6% of the time and Go receiver names like ``(b *Backend).foo``
14
+ — silently bypassed the fixer, which is how thousands of reversed/truncated
15
+ ranges survived into built caches.
16
+
17
+ This module inverts the order into two parts:
18
+
19
+ 1. The language backend supplies parser truth: ``{qualified_name: [(s, e), …]}``
20
+ (lists, because the same name can legitimately define several ranges).
21
+ 2. Each entry line is matched by NAME, and its range is converged to truth:
22
+ - any recognizable range token (dash, colon, en-dash, missing ``L``,
23
+ reversed, single-line) is REPLACED with the truth range;
24
+ - a matched name with no range token at all gets the truth range INSERTED;
25
+ - a name absent from truth whose range is clearly broken (reversed,
26
+ ``L1-L1`` placeholder, or missing) is SNAPPED to its innermost existing
27
+ container's range (e.g. a @dataclass-generated ``__init__`` → the class);
28
+ - otherwise the line is left untouched — never dropped, so a well-formed
29
+ range for something the parser can't see (module constants, generated
30
+ members) keeps the LLM's answer.
31
+
32
+ Multiple definitions sharing a name: qualified keys are preferred over bare
33
+ ones by ``_lookup``; among remaining candidates we pick the one whose start is
34
+ closest to the LLM's reported start, falling back to the first definition in
35
+ file order when the entry carried no usable range.
36
+ """
37
+ from __future__ import annotations
38
+
39
+ import re
40
+ from typing import Callable, Dict, List, Optional, Tuple
41
+
42
+ Range = Tuple[int, int]
43
+ Truth = Dict[str, List[Range]]
44
+ Resolver = Callable[[str, Optional[int], bool], Optional[Range]]
45
+
46
+ # Any parenthesized line token: (L12-L40) (12-40) (23:18) (L12:L40) (L12) (12)
47
+ # plus en/em-dash and tilde separators seen in off-format LLM output.
48
+ _RANGE_TOKEN = re.compile(
49
+ r"\(\s*L?(?P<s>\d+)\s*(?:(?P<sep>[-:–—~])\s*L?(?P<e>\d+))?\s*\)"
50
+ )
51
+
52
+ _BARE_ID = re.compile(r"^[A-Za-z_$~][\w$]*$")
53
+
54
+ # Leading tokens the LLM sometimes folds into the name ("func Foo", "pub fn bar").
55
+ _NAME_KEYWORDS = {
56
+ "func", "fn", "def", "function", "method", "class", "interface", "type",
57
+ "struct", "enum", "trait", "impl", "pub", "static", "async", "export",
58
+ "const", "var", "let", "val", "public", "private", "protected", "abstract",
59
+ "override", "suspend", "inline", "internal", "final", "new", "get", "set",
60
+ }
61
+
62
+ _TRAILING_SUFFIXES = (
63
+ ".get", ".set", ".<init>", ".init", ".constructor", ".ctor",
64
+ ".closure", ".lambda", ".function_",
65
+ )
66
+
67
+ _RECEIVER_RE = re.compile(r"^\([^)]*\)\s*\.") # Go receiver: (b *Backend).foo
68
+
69
+
70
+ def normalize_name(qn: str) -> str:
71
+ """Normalize separators to ``.``, drop receiver parens / keyword prefixes /
72
+ accessor suffixes, so lookup keys converge across LLM naming styles."""
73
+ qn = qn.strip()
74
+ qn = _RECEIVER_RE.sub("", qn) # "(b *Backend).foo" -> "foo"
75
+ qn = qn.replace("::", ".").replace("#", ".")
76
+ if " " in qn:
77
+ qn = qn.split()[-1]
78
+ while qn.endswith(_TRAILING_SUFFIXES):
79
+ for suf in _TRAILING_SUFFIXES:
80
+ if qn.endswith(suf):
81
+ qn = qn[: -len(suf)]
82
+ break
83
+ while ".." in qn:
84
+ qn = qn.replace("..", ".")
85
+ return qn.strip(".")
86
+
87
+
88
+ def _bare_tail(name: str) -> str:
89
+ tail = normalize_name(name).rsplit(".", 1)[-1]
90
+ return re.sub(r"<.*>$", "", tail) # strip generics: Foo<T> -> Foo
91
+
92
+
93
+ def _plausible_name(name: str) -> bool:
94
+ """Loose gate for heads that sit directly in front of a range token."""
95
+ name = name.strip()
96
+ if not name or len(name) > 160 or "/" in name or "`" in name:
97
+ return False
98
+ if " " in name and not _RECEIVER_RE.match(name):
99
+ toks = name.split()
100
+ if not all(t.lower() in _NAME_KEYWORDS for t in toks[:-1]):
101
+ return False
102
+ return bool(_BARE_ID.match(_bare_tail(name)))
103
+
104
+
105
+ def _strict_name(name: str) -> bool:
106
+ """Tight gate for the no-range insertion path (highest prose risk)."""
107
+ name = name.strip()
108
+ if not name or " " in name and not _RECEIVER_RE.match(name):
109
+ return False
110
+ if name[0] in "#-*>|":
111
+ return False
112
+ return _plausible_name(name)
113
+
114
+
115
+ def _lookup(truth: Truth, qn: str) -> List[Range]:
116
+ """Qualified match first; peel leading segments; finally bare name."""
117
+ norm = normalize_name(qn)
118
+ if not norm:
119
+ return []
120
+ if norm in truth:
121
+ return truth[norm]
122
+ parts = norm.split(".")
123
+ # Peel LLM-invented leading prefixes (mypkg.Foo.bar -> Foo.bar), but only
124
+ # accept still-qualified suffixes here — the bare tail is gated below.
125
+ for i in range(1, len(parts)):
126
+ suffix = ".".join(parts[i:])
127
+ if "." in suffix and suffix in truth:
128
+ return truth[suffix]
129
+ # Bare-name fallback — but NOT when the entry's own container exists in
130
+ # truth: then this member genuinely isn't defined there (generated or
131
+ # hallucinated), and borrowing a same-named member of another container
132
+ # would mis-anchor it (Dataclass.__init__ -> SomeOtherClass.__init__).
133
+ for i in range(len(parts) - 1):
134
+ container = ".".join(parts[i:-1])
135
+ if container and container in truth:
136
+ return []
137
+ return truth.get(_bare_tail(qn), [])
138
+
139
+
140
+ def make_resolver(truth: Truth) -> Resolver:
141
+ """Build a ``resolve(name, reported_start, allow_container)`` over truth.
142
+
143
+ ``allow_container=True`` additionally tries the name's enclosing scopes
144
+ (``A.B.c`` → ``A.B`` → ``A``) — used only when the entry's own range is
145
+ already known to be broken, so a fabricated member at least points the
146
+ agent at the right class body.
147
+ """
148
+
149
+ def resolve(name: str, reported_start: Optional[int] = None,
150
+ allow_container: bool = False) -> Optional[Range]:
151
+ cands = _lookup(truth, name)
152
+ if not cands and allow_container:
153
+ parent = normalize_name(name)
154
+ while "." in parent and not cands:
155
+ parent = parent.rsplit(".", 1)[0]
156
+ cands = truth.get(parent, [])
157
+ if not cands:
158
+ return None
159
+ if len(cands) > 1 and reported_start is not None:
160
+ return min(cands, key=lambda c: abs(c[0] - reported_start))
161
+ return cands[0]
162
+
163
+ return resolve
164
+
165
+
166
+ def _is_broken(s: int, e: Optional[int]) -> bool:
167
+ if e is None: # single-line token: not broken, but improvable
168
+ return False
169
+ return e < s or (s == 1 and e == 1)
170
+
171
+
172
+ def anchor_entries(functions_text: str, resolve: Resolver) -> str:
173
+ """Apply the name-first convergence rules line by line (never drops a line)."""
174
+ out: List[str] = []
175
+ for line in functions_text.splitlines():
176
+ out.append(_anchor_line(line, resolve))
177
+ joined = "\n".join(out)
178
+ if functions_text.endswith("\n") and not joined.endswith("\n"):
179
+ joined += "\n"
180
+ return joined
181
+
182
+
183
+ def _anchor_line(line: str, resolve: Resolver) -> str:
184
+ m = _RANGE_TOKEN.search(line)
185
+ if m:
186
+ head = line[: m.start()].rstrip()
187
+ after = line[m.end():]
188
+ # Real entries read "name (range): desc" — require the colon.
189
+ if head and re.match(r"\s*:", after) and _plausible_name(head.strip()):
190
+ s = int(m.group("s"))
191
+ e = int(m.group("e")) if m.group("e") else None
192
+ best = resolve(head.strip(), s, False)
193
+ if best is None:
194
+ # Name unknown to the parser: snap to the enclosing container
195
+ # when the range is clearly broken OR points outside it.
196
+ cont = resolve(head.strip(), s, True)
197
+ if cont is not None and (
198
+ _is_broken(s, e) or not (cont[0] <= s <= cont[1])
199
+ ):
200
+ best = cont
201
+ if best is None:
202
+ # No truth to converge to — still canonicalize the token so
203
+ # downstream dash-only parsers (search pass, compliance) see it.
204
+ lo, hi = (s, e if e is not None else s)
205
+ if hi < lo:
206
+ lo, hi = hi, lo
207
+ best = (lo, hi)
208
+ return f"{head} (L{best[0]}-L{best[1]}){after}"
209
+ # fall through: token may live in the description of a range-less entry
210
+
211
+ head, sep, rest = line.partition(":")
212
+ if not sep:
213
+ return line
214
+ name = head.strip()
215
+ if not _strict_name(name):
216
+ return line
217
+ best = resolve(name, None, False) or resolve(name, None, True)
218
+ if best is None:
219
+ return line
220
+ return f"{head.rstrip()} (L{best[0]}-L{best[1]}):{rest}"
@@ -8,28 +8,38 @@ whenever the qualified name matches. Description quality is unchanged.
8
8
  import ast
9
9
  import re
10
10
 
11
+ from .anchor import Truth, make_resolver, anchor_entries
12
+
11
13
 
12
14
  _LINE_RANGE_RE = re.compile(r"\(L?\d+\s*-\s*L?\d+\)")
13
15
  _ENTRY_RE = re.compile(r"^([\w.:_<>]+)\s+\(L?\d+\s*-\s*L?\d+\)\s*:?")
14
16
 
15
17
 
16
- def _python_ast_truth(source: str) -> dict[str, tuple[int, int]]:
17
- """{qualified_name: (start_lineno, end_lineno)} for every def in source.
18
+ def _python_ast_truth(source: str) -> Truth:
19
+ """{qualified_name: [(start_lineno, end_lineno), …]} for every def in source.
18
20
 
19
21
  Methods qualified as Class.method, nested funcs as Outer.inner AND
20
- Outer.<locals>.inner so either naming convention matches."""
22
+ Outer.<locals>.inner so either naming convention matches. A bare-name
23
+ fallback key (last segment) is also populated so entries that drop the
24
+ class prefix still resolve; the resolver prefers qualified keys."""
21
25
  try:
22
26
  tree = ast.parse(source)
23
27
  except (SyntaxError, ValueError):
24
28
  return {}
25
29
 
26
- out: dict[str, tuple[int, int]] = {}
30
+ out: Truth = {}
31
+
32
+ def add(key: str, rng: tuple[int, int]) -> None:
33
+ out.setdefault(key, []).append(rng)
27
34
 
28
35
  def walk(node, prefix: str = ""):
29
36
  for child in ast.iter_child_nodes(node):
30
37
  if isinstance(child, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)):
31
38
  qname = f"{prefix}{child.name}" if prefix else child.name
32
- out[qname] = (child.lineno, child.end_lineno or child.lineno)
39
+ rng = (child.lineno, child.end_lineno or child.lineno)
40
+ add(qname, rng)
41
+ if "." in qname:
42
+ add(child.name, rng)
33
43
  walk(child, prefix=f"{qname}.")
34
44
  if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
35
45
  walk(child, prefix=f"{qname}.<locals>.")
@@ -39,38 +49,16 @@ def _python_ast_truth(source: str) -> dict[str, tuple[int, int]]:
39
49
 
40
50
 
41
51
  def fix_python_line_ranges(source: str, functions_text: str) -> str:
42
- """Replace LLM-emitted line ranges with AST ground truth for Python files.
52
+ """Converge LLM-emitted line ranges to AST ground truth for Python files.
43
53
 
44
- Lines whose qualified name matches an AST def get their (Lstart-Lend)
45
- rewritten to the real range. Unmatched names (likely hallucinations or
46
- deeply nested helpers AST didn't enumerate) are left untouched — kept
47
- so a downstream judge can flag them, not silently dropped."""
54
+ Name-first (see ``anchor``): entries are matched by qualified name and
55
+ their range replaced/inserted/snapped regardless of the LLM's range
56
+ formatting. Names AST can't see with a well-formed range are left
57
+ untouched — kept, never dropped."""
48
58
  truth = _python_ast_truth(source)
49
59
  if not truth:
50
60
  return functions_text
51
-
52
- out_lines: list[str] = []
53
- for line in functions_text.splitlines():
54
- m = _ENTRY_RE.match(line.strip())
55
- if not m:
56
- out_lines.append(line)
57
- continue
58
- qname = m.group(1).rstrip(".")
59
- match_key = qname if qname in truth else None
60
- if match_key is None:
61
- for k in truth:
62
- if k.endswith(qname) or qname.endswith(k):
63
- match_key = k
64
- break
65
- if match_key is None:
66
- out_lines.append(line)
67
- continue
68
- ls, le = truth[match_key]
69
- new_line = _LINE_RANGE_RE.sub(f"(L{ls}-L{le})", line, count=1)
70
- if match_key != qname:
71
- new_line = new_line.replace(qname, match_key, 1)
72
- out_lines.append(new_line)
73
- return "\n".join(out_lines)
61
+ return anchor_entries(functions_text, make_resolver(truth))
74
62
 
75
63
 
76
64
  def fill_missing_python_classes(source: str, functions_text: str) -> str:
@@ -39,6 +39,12 @@ import sys
39
39
  from typing import Dict, List, Optional, Tuple
40
40
 
41
41
  from ...lib.log import get_logger
42
+ from .anchor import (
43
+ make_resolver,
44
+ normalize_name,
45
+ anchor_entries,
46
+ _lookup as anchor_lookup,
47
+ )
42
48
 
43
49
  _log = get_logger(__name__)
44
50
 
@@ -73,13 +79,6 @@ _CONTAINER_TYPES = {
73
79
 
74
80
  _TRAILING_SUFFIXES = (".get",".set",".<init>",".init",".constructor",".ctor",".closure",".lambda",".function_")
75
81
 
76
- _ENTRY_RE = re.compile(
77
- r"^(?P<lead>\s*)(?P<name>[^()]+?)\s*\(\s*L?(?P<s>\d+)\s*-\s*L?(?P<e>\d+)\s*\)\s*:(?P<rest>.*)$"
78
- )
79
- _ENTRY_SINGLE_RE = re.compile(
80
- r"^(?P<lead>\s*)(?P<name>[^()]+?)\s*\(\s*L?(?P<line>\d+)\s*\)\s*:(?P<rest>.*)$"
81
- )
82
-
83
82
 
84
83
  _AUTO_INSTALL_ENV = "SOURCEINDEX_AUTO_INSTALL_GRAMMARS"
85
84
 
@@ -177,29 +176,9 @@ def _get_parser(language_key: str) -> Optional[object]:
177
176
  return parser
178
177
 
179
178
 
180
- def _normalize_name(qn: str) -> str:
181
- """Normalize separators to ``.`` and strip accessor/ctor/closure suffixes.
182
-
183
- Also drops whitespace-prefixed language keywords the LLM sometimes folds
184
- into the qualified name (``func Foo``, ``pub fn bar``, ``def quux``).
185
- """
186
- qn = qn.replace("::", ".").replace("#", ".").strip()
187
- # Keep only the last whitespace-separated token: the actual identifier path
188
- if " " in qn:
189
- qn = qn.split()[-1]
190
- while qn.endswith(_TRAILING_SUFFIXES):
191
- for suf in _TRAILING_SUFFIXES:
192
- if qn.endswith(suf):
193
- qn = qn[: -len(suf)]
194
- break
195
- while ".." in qn:
196
- qn = qn.replace("..", ".")
197
- return qn.strip(".")
198
-
199
-
200
- def _bare_name(qn: str) -> str:
201
- qn = _normalize_name(qn)
202
- return qn.rsplit(".", 1)[-1]
179
+ # Name normalization + lookup now live in the shared name-first anchorer.
180
+ _normalize_name = normalize_name # kept as aliases for any external callers
181
+ _bare_name = lambda qn: normalize_name(qn).rsplit(".", 1)[-1] # noqa: E731
203
182
 
204
183
 
205
184
  def _walk_defs(parser, src: bytes, def_types: Tuple[str, ...]):
@@ -253,7 +232,11 @@ def _walk_defs(parser, src: bytes, def_types: Tuple[str, ...]):
253
232
  inner = deep_first_identifier(node)
254
233
  return text(inner) if inner else None
255
234
 
256
- def walk(node, prefix: str = ""):
235
+ # Iterative pre-order DFS: deeply nested trees (minified/JSX-heavy files)
236
+ # blow Python's recursion limit with a recursive generator.
237
+ stack = [(tree.root_node, "")]
238
+ while stack:
239
+ node, prefix = stack.pop()
257
240
  new_prefix = prefix
258
241
  if node.type in def_types:
259
242
  nm = name_of(node)
@@ -262,10 +245,8 @@ def _walk_defs(parser, src: bytes, def_types: Tuple[str, ...]):
262
245
  yield (full, node.start_point[0] + 1, node.end_point[0] + 1)
263
246
  if node.type in _CONTAINER_TYPES:
264
247
  new_prefix = full
265
- for c in node.children:
266
- yield from walk(c, new_prefix)
267
-
268
- yield from walk(tree.root_node)
248
+ for c in reversed(node.children):
249
+ stack.append((c, new_prefix))
269
250
 
270
251
 
271
252
  def _build_truth(parser, src: bytes, def_types: Tuple[str, ...]) -> Dict[str, List[Tuple[int, int]]]:
@@ -324,24 +305,16 @@ def collect_skeleton(language_key: str, source: str) -> List[Tuple[str, int, int
324
305
  return out
325
306
 
326
307
 
327
- def _lookup(truth: Dict[str, List[Tuple[int, int]]], qn: str) -> List[Tuple[int, int]]:
328
- """Match by qualified name; if missing, peel leading segments off; finally bare."""
329
- norm = _normalize_name(qn)
330
- if not norm:
331
- return []
332
- if norm in truth:
333
- return truth[norm]
334
- parts = norm.split(".")
335
- for i in range(1, len(parts)):
336
- suffix = ".".join(parts[i:])
337
- if suffix in truth:
338
- return truth[suffix]
339
- bare = parts[-1]
340
- return truth.get(bare, [])
308
+ _lookup = anchor_lookup # kept as alias for any external callers
341
309
 
342
310
 
343
311
  def fix_line_ranges(language_key: str, source: str, functions_text: str) -> str:
344
- """Replace LLM-emitted line ranges with tree-sitter truth where possible.
312
+ """Converge LLM-emitted line ranges to tree-sitter truth (name-first).
313
+
314
+ Entry lines are matched by NAME via the shared anchorer, so off-format
315
+ ranges (``(23:18)`` colon style, missing ``L`` prefixes, single-line
316
+ tokens) and receiver-syntax names (``(b *Backend).foo``) are corrected
317
+ too — the old strict-regex path silently skipped both.
345
318
 
346
319
  Returns ``functions_text`` unchanged on:
347
320
  - language not in ``LANG_CFG`` (no grammar configured),
@@ -372,26 +345,4 @@ def fix_line_ranges(language_key: str, source: str, functions_text: str) -> str:
372
345
  )
373
346
  return functions_text
374
347
 
375
- out_lines: List[str] = []
376
- for line in functions_text.splitlines():
377
- m = _ENTRY_RE.match(line)
378
- single = False
379
- if not m:
380
- m = _ENTRY_SINGLE_RE.match(line)
381
- single = bool(m)
382
- if not m:
383
- out_lines.append(line)
384
- continue
385
- name = m.group("name").strip()
386
- reported_start = int(m.group("line") if single else m.group("s"))
387
- cands = _lookup(truth, name)
388
- if not cands:
389
- out_lines.append(line)
390
- continue
391
- best = min(cands, key=lambda c: abs(c[0] - reported_start))
392
- if single:
393
- line = f"{m.group('lead')}{name} (L{best[0]}):{m.group('rest')}"
394
- else:
395
- line = f"{m.group('lead')}{name} (L{best[0]}-L{best[1]}):{m.group('rest')}"
396
- out_lines.append(line)
397
- return "\n".join(out_lines)
348
+ return anchor_entries(functions_text, make_resolver(truth))
@@ -131,10 +131,11 @@ def main() -> None:
131
131
  # No choices=: argparse rejects the empty default against choices for
132
132
  # nargs="*" positionals. cmd_init validates against KNOWN_AGENTS.
133
133
  metavar="agent",
134
- help="Coding agent(s) to set up: 'claude', 'opencode' — e.g. `sourceindex init "
135
- "opencode` or `sourceindex init claude opencode`. Omit to auto-detect from "
136
- "the repo (.claude/ or CLAUDE.md -> claude; .opencode/ or AGENTS.md -> "
137
- "opencode; neither -> claude).",
134
+ help="Coding agent(s) to set up: 'claude', 'opencode', 'codex' — e.g. `sourceindex init "
135
+ "opencode` or `sourceindex init claude codex`. Omit to auto-detect from "
136
+ "the repo (.claude/ or CLAUDE.md -> claude; .opencode/ -> opencode; "
137
+ ".codex/ -> codex; AGENTS.md with neither dir -> opencode and codex; "
138
+ "none -> claude).",
138
139
  )
139
140
  p_init.add_argument("--repo-root", default=".", help="Repository root (default: .)")
140
141
  p_init.add_argument(