SourceIndex 0.1.3__tar.gz → 0.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sourceindex-0.1.3 → sourceindex-0.1.4}/GETTING_STARTED.md +14 -12
- {sourceindex-0.1.3 → sourceindex-0.1.4}/PKG-INFO +19 -14
- {sourceindex-0.1.3 → sourceindex-0.1.4}/pyproject.toml +8 -1
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/__init__.py +1 -1
- sourceindex-0.1.4/sourceindex/build/linerange/anchor.py +220 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/linerange/python_ast.py +21 -33
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/linerange/treesitter.py +24 -73
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/__init__.py +5 -4
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/commands.py +14 -1
- sourceindex-0.1.4/sourceindex/cli/consent.py +118 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/install.py +270 -50
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/upgrade.py +51 -18
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/client.py +90 -23
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/lifecycle.py +7 -7
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/protocol.py +5 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/server.py +11 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/env.py +9 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/.gitignore +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/LICENSE +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/__init__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/indexer.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/linerange/__init__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/prompts.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/state.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/build/walker.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/__init__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/savings.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/savings_summary.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/claudecode/statusline.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/__main__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/cli/api_key.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/__init__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/crypto.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/keyring_store.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/daemon/store.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/__init__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/backend.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/cost.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/errors.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/git.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/languages.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/llm.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/log.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/registry.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/lib/timing.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/__init__.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/experiments.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/imports.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/passes.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/prompts.py +0 -0
- {sourceindex-0.1.3 → sourceindex-0.1.4}/sourceindex/search/roadmap.py +0 -0
|
@@ -10,8 +10,8 @@ weird" — would be hugely helpful to hear. No feedback is too small.
|
|
|
10
10
|
## What it is
|
|
11
11
|
|
|
12
12
|
When you ask a coding agent to fix or build something, it has to find the
|
|
13
|
-
relevant code first. By default Claude Code and
|
|
14
|
-
your session. SourceIndex does it ahead of time so the agent jumps straight
|
|
13
|
+
relevant code first. By default Claude Code, opencode, and Codex do that
|
|
14
|
+
live, in your session. SourceIndex does it ahead of time so the agent jumps straight
|
|
15
15
|
to the right files.
|
|
16
16
|
|
|
17
17
|
**Best on an existing repo you didn't fully write yourself.** Not very useful
|
|
@@ -22,8 +22,8 @@ on a brand-new empty project.
|
|
|
22
22
|
You'll need Python 3.10+ and a SourceIndex API key (starts with `sk-si-`).
|
|
23
23
|
Apply for access at https://sourceindex.dev/access.
|
|
24
24
|
|
|
25
|
-
**Easiest path:** open this file in Claude Code or
|
|
26
|
-
sourceindex up in this repo." It can run the steps for you.
|
|
25
|
+
**Easiest path:** open this file in Claude Code, opencode, or Codex and tell
|
|
26
|
+
it "set sourceindex up in this repo." It can run the steps for you.
|
|
27
27
|
|
|
28
28
|
If you'd rather do it by hand:
|
|
29
29
|
|
|
@@ -33,16 +33,18 @@ cd /path/to/your/repo
|
|
|
33
33
|
sourceindex init # auto-detects your agent(s)
|
|
34
34
|
sourceindex init claude # set up Claude Code only
|
|
35
35
|
sourceindex init opencode # set up opencode only
|
|
36
|
+
sourceindex init codex # set up Codex only
|
|
36
37
|
```
|
|
37
38
|
|
|
38
39
|
It'll prompt for the key, set up some git hooks plus a subagent for the
|
|
39
40
|
coding agent your repo already uses (a `CLAUDE.md` or `.claude/` means
|
|
40
|
-
Claude Code;
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
41
|
+
Claude Code; a `.opencode/` means opencode; a `.codex/` means Codex; an
|
|
42
|
+
`AGENTS.md` alone means both opencode and Codex; none of those means Claude
|
|
43
|
+
Code), and scan your repo. A few minutes, once per repo. You can also name
|
|
44
|
+
several at once (`sourceindex init claude opencode codex`), and it's safe
|
|
45
|
+
to re-run init later to add another agent.
|
|
46
|
+
If you don't have a `CLAUDE.md` (Claude Code) or `AGENTS.md` (opencode,
|
|
47
|
+
Codex) yet, run `/init` in your agent first.
|
|
46
48
|
|
|
47
49
|
To replace a rotated or expired key later:
|
|
48
50
|
|
|
@@ -68,8 +70,8 @@ global default.
|
|
|
68
70
|
|
|
69
71
|
## Day-to-day
|
|
70
72
|
|
|
71
|
-
Nothing. Open the repo in Claude Code or
|
|
72
|
-
SourceIndex on its own.
|
|
73
|
+
Nothing. Open the repo in Claude Code, opencode, or Codex as usual — it'll
|
|
74
|
+
use SourceIndex on its own.
|
|
73
75
|
|
|
74
76
|
## Feedback I'd love
|
|
75
77
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: SourceIndex
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.4
|
|
4
4
|
Summary: Codebase index for agentic coding
|
|
5
5
|
Author: MANTEON PTE. LTD.
|
|
6
6
|
License-Expression: LicenseRef-Proprietary
|
|
@@ -23,6 +23,9 @@ Requires-Dist: tree-sitter-ruby>=0.23
|
|
|
23
23
|
Requires-Dist: tree-sitter-rust>=0.23
|
|
24
24
|
Requires-Dist: tree-sitter-typescript>=0.23
|
|
25
25
|
Requires-Dist: tree-sitter>=0.23
|
|
26
|
+
Provides-Extra: test
|
|
27
|
+
Requires-Dist: pytest; extra == 'test'
|
|
28
|
+
Requires-Dist: tomli; (python_version < '3.11') and extra == 'test'
|
|
26
29
|
Description-Content-Type: text/markdown
|
|
27
30
|
|
|
28
31
|
# Getting Started with SourceIndex
|
|
@@ -37,8 +40,8 @@ weird" — would be hugely helpful to hear. No feedback is too small.
|
|
|
37
40
|
## What it is
|
|
38
41
|
|
|
39
42
|
When you ask a coding agent to fix or build something, it has to find the
|
|
40
|
-
relevant code first. By default Claude Code and
|
|
41
|
-
your session. SourceIndex does it ahead of time so the agent jumps straight
|
|
43
|
+
relevant code first. By default Claude Code, opencode, and Codex do that
|
|
44
|
+
live, in your session. SourceIndex does it ahead of time so the agent jumps straight
|
|
42
45
|
to the right files.
|
|
43
46
|
|
|
44
47
|
**Best on an existing repo you didn't fully write yourself.** Not very useful
|
|
@@ -49,8 +52,8 @@ on a brand-new empty project.
|
|
|
49
52
|
You'll need Python 3.10+ and a SourceIndex API key (starts with `sk-si-`).
|
|
50
53
|
Apply for access at https://sourceindex.dev/access.
|
|
51
54
|
|
|
52
|
-
**Easiest path:** open this file in Claude Code or
|
|
53
|
-
sourceindex up in this repo." It can run the steps for you.
|
|
55
|
+
**Easiest path:** open this file in Claude Code, opencode, or Codex and tell
|
|
56
|
+
it "set sourceindex up in this repo." It can run the steps for you.
|
|
54
57
|
|
|
55
58
|
If you'd rather do it by hand:
|
|
56
59
|
|
|
@@ -60,16 +63,18 @@ cd /path/to/your/repo
|
|
|
60
63
|
sourceindex init # auto-detects your agent(s)
|
|
61
64
|
sourceindex init claude # set up Claude Code only
|
|
62
65
|
sourceindex init opencode # set up opencode only
|
|
66
|
+
sourceindex init codex # set up Codex only
|
|
63
67
|
```
|
|
64
68
|
|
|
65
69
|
It'll prompt for the key, set up some git hooks plus a subagent for the
|
|
66
70
|
coding agent your repo already uses (a `CLAUDE.md` or `.claude/` means
|
|
67
|
-
Claude Code;
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
71
|
+
Claude Code; a `.opencode/` means opencode; a `.codex/` means Codex; an
|
|
72
|
+
`AGENTS.md` alone means both opencode and Codex; none of those means Claude
|
|
73
|
+
Code), and scan your repo. A few minutes, once per repo. You can also name
|
|
74
|
+
several at once (`sourceindex init claude opencode codex`), and it's safe
|
|
75
|
+
to re-run init later to add another agent.
|
|
76
|
+
If you don't have a `CLAUDE.md` (Claude Code) or `AGENTS.md` (opencode,
|
|
77
|
+
Codex) yet, run `/init` in your agent first.
|
|
73
78
|
|
|
74
79
|
To replace a rotated or expired key later:
|
|
75
80
|
|
|
@@ -95,8 +100,8 @@ global default.
|
|
|
95
100
|
|
|
96
101
|
## Day-to-day
|
|
97
102
|
|
|
98
|
-
Nothing. Open the repo in Claude Code or
|
|
99
|
-
SourceIndex on its own.
|
|
103
|
+
Nothing. Open the repo in Claude Code, opencode, or Codex as usual — it'll
|
|
104
|
+
use SourceIndex on its own.
|
|
100
105
|
|
|
101
106
|
## Feedback I'd love
|
|
102
107
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "SourceIndex"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.4"
|
|
8
8
|
description = "Codebase index for agentic coding"
|
|
9
9
|
readme = "GETTING_STARTED.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -47,6 +47,13 @@ dependencies = [
|
|
|
47
47
|
"tree-sitter-typescript>=0.23",
|
|
48
48
|
]
|
|
49
49
|
|
|
50
|
+
[project.optional-dependencies]
|
|
51
|
+
test = [
|
|
52
|
+
"pytest",
|
|
53
|
+
# tomllib backport for the TOML-validity tests on the 3.10 floor.
|
|
54
|
+
"tomli; python_version < '3.11'",
|
|
55
|
+
]
|
|
56
|
+
|
|
50
57
|
[project.scripts]
|
|
51
58
|
sourceindex = "sourceindex.cli:main"
|
|
52
59
|
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""Anchor tier-2 entries to parser truth, name-first.
|
|
2
|
+
|
|
3
|
+
Shared by the python-ast and tree-sitter fixers. This is a PERMANENT
|
|
4
|
+
build-pipeline stage (every ``init`` runs it right after the LLM responds),
|
|
5
|
+
not a repair utility — LLM-emitted line numbers are unreliable for every
|
|
6
|
+
builder model, and this stage is the deterministic guarantee that cached
|
|
7
|
+
ranges are parser truth whenever the symbol exists. On perfect LLM output
|
|
8
|
+
it is a no-op.
|
|
9
|
+
|
|
10
|
+
The original fixers were format-first: a strict regex had to recognize the
|
|
11
|
+
whole entry line (canonical ``name (Lstart-Lend): desc``) before any range
|
|
12
|
+
correction happened. Models that drift off-format — Gemma emits ``(23:18)``
|
|
13
|
+
colon ranges ~6% of the time and Go receiver names like ``(b *Backend).foo``
|
|
14
|
+
— silently bypassed the fixer, which is how thousands of reversed/truncated
|
|
15
|
+
ranges survived into built caches.
|
|
16
|
+
|
|
17
|
+
This module inverts the order into two parts:
|
|
18
|
+
|
|
19
|
+
1. The language backend supplies parser truth: ``{qualified_name: [(s, e), …]}``
|
|
20
|
+
(lists, because the same name can legitimately define several ranges).
|
|
21
|
+
2. Each entry line is matched by NAME, and its range is converged to truth:
|
|
22
|
+
- any recognizable range token (dash, colon, en-dash, missing ``L``,
|
|
23
|
+
reversed, single-line) is REPLACED with the truth range;
|
|
24
|
+
- a matched name with no range token at all gets the truth range INSERTED;
|
|
25
|
+
- a name absent from truth whose range is clearly broken (reversed,
|
|
26
|
+
``L1-L1`` placeholder, or missing) is SNAPPED to its innermost existing
|
|
27
|
+
container's range (e.g. a @dataclass-generated ``__init__`` → the class);
|
|
28
|
+
- otherwise the line is left untouched — never dropped, so a well-formed
|
|
29
|
+
range for something the parser can't see (module constants, generated
|
|
30
|
+
members) keeps the LLM's answer.
|
|
31
|
+
|
|
32
|
+
Multiple definitions sharing a name: qualified keys are preferred over bare
|
|
33
|
+
ones by ``_lookup``; among remaining candidates we pick the one whose start is
|
|
34
|
+
closest to the LLM's reported start, falling back to the first definition in
|
|
35
|
+
file order when the entry carried no usable range.
|
|
36
|
+
"""
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import re
|
|
40
|
+
from typing import Callable, Dict, List, Optional, Tuple
|
|
41
|
+
|
|
42
|
+
Range = Tuple[int, int]
|
|
43
|
+
Truth = Dict[str, List[Range]]
|
|
44
|
+
Resolver = Callable[[str, Optional[int], bool], Optional[Range]]
|
|
45
|
+
|
|
46
|
+
# Any parenthesized line token: (L12-L40) (12-40) (23:18) (L12:L40) (L12) (12)
|
|
47
|
+
# plus en/em-dash and tilde separators seen in off-format LLM output.
|
|
48
|
+
_RANGE_TOKEN = re.compile(
|
|
49
|
+
r"\(\s*L?(?P<s>\d+)\s*(?:(?P<sep>[-:–—~])\s*L?(?P<e>\d+))?\s*\)"
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
_BARE_ID = re.compile(r"^[A-Za-z_$~][\w$]*$")
|
|
53
|
+
|
|
54
|
+
# Leading tokens the LLM sometimes folds into the name ("func Foo", "pub fn bar").
|
|
55
|
+
_NAME_KEYWORDS = {
|
|
56
|
+
"func", "fn", "def", "function", "method", "class", "interface", "type",
|
|
57
|
+
"struct", "enum", "trait", "impl", "pub", "static", "async", "export",
|
|
58
|
+
"const", "var", "let", "val", "public", "private", "protected", "abstract",
|
|
59
|
+
"override", "suspend", "inline", "internal", "final", "new", "get", "set",
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
_TRAILING_SUFFIXES = (
|
|
63
|
+
".get", ".set", ".<init>", ".init", ".constructor", ".ctor",
|
|
64
|
+
".closure", ".lambda", ".function_",
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
_RECEIVER_RE = re.compile(r"^\([^)]*\)\s*\.") # Go receiver: (b *Backend).foo
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def normalize_name(qn: str) -> str:
|
|
71
|
+
"""Normalize separators to ``.``, drop receiver parens / keyword prefixes /
|
|
72
|
+
accessor suffixes, so lookup keys converge across LLM naming styles."""
|
|
73
|
+
qn = qn.strip()
|
|
74
|
+
qn = _RECEIVER_RE.sub("", qn) # "(b *Backend).foo" -> "foo"
|
|
75
|
+
qn = qn.replace("::", ".").replace("#", ".")
|
|
76
|
+
if " " in qn:
|
|
77
|
+
qn = qn.split()[-1]
|
|
78
|
+
while qn.endswith(_TRAILING_SUFFIXES):
|
|
79
|
+
for suf in _TRAILING_SUFFIXES:
|
|
80
|
+
if qn.endswith(suf):
|
|
81
|
+
qn = qn[: -len(suf)]
|
|
82
|
+
break
|
|
83
|
+
while ".." in qn:
|
|
84
|
+
qn = qn.replace("..", ".")
|
|
85
|
+
return qn.strip(".")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _bare_tail(name: str) -> str:
|
|
89
|
+
tail = normalize_name(name).rsplit(".", 1)[-1]
|
|
90
|
+
return re.sub(r"<.*>$", "", tail) # strip generics: Foo<T> -> Foo
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _plausible_name(name: str) -> bool:
|
|
94
|
+
"""Loose gate for heads that sit directly in front of a range token."""
|
|
95
|
+
name = name.strip()
|
|
96
|
+
if not name or len(name) > 160 or "/" in name or "`" in name:
|
|
97
|
+
return False
|
|
98
|
+
if " " in name and not _RECEIVER_RE.match(name):
|
|
99
|
+
toks = name.split()
|
|
100
|
+
if not all(t.lower() in _NAME_KEYWORDS for t in toks[:-1]):
|
|
101
|
+
return False
|
|
102
|
+
return bool(_BARE_ID.match(_bare_tail(name)))
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _strict_name(name: str) -> bool:
|
|
106
|
+
"""Tight gate for the no-range insertion path (highest prose risk)."""
|
|
107
|
+
name = name.strip()
|
|
108
|
+
if not name or " " in name and not _RECEIVER_RE.match(name):
|
|
109
|
+
return False
|
|
110
|
+
if name[0] in "#-*>|":
|
|
111
|
+
return False
|
|
112
|
+
return _plausible_name(name)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _lookup(truth: Truth, qn: str) -> List[Range]:
|
|
116
|
+
"""Qualified match first; peel leading segments; finally bare name."""
|
|
117
|
+
norm = normalize_name(qn)
|
|
118
|
+
if not norm:
|
|
119
|
+
return []
|
|
120
|
+
if norm in truth:
|
|
121
|
+
return truth[norm]
|
|
122
|
+
parts = norm.split(".")
|
|
123
|
+
# Peel LLM-invented leading prefixes (mypkg.Foo.bar -> Foo.bar), but only
|
|
124
|
+
# accept still-qualified suffixes here — the bare tail is gated below.
|
|
125
|
+
for i in range(1, len(parts)):
|
|
126
|
+
suffix = ".".join(parts[i:])
|
|
127
|
+
if "." in suffix and suffix in truth:
|
|
128
|
+
return truth[suffix]
|
|
129
|
+
# Bare-name fallback — but NOT when the entry's own container exists in
|
|
130
|
+
# truth: then this member genuinely isn't defined there (generated or
|
|
131
|
+
# hallucinated), and borrowing a same-named member of another container
|
|
132
|
+
# would mis-anchor it (Dataclass.__init__ -> SomeOtherClass.__init__).
|
|
133
|
+
for i in range(len(parts) - 1):
|
|
134
|
+
container = ".".join(parts[i:-1])
|
|
135
|
+
if container and container in truth:
|
|
136
|
+
return []
|
|
137
|
+
return truth.get(_bare_tail(qn), [])
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def make_resolver(truth: Truth) -> Resolver:
|
|
141
|
+
"""Build a ``resolve(name, reported_start, allow_container)`` over truth.
|
|
142
|
+
|
|
143
|
+
``allow_container=True`` additionally tries the name's enclosing scopes
|
|
144
|
+
(``A.B.c`` → ``A.B`` → ``A``) — used only when the entry's own range is
|
|
145
|
+
already known to be broken, so a fabricated member at least points the
|
|
146
|
+
agent at the right class body.
|
|
147
|
+
"""
|
|
148
|
+
|
|
149
|
+
def resolve(name: str, reported_start: Optional[int] = None,
|
|
150
|
+
allow_container: bool = False) -> Optional[Range]:
|
|
151
|
+
cands = _lookup(truth, name)
|
|
152
|
+
if not cands and allow_container:
|
|
153
|
+
parent = normalize_name(name)
|
|
154
|
+
while "." in parent and not cands:
|
|
155
|
+
parent = parent.rsplit(".", 1)[0]
|
|
156
|
+
cands = truth.get(parent, [])
|
|
157
|
+
if not cands:
|
|
158
|
+
return None
|
|
159
|
+
if len(cands) > 1 and reported_start is not None:
|
|
160
|
+
return min(cands, key=lambda c: abs(c[0] - reported_start))
|
|
161
|
+
return cands[0]
|
|
162
|
+
|
|
163
|
+
return resolve
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _is_broken(s: int, e: Optional[int]) -> bool:
|
|
167
|
+
if e is None: # single-line token: not broken, but improvable
|
|
168
|
+
return False
|
|
169
|
+
return e < s or (s == 1 and e == 1)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def anchor_entries(functions_text: str, resolve: Resolver) -> str:
|
|
173
|
+
"""Apply the name-first convergence rules line by line (never drops a line)."""
|
|
174
|
+
out: List[str] = []
|
|
175
|
+
for line in functions_text.splitlines():
|
|
176
|
+
out.append(_anchor_line(line, resolve))
|
|
177
|
+
joined = "\n".join(out)
|
|
178
|
+
if functions_text.endswith("\n") and not joined.endswith("\n"):
|
|
179
|
+
joined += "\n"
|
|
180
|
+
return joined
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _anchor_line(line: str, resolve: Resolver) -> str:
|
|
184
|
+
m = _RANGE_TOKEN.search(line)
|
|
185
|
+
if m:
|
|
186
|
+
head = line[: m.start()].rstrip()
|
|
187
|
+
after = line[m.end():]
|
|
188
|
+
# Real entries read "name (range): desc" — require the colon.
|
|
189
|
+
if head and re.match(r"\s*:", after) and _plausible_name(head.strip()):
|
|
190
|
+
s = int(m.group("s"))
|
|
191
|
+
e = int(m.group("e")) if m.group("e") else None
|
|
192
|
+
best = resolve(head.strip(), s, False)
|
|
193
|
+
if best is None:
|
|
194
|
+
# Name unknown to the parser: snap to the enclosing container
|
|
195
|
+
# when the range is clearly broken OR points outside it.
|
|
196
|
+
cont = resolve(head.strip(), s, True)
|
|
197
|
+
if cont is not None and (
|
|
198
|
+
_is_broken(s, e) or not (cont[0] <= s <= cont[1])
|
|
199
|
+
):
|
|
200
|
+
best = cont
|
|
201
|
+
if best is None:
|
|
202
|
+
# No truth to converge to — still canonicalize the token so
|
|
203
|
+
# downstream dash-only parsers (search pass, compliance) see it.
|
|
204
|
+
lo, hi = (s, e if e is not None else s)
|
|
205
|
+
if hi < lo:
|
|
206
|
+
lo, hi = hi, lo
|
|
207
|
+
best = (lo, hi)
|
|
208
|
+
return f"{head} (L{best[0]}-L{best[1]}){after}"
|
|
209
|
+
# fall through: token may live in the description of a range-less entry
|
|
210
|
+
|
|
211
|
+
head, sep, rest = line.partition(":")
|
|
212
|
+
if not sep:
|
|
213
|
+
return line
|
|
214
|
+
name = head.strip()
|
|
215
|
+
if not _strict_name(name):
|
|
216
|
+
return line
|
|
217
|
+
best = resolve(name, None, False) or resolve(name, None, True)
|
|
218
|
+
if best is None:
|
|
219
|
+
return line
|
|
220
|
+
return f"{head.rstrip()} (L{best[0]}-L{best[1]}):{rest}"
|
|
@@ -8,28 +8,38 @@ whenever the qualified name matches. Description quality is unchanged.
|
|
|
8
8
|
import ast
|
|
9
9
|
import re
|
|
10
10
|
|
|
11
|
+
from .anchor import Truth, make_resolver, anchor_entries
|
|
12
|
+
|
|
11
13
|
|
|
12
14
|
_LINE_RANGE_RE = re.compile(r"\(L?\d+\s*-\s*L?\d+\)")
|
|
13
15
|
_ENTRY_RE = re.compile(r"^([\w.:_<>]+)\s+\(L?\d+\s*-\s*L?\d+\)\s*:?")
|
|
14
16
|
|
|
15
17
|
|
|
16
|
-
def _python_ast_truth(source: str) ->
|
|
17
|
-
"""{qualified_name: (start_lineno, end_lineno)} for every def in source.
|
|
18
|
+
def _python_ast_truth(source: str) -> Truth:
|
|
19
|
+
"""{qualified_name: [(start_lineno, end_lineno), …]} for every def in source.
|
|
18
20
|
|
|
19
21
|
Methods qualified as Class.method, nested funcs as Outer.inner AND
|
|
20
|
-
Outer.<locals>.inner so either naming convention matches.
|
|
22
|
+
Outer.<locals>.inner so either naming convention matches. A bare-name
|
|
23
|
+
fallback key (last segment) is also populated so entries that drop the
|
|
24
|
+
class prefix still resolve; the resolver prefers qualified keys."""
|
|
21
25
|
try:
|
|
22
26
|
tree = ast.parse(source)
|
|
23
27
|
except (SyntaxError, ValueError):
|
|
24
28
|
return {}
|
|
25
29
|
|
|
26
|
-
out:
|
|
30
|
+
out: Truth = {}
|
|
31
|
+
|
|
32
|
+
def add(key: str, rng: tuple[int, int]) -> None:
|
|
33
|
+
out.setdefault(key, []).append(rng)
|
|
27
34
|
|
|
28
35
|
def walk(node, prefix: str = ""):
|
|
29
36
|
for child in ast.iter_child_nodes(node):
|
|
30
37
|
if isinstance(child, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
31
38
|
qname = f"{prefix}{child.name}" if prefix else child.name
|
|
32
|
-
|
|
39
|
+
rng = (child.lineno, child.end_lineno or child.lineno)
|
|
40
|
+
add(qname, rng)
|
|
41
|
+
if "." in qname:
|
|
42
|
+
add(child.name, rng)
|
|
33
43
|
walk(child, prefix=f"{qname}.")
|
|
34
44
|
if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
35
45
|
walk(child, prefix=f"{qname}.<locals>.")
|
|
@@ -39,38 +49,16 @@ def _python_ast_truth(source: str) -> dict[str, tuple[int, int]]:
|
|
|
39
49
|
|
|
40
50
|
|
|
41
51
|
def fix_python_line_ranges(source: str, functions_text: str) -> str:
|
|
42
|
-
"""
|
|
52
|
+
"""Converge LLM-emitted line ranges to AST ground truth for Python files.
|
|
43
53
|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
54
|
+
Name-first (see ``anchor``): entries are matched by qualified name and
|
|
55
|
+
their range replaced/inserted/snapped regardless of the LLM's range
|
|
56
|
+
formatting. Names AST can't see with a well-formed range are left
|
|
57
|
+
untouched — kept, never dropped."""
|
|
48
58
|
truth = _python_ast_truth(source)
|
|
49
59
|
if not truth:
|
|
50
60
|
return functions_text
|
|
51
|
-
|
|
52
|
-
out_lines: list[str] = []
|
|
53
|
-
for line in functions_text.splitlines():
|
|
54
|
-
m = _ENTRY_RE.match(line.strip())
|
|
55
|
-
if not m:
|
|
56
|
-
out_lines.append(line)
|
|
57
|
-
continue
|
|
58
|
-
qname = m.group(1).rstrip(".")
|
|
59
|
-
match_key = qname if qname in truth else None
|
|
60
|
-
if match_key is None:
|
|
61
|
-
for k in truth:
|
|
62
|
-
if k.endswith(qname) or qname.endswith(k):
|
|
63
|
-
match_key = k
|
|
64
|
-
break
|
|
65
|
-
if match_key is None:
|
|
66
|
-
out_lines.append(line)
|
|
67
|
-
continue
|
|
68
|
-
ls, le = truth[match_key]
|
|
69
|
-
new_line = _LINE_RANGE_RE.sub(f"(L{ls}-L{le})", line, count=1)
|
|
70
|
-
if match_key != qname:
|
|
71
|
-
new_line = new_line.replace(qname, match_key, 1)
|
|
72
|
-
out_lines.append(new_line)
|
|
73
|
-
return "\n".join(out_lines)
|
|
61
|
+
return anchor_entries(functions_text, make_resolver(truth))
|
|
74
62
|
|
|
75
63
|
|
|
76
64
|
def fill_missing_python_classes(source: str, functions_text: str) -> str:
|
|
@@ -39,6 +39,12 @@ import sys
|
|
|
39
39
|
from typing import Dict, List, Optional, Tuple
|
|
40
40
|
|
|
41
41
|
from ...lib.log import get_logger
|
|
42
|
+
from .anchor import (
|
|
43
|
+
make_resolver,
|
|
44
|
+
normalize_name,
|
|
45
|
+
anchor_entries,
|
|
46
|
+
_lookup as anchor_lookup,
|
|
47
|
+
)
|
|
42
48
|
|
|
43
49
|
_log = get_logger(__name__)
|
|
44
50
|
|
|
@@ -73,13 +79,6 @@ _CONTAINER_TYPES = {
|
|
|
73
79
|
|
|
74
80
|
_TRAILING_SUFFIXES = (".get",".set",".<init>",".init",".constructor",".ctor",".closure",".lambda",".function_")
|
|
75
81
|
|
|
76
|
-
_ENTRY_RE = re.compile(
|
|
77
|
-
r"^(?P<lead>\s*)(?P<name>[^()]+?)\s*\(\s*L?(?P<s>\d+)\s*-\s*L?(?P<e>\d+)\s*\)\s*:(?P<rest>.*)$"
|
|
78
|
-
)
|
|
79
|
-
_ENTRY_SINGLE_RE = re.compile(
|
|
80
|
-
r"^(?P<lead>\s*)(?P<name>[^()]+?)\s*\(\s*L?(?P<line>\d+)\s*\)\s*:(?P<rest>.*)$"
|
|
81
|
-
)
|
|
82
|
-
|
|
83
82
|
|
|
84
83
|
_AUTO_INSTALL_ENV = "SOURCEINDEX_AUTO_INSTALL_GRAMMARS"
|
|
85
84
|
|
|
@@ -177,29 +176,9 @@ def _get_parser(language_key: str) -> Optional[object]:
|
|
|
177
176
|
return parser
|
|
178
177
|
|
|
179
178
|
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
Also drops whitespace-prefixed language keywords the LLM sometimes folds
|
|
184
|
-
into the qualified name (``func Foo``, ``pub fn bar``, ``def quux``).
|
|
185
|
-
"""
|
|
186
|
-
qn = qn.replace("::", ".").replace("#", ".").strip()
|
|
187
|
-
# Keep only the last whitespace-separated token: the actual identifier path
|
|
188
|
-
if " " in qn:
|
|
189
|
-
qn = qn.split()[-1]
|
|
190
|
-
while qn.endswith(_TRAILING_SUFFIXES):
|
|
191
|
-
for suf in _TRAILING_SUFFIXES:
|
|
192
|
-
if qn.endswith(suf):
|
|
193
|
-
qn = qn[: -len(suf)]
|
|
194
|
-
break
|
|
195
|
-
while ".." in qn:
|
|
196
|
-
qn = qn.replace("..", ".")
|
|
197
|
-
return qn.strip(".")
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
def _bare_name(qn: str) -> str:
|
|
201
|
-
qn = _normalize_name(qn)
|
|
202
|
-
return qn.rsplit(".", 1)[-1]
|
|
179
|
+
# Name normalization + lookup now live in the shared name-first anchorer.
|
|
180
|
+
_normalize_name = normalize_name # kept as aliases for any external callers
|
|
181
|
+
_bare_name = lambda qn: normalize_name(qn).rsplit(".", 1)[-1] # noqa: E731
|
|
203
182
|
|
|
204
183
|
|
|
205
184
|
def _walk_defs(parser, src: bytes, def_types: Tuple[str, ...]):
|
|
@@ -253,7 +232,11 @@ def _walk_defs(parser, src: bytes, def_types: Tuple[str, ...]):
|
|
|
253
232
|
inner = deep_first_identifier(node)
|
|
254
233
|
return text(inner) if inner else None
|
|
255
234
|
|
|
256
|
-
|
|
235
|
+
# Iterative pre-order DFS: deeply nested trees (minified/JSX-heavy files)
|
|
236
|
+
# blow Python's recursion limit with a recursive generator.
|
|
237
|
+
stack = [(tree.root_node, "")]
|
|
238
|
+
while stack:
|
|
239
|
+
node, prefix = stack.pop()
|
|
257
240
|
new_prefix = prefix
|
|
258
241
|
if node.type in def_types:
|
|
259
242
|
nm = name_of(node)
|
|
@@ -262,10 +245,8 @@ def _walk_defs(parser, src: bytes, def_types: Tuple[str, ...]):
|
|
|
262
245
|
yield (full, node.start_point[0] + 1, node.end_point[0] + 1)
|
|
263
246
|
if node.type in _CONTAINER_TYPES:
|
|
264
247
|
new_prefix = full
|
|
265
|
-
for c in node.children:
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
yield from walk(tree.root_node)
|
|
248
|
+
for c in reversed(node.children):
|
|
249
|
+
stack.append((c, new_prefix))
|
|
269
250
|
|
|
270
251
|
|
|
271
252
|
def _build_truth(parser, src: bytes, def_types: Tuple[str, ...]) -> Dict[str, List[Tuple[int, int]]]:
|
|
@@ -324,24 +305,16 @@ def collect_skeleton(language_key: str, source: str) -> List[Tuple[str, int, int
|
|
|
324
305
|
return out
|
|
325
306
|
|
|
326
307
|
|
|
327
|
-
|
|
328
|
-
"""Match by qualified name; if missing, peel leading segments off; finally bare."""
|
|
329
|
-
norm = _normalize_name(qn)
|
|
330
|
-
if not norm:
|
|
331
|
-
return []
|
|
332
|
-
if norm in truth:
|
|
333
|
-
return truth[norm]
|
|
334
|
-
parts = norm.split(".")
|
|
335
|
-
for i in range(1, len(parts)):
|
|
336
|
-
suffix = ".".join(parts[i:])
|
|
337
|
-
if suffix in truth:
|
|
338
|
-
return truth[suffix]
|
|
339
|
-
bare = parts[-1]
|
|
340
|
-
return truth.get(bare, [])
|
|
308
|
+
_lookup = anchor_lookup # kept as alias for any external callers
|
|
341
309
|
|
|
342
310
|
|
|
343
311
|
def fix_line_ranges(language_key: str, source: str, functions_text: str) -> str:
|
|
344
|
-
"""
|
|
312
|
+
"""Converge LLM-emitted line ranges to tree-sitter truth (name-first).
|
|
313
|
+
|
|
314
|
+
Entry lines are matched by NAME via the shared anchorer, so off-format
|
|
315
|
+
ranges (``(23:18)`` colon style, missing ``L`` prefixes, single-line
|
|
316
|
+
tokens) and receiver-syntax names (``(b *Backend).foo``) are corrected
|
|
317
|
+
too — the old strict-regex path silently skipped both.
|
|
345
318
|
|
|
346
319
|
Returns ``functions_text`` unchanged on:
|
|
347
320
|
- language not in ``LANG_CFG`` (no grammar configured),
|
|
@@ -372,26 +345,4 @@ def fix_line_ranges(language_key: str, source: str, functions_text: str) -> str:
|
|
|
372
345
|
)
|
|
373
346
|
return functions_text
|
|
374
347
|
|
|
375
|
-
|
|
376
|
-
for line in functions_text.splitlines():
|
|
377
|
-
m = _ENTRY_RE.match(line)
|
|
378
|
-
single = False
|
|
379
|
-
if not m:
|
|
380
|
-
m = _ENTRY_SINGLE_RE.match(line)
|
|
381
|
-
single = bool(m)
|
|
382
|
-
if not m:
|
|
383
|
-
out_lines.append(line)
|
|
384
|
-
continue
|
|
385
|
-
name = m.group("name").strip()
|
|
386
|
-
reported_start = int(m.group("line") if single else m.group("s"))
|
|
387
|
-
cands = _lookup(truth, name)
|
|
388
|
-
if not cands:
|
|
389
|
-
out_lines.append(line)
|
|
390
|
-
continue
|
|
391
|
-
best = min(cands, key=lambda c: abs(c[0] - reported_start))
|
|
392
|
-
if single:
|
|
393
|
-
line = f"{m.group('lead')}{name} (L{best[0]}):{m.group('rest')}"
|
|
394
|
-
else:
|
|
395
|
-
line = f"{m.group('lead')}{name} (L{best[0]}-L{best[1]}):{m.group('rest')}"
|
|
396
|
-
out_lines.append(line)
|
|
397
|
-
return "\n".join(out_lines)
|
|
348
|
+
return anchor_entries(functions_text, make_resolver(truth))
|
|
@@ -131,10 +131,11 @@ def main() -> None:
|
|
|
131
131
|
# No choices=: argparse rejects the empty default against choices for
|
|
132
132
|
# nargs="*" positionals. cmd_init validates against KNOWN_AGENTS.
|
|
133
133
|
metavar="agent",
|
|
134
|
-
help="Coding agent(s) to set up: 'claude', 'opencode' — e.g. `sourceindex init "
|
|
135
|
-
"opencode` or `sourceindex init claude
|
|
136
|
-
"the repo (.claude/ or CLAUDE.md -> claude; .opencode/
|
|
137
|
-
"
|
|
134
|
+
help="Coding agent(s) to set up: 'claude', 'opencode', 'codex' — e.g. `sourceindex init "
|
|
135
|
+
"opencode` or `sourceindex init claude codex`. Omit to auto-detect from "
|
|
136
|
+
"the repo (.claude/ or CLAUDE.md -> claude; .opencode/ -> opencode; "
|
|
137
|
+
".codex/ -> codex; AGENTS.md with neither dir -> opencode and codex; "
|
|
138
|
+
"none -> claude).",
|
|
138
139
|
)
|
|
139
140
|
p_init.add_argument("--repo-root", default=".", help="Repository root (default: .)")
|
|
140
141
|
p_init.add_argument(
|