symgrep-codesearch 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- symgrep/__init__.py +0 -0
- symgrep/blocks.py +326 -0
- symgrep/classfile.py +221 -0
- symgrep/cli.py +588 -0
- symgrep/data_symbols.py +457 -0
- symgrep/docsymbols.py +192 -0
- symgrep/enclosing.py +727 -0
- symgrep/mcp_server.py +340 -0
- symgrep/mcp_structured.py +176 -0
- symgrep/report.py +440 -0
- symgrep/resolve.py +278 -0
- symgrep/resolve_js.py +328 -0
- symgrep/resolve_python.py +293 -0
- symgrep/ripgrep.py +213 -0
- symgrep/semantic.py +956 -0
- symgrep/structural.py +259 -0
- symgrep/structure.py +365 -0
- symgrep_codesearch-0.5.0.dist-info/METADATA +424 -0
- symgrep_codesearch-0.5.0.dist-info/RECORD +23 -0
- symgrep_codesearch-0.5.0.dist-info/WHEEL +5 -0
- symgrep_codesearch-0.5.0.dist-info/entry_points.txt +3 -0
- symgrep_codesearch-0.5.0.dist-info/licenses/LICENSE +21 -0
- symgrep_codesearch-0.5.0.dist-info/top_level.txt +1 -0
symgrep/__init__.py
ADDED
|
File without changes
|
symgrep/blocks.py
ADDED
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
"""Block-granular units: sub-function AST blocks + a scope signature, for search-by-MEANING.
|
|
2
|
+
|
|
3
|
+
WHY THIS MODULE EXISTS. `functions()` (semantic.py) indexes whole functions, so search-by-meaning finds a
|
|
4
|
+
duplicated FUNCTION but misses a duplicated BLOCK — the same coherent snippet living inside two otherwise-
|
|
5
|
+
different functions (the sub-function Type-4 clone honestreview's canonical_concerns axis asks for). This
|
|
6
|
+
module extracts those blocks and gives each a deterministic SCOPE SIGNATURE (the block as an "environment":
|
|
7
|
+
what it reads, what it writes, what it does), so retrieval separates a domain accumulation from a generic
|
|
8
|
+
loop by more than a prose sentence. See docs/BLOCK_CHUNKING_SPEC.md.
|
|
9
|
+
|
|
10
|
+
WHAT IS AND IS NOT A JUDGEMENT (the same line semantic.py draws):
|
|
11
|
+
· Extracting a block and computing its signature is PARSING — the tree-sitter AST settles it exactly, no
|
|
12
|
+
meaning decision. Alpha-renaming invariance (two clones with different local names → same signature) is a
|
|
13
|
+
property of dropping plain-identifier leaf NAMES, not of any classifier.
|
|
14
|
+
· Turning a block into a one-sentence job description is a judgement → an LLM (cli.py's describer), grounded
|
|
15
|
+
by the signature. Matching a query to those is RETRIEVAL (semantic.py). This module decides nothing about
|
|
16
|
+
duplication; it yields candidate units, and the caller's model judges "same job".
|
|
17
|
+
|
|
18
|
+
CHANGE-DETECTION IS CONTENT, NOT MTIME. A block's identity (its `key`) embeds the sha256 of its own body, and
|
|
19
|
+
`body_hash` is carried explicitly, so the build's incremental logic keys re-describe on CONTENT: a block that
|
|
20
|
+
merely moves keeps its key (no re-describe), a body that changes gets a new key. `stamp` (mtime:size) is
|
|
21
|
+
carried ONLY as the per-file freshness DISPLAY signal search_by_meaning renders — it never gates re-describe
|
|
22
|
+
(mirrors functions() / semantic.py's documented stamp-is-a-signal contract), so a `touch`/restore/sync that
|
|
23
|
+
resets mtime does NOT cause a reindex.
|
|
24
|
+
|
|
25
|
+
DEGRADES, NEVER DEPENDS. Block extraction is an OPTIONAL layer. If the tree-sitter pack is absent the layer is
|
|
26
|
+
UNAVAILABLE and says so ONCE (never a silent per-file swallow), returning no blocks while functions() keeps
|
|
27
|
+
indexing. A per-file cap engaging is COUNTED and REPORTED, never a silent drop. No broad `except` hides a
|
|
28
|
+
real failure: a genuine parse crash propagates as the bug it is.
|
|
29
|
+
|
|
30
|
+
Reuses structural.py's tree-sitter parse (byte spans + node.type, all TS_LANG_BY_EXT languages) and
|
|
31
|
+
enclosing.file_symbols for the enclosing-symbol nesting. No new dependency.
|
|
32
|
+
"""
|
|
33
|
+
import pathlib
|
|
34
|
+
import sys
|
|
35
|
+
from typing import Dict, List, Optional
|
|
36
|
+
|
|
37
|
+
from .semantic import _cfg_int, _hash8 # shared helpers (one definition — name-uniqueness)
|
|
38
|
+
|
|
39
|
+
# A block = a coherent standalone AST sub-tree. Matched by tree-sitter TYPE SUFFIX so ONE rule spans grammars
|
|
40
|
+
# (the C-family/python grammars share these type names; the suffix match tolerates the prefixes that differ).
|
|
41
|
+
BLOCK_TYPE_SUFFIXES = ("for_statement", "for_in_statement", "foreach_statement", "while_statement",
|
|
42
|
+
"if_statement", "switch_statement", "match_statement", "try_statement",
|
|
43
|
+
"with_statement", "case_statement", "catch_clause")
|
|
44
|
+
# Nodes whose leaf identifier IS semantically meaningful across clones (kept in the signature): a call target
|
|
45
|
+
# and an accessed attribute/member/field/property name. A PLAIN variable identifier is dropped (alpha-
|
|
46
|
+
# invariance) but counted, so the signature is shape + API surface, not variable names.
|
|
47
|
+
_CALL_SUFFIXES = ("call", "call_expression", "method_invocation", "function_call")
|
|
48
|
+
_ATTR_SUFFIXES = ("attribute", "member_expression", "field_expression", "selector_expression",
|
|
49
|
+
"member_access_expression", "scoped_identifier")
|
|
50
|
+
_ASSIGN_SUFFIXES = ("assignment", "assignment_expression", "augmented_assignment",
|
|
51
|
+
"assignment_statement", "variable_declarator", "short_var_declaration")
|
|
52
|
+
|
|
53
|
+
MIN_BLOCK_LINES_DEFAULT = 3
|
|
54
|
+
MIN_BLOCK_CHARS_DEFAULT = 60
|
|
55
|
+
BLOCK_MAX_PER_FILE_DEFAULT = 200
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class BlockLayerUnavailable(RuntimeError):
|
|
59
|
+
"""The tree-sitter pack needed for block extraction is not installed. Raised so the OPTIONAL block layer
|
|
60
|
+
says 'unavailable' once, loudly, rather than silently returning no blocks — unavailable ≠ empty."""
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _lang_of(path: pathlib.Path) -> Optional[str]:
|
|
64
|
+
from .enclosing import TS_LANG_BY_EXT
|
|
65
|
+
return TS_LANG_BY_EXT.get(path.suffix.lower())
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _is_func_node(t: str) -> bool:
|
|
69
|
+
return (t.endswith("function_definition") or t.endswith("method_definition")
|
|
70
|
+
or t in ("function_declaration", "method_declaration", "function_item", "arrow_function"))
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _func_nodes(root):
|
|
74
|
+
"""All function/method nodes in the tree (stack walk, like structural.search_source)."""
|
|
75
|
+
out, stack = [], [root]
|
|
76
|
+
while stack:
|
|
77
|
+
n = stack.pop()
|
|
78
|
+
if _is_func_node(n.type):
|
|
79
|
+
out.append(n)
|
|
80
|
+
stack.extend(n.children)
|
|
81
|
+
return out
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _block_nodes(fn_node):
|
|
85
|
+
"""Qualifying block nodes inside a function — OUTERMOST WINS (do not descend into a collected block, so a
|
|
86
|
+
nested if inside a for is not indexed twice and blocks never overlap)."""
|
|
87
|
+
out, stack = [], list(fn_node.children)
|
|
88
|
+
while stack:
|
|
89
|
+
n = stack.pop()
|
|
90
|
+
if any(n.type.endswith(suf) for suf in BLOCK_TYPE_SUFFIXES):
|
|
91
|
+
out.append(n)
|
|
92
|
+
continue # outermost wins — do NOT descend into this block's children
|
|
93
|
+
stack.extend(n.children)
|
|
94
|
+
return out
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _suffix_any(t: str, suffixes) -> bool:
|
|
98
|
+
return any(t.endswith(s) or t == s for s in suffixes)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _rightmost_identifier(node, data: bytes) -> Optional[str]:
|
|
102
|
+
"""The last identifier leaf under `node` (obj.attr.method → 'method'); the meaningful API name."""
|
|
103
|
+
order = []
|
|
104
|
+
stack = [node]
|
|
105
|
+
while stack:
|
|
106
|
+
n = stack.pop()
|
|
107
|
+
if n.type.endswith("identifier") or n.type.endswith("property_identifier") \
|
|
108
|
+
or n.type.endswith("field_identifier"):
|
|
109
|
+
order.append((n.start_byte, n))
|
|
110
|
+
stack.extend(n.children)
|
|
111
|
+
if not order:
|
|
112
|
+
return None
|
|
113
|
+
order.sort()
|
|
114
|
+
last = order[-1][1]
|
|
115
|
+
txt = data[last.start_byte:last.end_byte].decode("utf-8", "ignore")
|
|
116
|
+
return txt or None
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def scope_signature(node, data: bytes) -> str:
|
|
120
|
+
"""A canonical, alpha-renaming-invariant SHAPE+API signature of a block — the block as an environment.
|
|
121
|
+
|
|
122
|
+
Deterministic parsing, no judgement. Captures: the control-type skeleton (order-sensitive), the sorted
|
|
123
|
+
unique call targets + accessed member names (the API surface — these ARE meaningful across clones), the
|
|
124
|
+
operator multiset, literal KINDS (not values), and counts of assigned vs read plain identifiers (a size
|
|
125
|
+
signal for in/out, names dropped for alpha-invariance). Two Type-4 clones that differ only in local
|
|
126
|
+
variable names produce the SAME signature; two blocks that call different methods do not.
|
|
127
|
+
|
|
128
|
+
This is retrieval enrichment + a $0 exact-clone pre-filter, explicitly NOT a resolved dataflow: it does
|
|
129
|
+
not prove two blocks are the same, it makes the embedding and the describe prompt see the block's shape.
|
|
130
|
+
"""
|
|
131
|
+
ctl = [] # control skeleton, in walk order
|
|
132
|
+
calls, attrs, ops = set(), set(), []
|
|
133
|
+
lits = set()
|
|
134
|
+
assigned, read = 0, 0
|
|
135
|
+
assign_target_ids = set()
|
|
136
|
+
|
|
137
|
+
def _collect_targets(n):
|
|
138
|
+
kids = n.named_children if n.named_children else n.children
|
|
139
|
+
if kids:
|
|
140
|
+
st = [kids[0]]
|
|
141
|
+
while st:
|
|
142
|
+
m = st.pop()
|
|
143
|
+
if m.type.endswith("identifier"):
|
|
144
|
+
assign_target_ids.add(m.id)
|
|
145
|
+
st.extend(m.children)
|
|
146
|
+
|
|
147
|
+
stack = [node]
|
|
148
|
+
while stack:
|
|
149
|
+
n = stack.pop()
|
|
150
|
+
t = n.type
|
|
151
|
+
if any(t.endswith(suf) for suf in BLOCK_TYPE_SUFFIXES):
|
|
152
|
+
ctl.append(next(s for s in BLOCK_TYPE_SUFFIXES if t.endswith(s)))
|
|
153
|
+
if _suffix_any(t, _ASSIGN_SUFFIXES):
|
|
154
|
+
_collect_targets(n)
|
|
155
|
+
if _suffix_any(t, _CALL_SUFFIXES):
|
|
156
|
+
fnc = n.child_by_field_name("function") if hasattr(n, "child_by_field_name") else None
|
|
157
|
+
target = fnc or (n.named_children[0] if n.named_children else None)
|
|
158
|
+
if target is not None:
|
|
159
|
+
name = _rightmost_identifier(target, data)
|
|
160
|
+
if name:
|
|
161
|
+
calls.add(name)
|
|
162
|
+
if _suffix_any(t, _ATTR_SUFFIXES):
|
|
163
|
+
name = _rightmost_identifier(n, data)
|
|
164
|
+
if name:
|
|
165
|
+
attrs.add(name)
|
|
166
|
+
if t.endswith("identifier"):
|
|
167
|
+
if n.id in assign_target_ids:
|
|
168
|
+
assigned += 1
|
|
169
|
+
else:
|
|
170
|
+
read += 1
|
|
171
|
+
elif t.endswith("string") or t == "string_literal":
|
|
172
|
+
lits.add("str")
|
|
173
|
+
elif t.endswith("integer") or t.endswith("number") or t == "int_literal" or t.endswith("float"):
|
|
174
|
+
lits.add("num")
|
|
175
|
+
elif t in ("true", "false", "null", "nil"):
|
|
176
|
+
lits.add("bool")
|
|
177
|
+
elif not n.children and not n.is_named:
|
|
178
|
+
tok = data[n.start_byte:n.end_byte].decode("utf-8", "ignore")
|
|
179
|
+
if tok and all(c in "+-*/%<>=!&|^~" for c in tok) and tok != "=":
|
|
180
|
+
ops.append(tok)
|
|
181
|
+
stack.extend(n.children)
|
|
182
|
+
parts = [
|
|
183
|
+
"ctl:" + ">".join(ctl),
|
|
184
|
+
"call:" + ",".join(sorted(calls)),
|
|
185
|
+
"attr:" + ",".join(sorted(attrs)),
|
|
186
|
+
"op:" + ",".join(sorted(ops)),
|
|
187
|
+
"lit:" + ",".join(sorted(lits)),
|
|
188
|
+
f"io:{read}r/{assigned}w",
|
|
189
|
+
]
|
|
190
|
+
return " ".join(parts)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _parsers_for(langs) -> Dict[str, object]:
|
|
194
|
+
"""Acquire a tree-sitter parser per language ONCE. If the pack is absent, raise BlockLayerUnavailable —
|
|
195
|
+
the OPTIONAL layer is unavailable and says so once, never a per-file silent swallow. A concrete parser
|
|
196
|
+
that cannot be built for one known language is left out (that language simply yields no blocks); a missing
|
|
197
|
+
PACK (ImportError) fails the whole layer loudly because nothing can parse."""
|
|
198
|
+
from . import structural
|
|
199
|
+
from .enclosing import TreeSitterUnavailable
|
|
200
|
+
parsers = {}
|
|
201
|
+
for lang in sorted(set(langs)):
|
|
202
|
+
try:
|
|
203
|
+
parsers[lang] = structural._get_parser(lang)
|
|
204
|
+
except (ImportError, TreeSitterUnavailable) as e:
|
|
205
|
+
raise BlockLayerUnavailable(
|
|
206
|
+
"no usable tree-sitter grammar pack, so block extraction is UNAVAILABLE — this is NOT "
|
|
207
|
+
"'no blocks found'. Install tree-sitter-language-pack, or leave SYMGREP_INDEX_BLOCKS unset. "
|
|
208
|
+
f"[{e}]") from e
|
|
209
|
+
return parsers
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def blocks(roots: List[str]) -> List[dict]:
|
|
213
|
+
"""Every qualifying sub-function block under `roots`, mirroring functions()' dict shape plus block fields.
|
|
214
|
+
|
|
215
|
+
Returned dict: {key, qual, path, lineno, body, body_hash, stamp, unit:"block", parent_key, block_start,
|
|
216
|
+
block_end, signature, sig_hash}. Key = "<abspath>::<parent_sym>#<block_kind>@<body_hash8>" — CONTENT-keyed
|
|
217
|
+
(a block that moves keeps its key; a changed body gets a new one). Identical blocks in one function
|
|
218
|
+
collide on key → deduped (they ARE the same content). Bounded by min-size + per-file cap; skips trivia.
|
|
219
|
+
Every drop (per-file cap) and skip (unparsed file) is counted and REPORTED, never silent.
|
|
220
|
+
"""
|
|
221
|
+
from .enclosing import file_symbols
|
|
222
|
+
from .semantic import INDEXABLE_EXTS, _is_generated, _looks_minified, _module_name
|
|
223
|
+
|
|
224
|
+
min_lines = _cfg_int("SYMGREP_BLOCK_MIN_LINES", MIN_BLOCK_LINES_DEFAULT)
|
|
225
|
+
min_chars = _cfg_int("SYMGREP_BLOCK_MIN_CHARS", MIN_BLOCK_CHARS_DEFAULT)
|
|
226
|
+
max_per_file = _cfg_int("SYMGREP_BLOCK_MAX_PER_FILE", BLOCK_MAX_PER_FILE_DEFAULT)
|
|
227
|
+
|
|
228
|
+
# Enumerate candidate files first, so parsers are acquired once for exactly the languages present.
|
|
229
|
+
cand = []
|
|
230
|
+
for root in roots:
|
|
231
|
+
rp = pathlib.Path(root)
|
|
232
|
+
files = [rp] if rp.is_file() else sorted(
|
|
233
|
+
p for p in rp.rglob("*")
|
|
234
|
+
if p.is_file() and p.suffix.lower() in INDEXABLE_EXTS and not _is_generated(p, rp))
|
|
235
|
+
for f in files:
|
|
236
|
+
lang = _lang_of(f)
|
|
237
|
+
if lang:
|
|
238
|
+
cand.append((f, lang))
|
|
239
|
+
if not cand:
|
|
240
|
+
return []
|
|
241
|
+
parsers = _parsers_for(l for _f, l in cand) # raises BlockLayerUnavailable if the pack is absent (loud)
|
|
242
|
+
|
|
243
|
+
out = []
|
|
244
|
+
capped_files = 0 # files where the per-file cap engaged (blocks beyond the cap intentionally omitted)
|
|
245
|
+
capped_blocks = 0 # count of qualifying blocks omitted by the cap — the honest denominator
|
|
246
|
+
parse_skipped = 0 # files whose bytes would not parse (counted, not swallowed)
|
|
247
|
+
for f, lang in cand:
|
|
248
|
+
parser = parsers.get(lang)
|
|
249
|
+
if parser is None:
|
|
250
|
+
continue
|
|
251
|
+
try:
|
|
252
|
+
st = f.stat()
|
|
253
|
+
if st.st_size > 0 and getattr(st, "st_blocks", 1) == 0: # iCloud dataless guard (as functions())
|
|
254
|
+
continue
|
|
255
|
+
text = f.read_text(encoding="utf-8", errors="ignore")
|
|
256
|
+
except OSError:
|
|
257
|
+
continue
|
|
258
|
+
if _looks_minified(text):
|
|
259
|
+
continue
|
|
260
|
+
data = text.encode("utf-8")
|
|
261
|
+
tree = parser.parse(data) # valid bytes parse; a genuine crash propagates (a real bug)
|
|
262
|
+
if tree is None or tree.root_node is None:
|
|
263
|
+
parse_skipped += 1
|
|
264
|
+
continue
|
|
265
|
+
fs = file_symbols(str(f))
|
|
266
|
+
stamp = f"{st.st_mtime_ns}:{st.st_size}" # DISPLAY freshness signal only — identity is body_hash
|
|
267
|
+
mod = _module_name(f)
|
|
268
|
+
n_file = 0
|
|
269
|
+
capped_here = False
|
|
270
|
+
seen_keys = set()
|
|
271
|
+
for fn in _func_nodes(tree.root_node):
|
|
272
|
+
fn_line = fn.start_point[0] + 1
|
|
273
|
+
parent_sym = _enclosing_name(fs, fn_line) or f"<anon@{fn_line}>"
|
|
274
|
+
for blk in _block_nodes(fn):
|
|
275
|
+
body = data[blk.start_byte:blk.end_byte].decode("utf-8", "ignore")
|
|
276
|
+
if body.count("\n") + 1 < min_lines or len(body) < min_chars:
|
|
277
|
+
continue
|
|
278
|
+
if n_file >= max_per_file: # cap engaged — COUNT the omission, do not silently drop
|
|
279
|
+
capped_blocks += 1
|
|
280
|
+
capped_here = True
|
|
281
|
+
continue
|
|
282
|
+
bh = _hash8(body)
|
|
283
|
+
block_kind = next(s for s in BLOCK_TYPE_SUFFIXES if blk.type.endswith(s))
|
|
284
|
+
key = f"{f.resolve()}::{parent_sym}#{block_kind}@{bh}"
|
|
285
|
+
if key in seen_keys: # exact-dup within this function → dedup (same content)
|
|
286
|
+
continue
|
|
287
|
+
seen_keys.add(key)
|
|
288
|
+
sig = scope_signature(blk, data)
|
|
289
|
+
start_line = blk.start_point[0] + 1
|
|
290
|
+
out.append({
|
|
291
|
+
"key": key,
|
|
292
|
+
"qual": f"{mod}.{parent_sym}#{block_kind}@{bh}", # whitespace-free (MCP contract)
|
|
293
|
+
"path": str(f), "lineno": start_line, "body": body, "body_hash": bh, "stamp": stamp,
|
|
294
|
+
"unit": "block", "parent_key": f"{f.resolve()}::{parent_sym}",
|
|
295
|
+
"block_start": start_line, "block_end": blk.end_point[0] + 1,
|
|
296
|
+
"signature": sig, "sig_hash": _hash8(sig),
|
|
297
|
+
})
|
|
298
|
+
n_file += 1
|
|
299
|
+
if capped_here:
|
|
300
|
+
capped_files += 1
|
|
301
|
+
if capped_blocks or parse_skipped:
|
|
302
|
+
sys.stderr.write(
|
|
303
|
+
f"symgrep block-index: {len(out)} block(s); "
|
|
304
|
+
f"OMITTED {capped_blocks} block(s) past the per-file cap ({max_per_file}) in {capped_files} file(s) "
|
|
305
|
+
f"[raise SYMGREP_BLOCK_MAX_PER_FILE]; {parse_skipped} file(s) would not parse (counted, not "
|
|
306
|
+
f"described).\n")
|
|
307
|
+
return out
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _enclosing_name(fs, line: int) -> Optional[str]:
|
|
311
|
+
"""The innermost enclosing symbol name at `line`, via FileSymbols.enclosing when available, else a manual
|
|
312
|
+
innermost-span scan over fs.symbols."""
|
|
313
|
+
if fs is None:
|
|
314
|
+
return None
|
|
315
|
+
enc = getattr(fs, "enclosing", None)
|
|
316
|
+
if callable(enc):
|
|
317
|
+
sym = enc(line)
|
|
318
|
+
if sym is not None:
|
|
319
|
+
return sym.name
|
|
320
|
+
best = None
|
|
321
|
+
for s in (getattr(fs, "symbols", None) or []):
|
|
322
|
+
if s.start is None or s.end is None:
|
|
323
|
+
continue
|
|
324
|
+
if s.start <= line <= s.end and (best is None or s.start > best.start):
|
|
325
|
+
best = s
|
|
326
|
+
return best.name if best else None
|
symgrep/classfile.py
ADDED
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""Symbols from a compiled JVM `.class` file — the constant pool read directly, no JDK.
|
|
2
|
+
|
|
3
|
+
WHY THIS IS DIFFERENT FROM EVERY OTHER SYMBOL PATH. Source (.java, .py, …) is TEXT: ripgrep matches it
|
|
4
|
+
and tree-sitter/ast gives exact spans. A `.class` is BINARY bytecode — ripgrep, symgrep's whole Tier 1,
|
|
5
|
+
cannot search it, and there is no source text to grep. But the JVM classfile format (JVMS §4) is fixed
|
|
6
|
+
and self-describing: the constant pool holds the class name, every method name + type descriptor, and
|
|
7
|
+
every field name. Extracting them is reading a documented binary layout — PARSING, exactly like
|
|
8
|
+
tree-sitter reads a grammar, so it carries no model call and its `basis` is "classfile" (honest: this
|
|
9
|
+
came from bytecode, not source).
|
|
10
|
+
|
|
11
|
+
WHAT IT CAN AND CANNOT KNOW (stated, never blurred):
|
|
12
|
+
· CAN: class name, method names + descriptors (overloads disambiguated), field names — the public
|
|
13
|
+
shape a `file_symbols` consumer wants when the source is a dependency it does not have.
|
|
14
|
+
· Line spans: ONLY if the class was compiled with debug info (the optional Code→LineNumberTable
|
|
15
|
+
attribute). Present → real spans; absent → start=0, end=None, and the basis still says "classfile"
|
|
16
|
+
so a consumer never mistakes a missing span for line 0.
|
|
17
|
+
· CANNOT: local variable names, comments, or a trustworthy one-sentence "what it does" — bytecode is
|
|
18
|
+
a lossy witness of intent, which is why `.class` feeds the STRUCTURAL file_symbols path and is
|
|
19
|
+
deliberately kept OUT of the semantic (meaning) index. Describing a method from its bytecode would
|
|
20
|
+
be a confident-wrong-answer; if you have the source, index the source.
|
|
21
|
+
|
|
22
|
+
This slots into enclosing.file_symbols so the existing `file_symbols(path)` MCP tool answers for a
|
|
23
|
+
`.class` with no new surface — a dependency jar's symbols become queryable without unzipping to source.
|
|
24
|
+
"""
|
|
25
|
+
import struct
|
|
26
|
+
from typing import Dict, List, Optional, Tuple
|
|
27
|
+
|
|
28
|
+
# Constant-pool tag → how many bytes its body occupies after the 1-byte tag, and whether it eats an
|
|
29
|
+
# extra pool slot (Long/Double do — JVMS §4.4.5). This is the format, not a heuristic.
|
|
30
|
+
_CP_LAYOUT: Dict[int, Tuple[int, bool]] = {
|
|
31
|
+
1: (-1, False), # Utf8: u2 length then <length> bytes (handled specially)
|
|
32
|
+
3: (4, False), # Integer
|
|
33
|
+
4: (4, False), # Float
|
|
34
|
+
5: (8, True), # Long — takes TWO pool slots
|
|
35
|
+
6: (8, True), # Double — takes TWO pool slots
|
|
36
|
+
7: (2, False), # Class → name_index
|
|
37
|
+
8: (2, False), # String
|
|
38
|
+
9: (4, False), # Fieldref
|
|
39
|
+
10: (4, False), # Methodref
|
|
40
|
+
11: (4, False), # InterfaceMethodref
|
|
41
|
+
12: (4, False), # NameAndType
|
|
42
|
+
15: (3, False), # MethodHandle
|
|
43
|
+
16: (2, False), # MethodType
|
|
44
|
+
17: (4, False), # Dynamic
|
|
45
|
+
18: (4, False), # InvokeDynamic
|
|
46
|
+
19: (2, False), # Module
|
|
47
|
+
20: (2, False), # Package
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
MAGIC = 0xCAFEBABE
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class NotAClassFile(ValueError):
|
|
54
|
+
pass
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class _Reader:
|
|
58
|
+
def __init__(self, data: bytes):
|
|
59
|
+
self.d = data
|
|
60
|
+
self.i = 0
|
|
61
|
+
|
|
62
|
+
def u1(self) -> int:
|
|
63
|
+
v = self.d[self.i]
|
|
64
|
+
self.i += 1
|
|
65
|
+
return v
|
|
66
|
+
|
|
67
|
+
def u2(self) -> int:
|
|
68
|
+
v = struct.unpack_from(">H", self.d, self.i)[0]
|
|
69
|
+
self.i += 2
|
|
70
|
+
return v
|
|
71
|
+
|
|
72
|
+
def u4(self) -> int:
|
|
73
|
+
v = struct.unpack_from(">I", self.d, self.i)[0]
|
|
74
|
+
self.i += 4
|
|
75
|
+
return v
|
|
76
|
+
|
|
77
|
+
def take(self, n: int) -> bytes:
|
|
78
|
+
v = self.d[self.i:self.i + n]
|
|
79
|
+
self.i += n
|
|
80
|
+
return v
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _mutf8(b: bytes) -> str:
|
|
84
|
+
"""Decode JVM "modified UTF-8" (JVMS 4.4.7): U+0000 encoded as 0xC0 0x80, supplementary chars as
|
|
85
|
+
CESU-8 surrogate pairs. A plain utf-8 decode silently corrupts both. Fast path for the ASCII/BMP
|
|
86
|
+
names that dominate; the manual walk runs only when the modified encodings are actually present."""
|
|
87
|
+
if 0xC0 not in b and 0xED not in b:
|
|
88
|
+
try:
|
|
89
|
+
return b.decode("utf-8")
|
|
90
|
+
except UnicodeDecodeError:
|
|
91
|
+
pass
|
|
92
|
+
out = []
|
|
93
|
+
i, n = 0, len(b)
|
|
94
|
+
while i < n:
|
|
95
|
+
c = b[i]
|
|
96
|
+
if c == 0xC0 and i + 1 < n and b[i + 1] == 0x80:
|
|
97
|
+
out.append("\x00"); i += 2
|
|
98
|
+
elif c < 0x80:
|
|
99
|
+
out.append(chr(c)); i += 1
|
|
100
|
+
elif c == 0xED and i + 5 < n and (b[i + 1] & 0xF0) == 0xA0: # 6-byte CESU-8 surrogate pair
|
|
101
|
+
hi = 0xD000 | ((b[i + 1] & 0x3F) << 6) | (b[i + 2] & 0x3F)
|
|
102
|
+
lo = 0xD000 | ((b[i + 4] & 0x3F) << 6) | (b[i + 5] & 0x3F)
|
|
103
|
+
out.append(chr(0x10000 + (((hi - 0xD800) << 10) | (lo - 0xDC00)))); i += 6
|
|
104
|
+
elif 0xC0 <= c < 0xE0 and i + 1 < n:
|
|
105
|
+
out.append(chr(((c & 0x1F) << 6) | (b[i + 1] & 0x3F))); i += 2
|
|
106
|
+
elif 0xE0 <= c < 0xF0 and i + 2 < n:
|
|
107
|
+
out.append(chr(((c & 0x0F) << 12) | ((b[i + 1] & 0x3F) << 6) | (b[i + 2] & 0x3F))); i += 3
|
|
108
|
+
else:
|
|
109
|
+
out.append("�"); i += 1
|
|
110
|
+
return "".join(out)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _parse_constant_pool(r: _Reader) -> Dict[int, object]:
|
|
114
|
+
count = r.u2()
|
|
115
|
+
pool: Dict[int, object] = {}
|
|
116
|
+
idx = 1
|
|
117
|
+
while idx < count:
|
|
118
|
+
tag = r.u1()
|
|
119
|
+
if tag == 1: # Utf8
|
|
120
|
+
length = r.u2()
|
|
121
|
+
pool[idx] = ("Utf8", _mutf8(r.take(length)))
|
|
122
|
+
elif tag == 7: # Class → name_index
|
|
123
|
+
pool[idx] = ("Class", r.u2())
|
|
124
|
+
elif tag in _CP_LAYOUT:
|
|
125
|
+
size, wide = _CP_LAYOUT[tag]
|
|
126
|
+
body = r.take(size)
|
|
127
|
+
pool[idx] = (tag, body)
|
|
128
|
+
if wide:
|
|
129
|
+
idx += 1 # Long/Double occupy this slot AND the next
|
|
130
|
+
else:
|
|
131
|
+
raise NotAClassFile(f"unknown constant-pool tag {tag} at index {idx}")
|
|
132
|
+
idx += 1
|
|
133
|
+
return pool
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _utf8(pool: Dict[int, object], index: int) -> Optional[str]:
|
|
137
|
+
v = pool.get(index)
|
|
138
|
+
return v[1] if isinstance(v, tuple) and v[0] == "Utf8" else None
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _class_name(pool: Dict[int, object], class_index: int) -> Optional[str]:
|
|
142
|
+
v = pool.get(class_index)
|
|
143
|
+
if isinstance(v, tuple) and v[0] == "Class":
|
|
144
|
+
raw = _utf8(pool, v[1])
|
|
145
|
+
return raw.replace("/", ".") if raw else None # internal "a/b/C" → "a.b.C"
|
|
146
|
+
return None
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _member_symbols(r: _Reader, pool: Dict[int, object], class_name: str, kind: str) -> List[dict]:
|
|
150
|
+
out: List[dict] = []
|
|
151
|
+
count = r.u2()
|
|
152
|
+
for _ in range(count):
|
|
153
|
+
r.u2() # access_flags
|
|
154
|
+
name = _utf8(pool, r.u2())
|
|
155
|
+
desc = _utf8(pool, r.u2())
|
|
156
|
+
span: Optional[Tuple[int, int]] = None
|
|
157
|
+
attr_count = r.u2()
|
|
158
|
+
for _a in range(attr_count):
|
|
159
|
+
attr_name = _utf8(pool, r.u2())
|
|
160
|
+
attr_len = r.u4()
|
|
161
|
+
body_start = r.i
|
|
162
|
+
if kind == "method" and attr_name == "Code":
|
|
163
|
+
span = _code_line_span(r, pool)
|
|
164
|
+
r.i = body_start + attr_len # always skip by declared length (robust)
|
|
165
|
+
out.append({"name": name, "desc": desc, "kind": kind, "span": span})
|
|
166
|
+
return out
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _code_line_span(r: _Reader, pool: Dict[int, object]) -> Optional[Tuple[int, int]]:
|
|
170
|
+
"""Read a Code attribute far enough to find its LineNumberTable (optional debug info)."""
|
|
171
|
+
r.u2(); r.u2() # max_stack, max_locals
|
|
172
|
+
code_len = r.u4()
|
|
173
|
+
r.i += code_len # skip bytecode
|
|
174
|
+
exc_len = r.u2()
|
|
175
|
+
r.i += exc_len * 8 # exception table
|
|
176
|
+
lines: List[int] = []
|
|
177
|
+
sub_count = r.u2()
|
|
178
|
+
for _ in range(sub_count):
|
|
179
|
+
sub_name = _utf8(pool, r.u2())
|
|
180
|
+
sub_len = r.u4()
|
|
181
|
+
start = r.i
|
|
182
|
+
if sub_name == "LineNumberTable":
|
|
183
|
+
n = r.u2()
|
|
184
|
+
for _e in range(n):
|
|
185
|
+
r.u2() # start_pc
|
|
186
|
+
lines.append(r.u2()) # line_number
|
|
187
|
+
r.i = start + sub_len
|
|
188
|
+
return (min(lines), max(lines)) if lines else None
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def classfile_symbols(data: bytes):
|
|
192
|
+
"""-> (class_name, [Symbol]) via enclosing.Symbol. Raises NotAClassFile on a non-class buffer."""
|
|
193
|
+
from .enclosing import FileSymbols, Symbol
|
|
194
|
+
r = _Reader(data)
|
|
195
|
+
if r.u4() != MAGIC:
|
|
196
|
+
raise NotAClassFile("bad magic (not 0xCAFEBABE)")
|
|
197
|
+
r.u2(); r.u2() # minor, major version
|
|
198
|
+
pool = _parse_constant_pool(r)
|
|
199
|
+
r.u2() # access_flags
|
|
200
|
+
class_name = _class_name(pool, r.u2()) or "(unknown)"
|
|
201
|
+
r.u2() # super_class
|
|
202
|
+
n_ifaces = r.u2()
|
|
203
|
+
r.i += n_ifaces * 2 # interfaces[]
|
|
204
|
+
fields = _member_symbols(r, pool, class_name, "field")
|
|
205
|
+
methods = _member_symbols(r, pool, class_name, "method")
|
|
206
|
+
|
|
207
|
+
# Disambiguate overloaded methods by appending the descriptor — keeps names UNIQUE (symgrep's ethos)
|
|
208
|
+
# only where they would collide, so the common case stays clean `Class.method`.
|
|
209
|
+
seen: Dict[str, int] = {}
|
|
210
|
+
for m in methods:
|
|
211
|
+
seen[m["name"]] = seen.get(m["name"], 0) + 1
|
|
212
|
+
syms: List[Symbol] = []
|
|
213
|
+
for m in methods:
|
|
214
|
+
base = f"{class_name}.{m['name']}"
|
|
215
|
+
name = f"{base}{m['desc']}" if seen[m["name"]] > 1 else base
|
|
216
|
+
start, end = (m["span"] if m["span"] else (0, None))
|
|
217
|
+
syms.append(Symbol(name=name, kind="method", start=start, end=end, basis="classfile"))
|
|
218
|
+
for f in fields:
|
|
219
|
+
syms.append(Symbol(name=f"{class_name}.{f['name']}", kind="field",
|
|
220
|
+
start=0, end=None, basis="classfile"))
|
|
221
|
+
return class_name, FileSymbols(symbols=syms, basis="classfile")
|