symgrep-codesearch 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
symgrep/__init__.py ADDED
File without changes
symgrep/blocks.py ADDED
@@ -0,0 +1,326 @@
1
+ """Block-granular units: sub-function AST blocks + a scope signature, for search-by-MEANING.
2
+
3
+ WHY THIS MODULE EXISTS. `functions()` (semantic.py) indexes whole functions, so search-by-meaning finds a
4
+ duplicated FUNCTION but misses a duplicated BLOCK — the same coherent snippet living inside two otherwise-
5
+ different functions (the sub-function Type-4 clone honestreview's canonical_concerns axis asks for). This
6
+ module extracts those blocks and gives each a deterministic SCOPE SIGNATURE (the block as an "environment":
7
+ what it reads, what it writes, what it does), so retrieval separates a domain accumulation from a generic
8
+ loop by more than a prose sentence. See docs/BLOCK_CHUNKING_SPEC.md.
9
+
10
+ WHAT IS AND IS NOT A JUDGEMENT (the same line semantic.py draws):
11
+ · Extracting a block and computing its signature is PARSING — the tree-sitter AST settles it exactly, no
12
+ meaning decision. Alpha-renaming invariance (two clones with different local names → same signature) is a
13
+ property of dropping plain-identifier leaf NAMES, not of any classifier.
14
+ · Turning a block into a one-sentence job description is a judgement → an LLM (cli.py's describer), grounded
15
+ by the signature. Matching a query to those is RETRIEVAL (semantic.py). This module decides nothing about
16
+ duplication; it yields candidate units, and the caller's model judges "same job".
17
+
18
+ CHANGE-DETECTION IS CONTENT, NOT MTIME. A block's identity (its `key`) embeds the sha256 of its own body, and
19
+ `body_hash` is carried explicitly, so the build's incremental logic keys re-describe on CONTENT: a block that
20
+ merely moves keeps its key (no re-describe), a body that changes gets a new key. `stamp` (mtime:size) is
21
+ carried ONLY as the per-file freshness DISPLAY signal search_by_meaning renders — it never gates re-describe
22
+ (mirrors functions() / semantic.py's documented stamp-is-a-signal contract), so a `touch`/restore/sync that
23
+ resets mtime does NOT cause a reindex.
24
+
25
+ DEGRADES, NEVER DEPENDS. Block extraction is an OPTIONAL layer. If the tree-sitter pack is absent the layer is
26
+ UNAVAILABLE and says so ONCE (never a silent per-file swallow), returning no blocks while functions() keeps
27
+ indexing. A per-file cap engaging is COUNTED and REPORTED, never a silent drop. No broad `except` hides a
28
+ real failure: a genuine parse crash propagates as the bug it is.
29
+
30
+ Reuses structural.py's tree-sitter parse (byte spans + node.type, all TS_LANG_BY_EXT languages) and
31
+ enclosing.file_symbols for the enclosing-symbol nesting. No new dependency.
32
+ """
33
+ import pathlib
34
+ import sys
35
+ from typing import Dict, List, Optional
36
+
37
+ from .semantic import _cfg_int, _hash8 # shared helpers (one definition — name-uniqueness)
38
+
39
+ # A block = a coherent standalone AST sub-tree. Matched by tree-sitter TYPE SUFFIX so ONE rule spans grammars
40
+ # (the C-family/python grammars share these type names; the suffix match tolerates the prefixes that differ).
41
+ BLOCK_TYPE_SUFFIXES = ("for_statement", "for_in_statement", "foreach_statement", "while_statement",
42
+ "if_statement", "switch_statement", "match_statement", "try_statement",
43
+ "with_statement", "case_statement", "catch_clause")
44
+ # Nodes whose leaf identifier IS semantically meaningful across clones (kept in the signature): a call target
45
+ # and an accessed attribute/member/field/property name. A PLAIN variable identifier is dropped (alpha-
46
+ # invariance) but counted, so the signature is shape + API surface, not variable names.
47
+ _CALL_SUFFIXES = ("call", "call_expression", "method_invocation", "function_call")
48
+ _ATTR_SUFFIXES = ("attribute", "member_expression", "field_expression", "selector_expression",
49
+ "member_access_expression", "scoped_identifier")
50
+ _ASSIGN_SUFFIXES = ("assignment", "assignment_expression", "augmented_assignment",
51
+ "assignment_statement", "variable_declarator", "short_var_declaration")
52
+
53
+ MIN_BLOCK_LINES_DEFAULT = 3
54
+ MIN_BLOCK_CHARS_DEFAULT = 60
55
+ BLOCK_MAX_PER_FILE_DEFAULT = 200
56
+
57
+
58
+ class BlockLayerUnavailable(RuntimeError):
59
+ """The tree-sitter pack needed for block extraction is not installed. Raised so the OPTIONAL block layer
60
+ says 'unavailable' once, loudly, rather than silently returning no blocks — unavailable ≠ empty."""
61
+
62
+
63
+ def _lang_of(path: pathlib.Path) -> Optional[str]:
64
+ from .enclosing import TS_LANG_BY_EXT
65
+ return TS_LANG_BY_EXT.get(path.suffix.lower())
66
+
67
+
68
+ def _is_func_node(t: str) -> bool:
69
+ return (t.endswith("function_definition") or t.endswith("method_definition")
70
+ or t in ("function_declaration", "method_declaration", "function_item", "arrow_function"))
71
+
72
+
73
+ def _func_nodes(root):
74
+ """All function/method nodes in the tree (stack walk, like structural.search_source)."""
75
+ out, stack = [], [root]
76
+ while stack:
77
+ n = stack.pop()
78
+ if _is_func_node(n.type):
79
+ out.append(n)
80
+ stack.extend(n.children)
81
+ return out
82
+
83
+
84
+ def _block_nodes(fn_node):
85
+ """Qualifying block nodes inside a function — OUTERMOST WINS (do not descend into a collected block, so a
86
+ nested if inside a for is not indexed twice and blocks never overlap)."""
87
+ out, stack = [], list(fn_node.children)
88
+ while stack:
89
+ n = stack.pop()
90
+ if any(n.type.endswith(suf) for suf in BLOCK_TYPE_SUFFIXES):
91
+ out.append(n)
92
+ continue # outermost wins — do NOT descend into this block's children
93
+ stack.extend(n.children)
94
+ return out
95
+
96
+
97
+ def _suffix_any(t: str, suffixes) -> bool:
98
+ return any(t.endswith(s) or t == s for s in suffixes)
99
+
100
+
101
+ def _rightmost_identifier(node, data: bytes) -> Optional[str]:
102
+ """The last identifier leaf under `node` (obj.attr.method → 'method'); the meaningful API name."""
103
+ order = []
104
+ stack = [node]
105
+ while stack:
106
+ n = stack.pop()
107
+ if n.type.endswith("identifier") or n.type.endswith("property_identifier") \
108
+ or n.type.endswith("field_identifier"):
109
+ order.append((n.start_byte, n))
110
+ stack.extend(n.children)
111
+ if not order:
112
+ return None
113
+ order.sort()
114
+ last = order[-1][1]
115
+ txt = data[last.start_byte:last.end_byte].decode("utf-8", "ignore")
116
+ return txt or None
117
+
118
+
119
+ def scope_signature(node, data: bytes) -> str:
120
+ """A canonical, alpha-renaming-invariant SHAPE+API signature of a block — the block as an environment.
121
+
122
+ Deterministic parsing, no judgement. Captures: the control-type skeleton (order-sensitive), the sorted
123
+ unique call targets + accessed member names (the API surface — these ARE meaningful across clones), the
124
+ operator multiset, literal KINDS (not values), and counts of assigned vs read plain identifiers (a size
125
+ signal for in/out, names dropped for alpha-invariance). Two Type-4 clones that differ only in local
126
+ variable names produce the SAME signature; two blocks that call different methods do not.
127
+
128
+ This is retrieval enrichment + a $0 exact-clone pre-filter, explicitly NOT a resolved dataflow: it does
129
+ not prove two blocks are the same, it makes the embedding and the describe prompt see the block's shape.
130
+ """
131
+ ctl = [] # control skeleton, in walk order
132
+ calls, attrs, ops = set(), set(), []
133
+ lits = set()
134
+ assigned, read = 0, 0
135
+ assign_target_ids = set()
136
+
137
+ def _collect_targets(n):
138
+ kids = n.named_children if n.named_children else n.children
139
+ if kids:
140
+ st = [kids[0]]
141
+ while st:
142
+ m = st.pop()
143
+ if m.type.endswith("identifier"):
144
+ assign_target_ids.add(m.id)
145
+ st.extend(m.children)
146
+
147
+ stack = [node]
148
+ while stack:
149
+ n = stack.pop()
150
+ t = n.type
151
+ if any(t.endswith(suf) for suf in BLOCK_TYPE_SUFFIXES):
152
+ ctl.append(next(s for s in BLOCK_TYPE_SUFFIXES if t.endswith(s)))
153
+ if _suffix_any(t, _ASSIGN_SUFFIXES):
154
+ _collect_targets(n)
155
+ if _suffix_any(t, _CALL_SUFFIXES):
156
+ fnc = n.child_by_field_name("function") if hasattr(n, "child_by_field_name") else None
157
+ target = fnc or (n.named_children[0] if n.named_children else None)
158
+ if target is not None:
159
+ name = _rightmost_identifier(target, data)
160
+ if name:
161
+ calls.add(name)
162
+ if _suffix_any(t, _ATTR_SUFFIXES):
163
+ name = _rightmost_identifier(n, data)
164
+ if name:
165
+ attrs.add(name)
166
+ if t.endswith("identifier"):
167
+ if n.id in assign_target_ids:
168
+ assigned += 1
169
+ else:
170
+ read += 1
171
+ elif t.endswith("string") or t == "string_literal":
172
+ lits.add("str")
173
+ elif t.endswith("integer") or t.endswith("number") or t == "int_literal" or t.endswith("float"):
174
+ lits.add("num")
175
+ elif t in ("true", "false", "null", "nil"):
176
+ lits.add("bool")
177
+ elif not n.children and not n.is_named:
178
+ tok = data[n.start_byte:n.end_byte].decode("utf-8", "ignore")
179
+ if tok and all(c in "+-*/%<>=!&|^~" for c in tok) and tok != "=":
180
+ ops.append(tok)
181
+ stack.extend(n.children)
182
+ parts = [
183
+ "ctl:" + ">".join(ctl),
184
+ "call:" + ",".join(sorted(calls)),
185
+ "attr:" + ",".join(sorted(attrs)),
186
+ "op:" + ",".join(sorted(ops)),
187
+ "lit:" + ",".join(sorted(lits)),
188
+ f"io:{read}r/{assigned}w",
189
+ ]
190
+ return " ".join(parts)
191
+
192
+
193
+ def _parsers_for(langs) -> Dict[str, object]:
194
+ """Acquire a tree-sitter parser per language ONCE. If the pack is absent, raise BlockLayerUnavailable —
195
+ the OPTIONAL layer is unavailable and says so once, never a per-file silent swallow. A concrete parser
196
+ that cannot be built for one known language is left out (that language simply yields no blocks); a missing
197
+ PACK (ImportError) fails the whole layer loudly because nothing can parse."""
198
+ from . import structural
199
+ from .enclosing import TreeSitterUnavailable
200
+ parsers = {}
201
+ for lang in sorted(set(langs)):
202
+ try:
203
+ parsers[lang] = structural._get_parser(lang)
204
+ except (ImportError, TreeSitterUnavailable) as e:
205
+ raise BlockLayerUnavailable(
206
+ "no usable tree-sitter grammar pack, so block extraction is UNAVAILABLE — this is NOT "
207
+ "'no blocks found'. Install tree-sitter-language-pack, or leave SYMGREP_INDEX_BLOCKS unset. "
208
+ f"[{e}]") from e
209
+ return parsers
210
+
211
+
212
+ def blocks(roots: List[str]) -> List[dict]:
213
+ """Every qualifying sub-function block under `roots`, mirroring functions()' dict shape plus block fields.
214
+
215
+ Returned dict: {key, qual, path, lineno, body, body_hash, stamp, unit:"block", parent_key, block_start,
216
+ block_end, signature, sig_hash}. Key = "<abspath>::<parent_sym>#<block_kind>@<body_hash8>" — CONTENT-keyed
217
+ (a block that moves keeps its key; a changed body gets a new one). Identical blocks in one function
218
+ collide on key → deduped (they ARE the same content). Bounded by min-size + per-file cap; skips trivia.
219
+ Every drop (per-file cap) and skip (unparsed file) is counted and REPORTED, never silent.
220
+ """
221
+ from .enclosing import file_symbols
222
+ from .semantic import INDEXABLE_EXTS, _is_generated, _looks_minified, _module_name
223
+
224
+ min_lines = _cfg_int("SYMGREP_BLOCK_MIN_LINES", MIN_BLOCK_LINES_DEFAULT)
225
+ min_chars = _cfg_int("SYMGREP_BLOCK_MIN_CHARS", MIN_BLOCK_CHARS_DEFAULT)
226
+ max_per_file = _cfg_int("SYMGREP_BLOCK_MAX_PER_FILE", BLOCK_MAX_PER_FILE_DEFAULT)
227
+
228
+ # Enumerate candidate files first, so parsers are acquired once for exactly the languages present.
229
+ cand = []
230
+ for root in roots:
231
+ rp = pathlib.Path(root)
232
+ files = [rp] if rp.is_file() else sorted(
233
+ p for p in rp.rglob("*")
234
+ if p.is_file() and p.suffix.lower() in INDEXABLE_EXTS and not _is_generated(p, rp))
235
+ for f in files:
236
+ lang = _lang_of(f)
237
+ if lang:
238
+ cand.append((f, lang))
239
+ if not cand:
240
+ return []
241
+ parsers = _parsers_for(l for _f, l in cand) # raises BlockLayerUnavailable if the pack is absent (loud)
242
+
243
+ out = []
244
+ capped_files = 0 # files where the per-file cap engaged (blocks beyond the cap intentionally omitted)
245
+ capped_blocks = 0 # count of qualifying blocks omitted by the cap — the honest denominator
246
+ parse_skipped = 0 # files whose bytes would not parse (counted, not swallowed)
247
+ for f, lang in cand:
248
+ parser = parsers.get(lang)
249
+ if parser is None:
250
+ continue
251
+ try:
252
+ st = f.stat()
253
+ if st.st_size > 0 and getattr(st, "st_blocks", 1) == 0: # iCloud dataless guard (as functions())
254
+ continue
255
+ text = f.read_text(encoding="utf-8", errors="ignore")
256
+ except OSError:
257
+ continue
258
+ if _looks_minified(text):
259
+ continue
260
+ data = text.encode("utf-8")
261
+ tree = parser.parse(data) # valid bytes parse; a genuine crash propagates (a real bug)
262
+ if tree is None or tree.root_node is None:
263
+ parse_skipped += 1
264
+ continue
265
+ fs = file_symbols(str(f))
266
+ stamp = f"{st.st_mtime_ns}:{st.st_size}" # DISPLAY freshness signal only — identity is body_hash
267
+ mod = _module_name(f)
268
+ n_file = 0
269
+ capped_here = False
270
+ seen_keys = set()
271
+ for fn in _func_nodes(tree.root_node):
272
+ fn_line = fn.start_point[0] + 1
273
+ parent_sym = _enclosing_name(fs, fn_line) or f"<anon@{fn_line}>"
274
+ for blk in _block_nodes(fn):
275
+ body = data[blk.start_byte:blk.end_byte].decode("utf-8", "ignore")
276
+ if body.count("\n") + 1 < min_lines or len(body) < min_chars:
277
+ continue
278
+ if n_file >= max_per_file: # cap engaged — COUNT the omission, do not silently drop
279
+ capped_blocks += 1
280
+ capped_here = True
281
+ continue
282
+ bh = _hash8(body)
283
+ block_kind = next(s for s in BLOCK_TYPE_SUFFIXES if blk.type.endswith(s))
284
+ key = f"{f.resolve()}::{parent_sym}#{block_kind}@{bh}"
285
+ if key in seen_keys: # exact-dup within this function → dedup (same content)
286
+ continue
287
+ seen_keys.add(key)
288
+ sig = scope_signature(blk, data)
289
+ start_line = blk.start_point[0] + 1
290
+ out.append({
291
+ "key": key,
292
+ "qual": f"{mod}.{parent_sym}#{block_kind}@{bh}", # whitespace-free (MCP contract)
293
+ "path": str(f), "lineno": start_line, "body": body, "body_hash": bh, "stamp": stamp,
294
+ "unit": "block", "parent_key": f"{f.resolve()}::{parent_sym}",
295
+ "block_start": start_line, "block_end": blk.end_point[0] + 1,
296
+ "signature": sig, "sig_hash": _hash8(sig),
297
+ })
298
+ n_file += 1
299
+ if capped_here:
300
+ capped_files += 1
301
+ if capped_blocks or parse_skipped:
302
+ sys.stderr.write(
303
+ f"symgrep block-index: {len(out)} block(s); "
304
+ f"OMITTED {capped_blocks} block(s) past the per-file cap ({max_per_file}) in {capped_files} file(s) "
305
+ f"[raise SYMGREP_BLOCK_MAX_PER_FILE]; {parse_skipped} file(s) would not parse (counted, not "
306
+ f"described).\n")
307
+ return out
308
+
309
+
310
+ def _enclosing_name(fs, line: int) -> Optional[str]:
311
+ """The innermost enclosing symbol name at `line`, via FileSymbols.enclosing when available, else a manual
312
+ innermost-span scan over fs.symbols."""
313
+ if fs is None:
314
+ return None
315
+ enc = getattr(fs, "enclosing", None)
316
+ if callable(enc):
317
+ sym = enc(line)
318
+ if sym is not None:
319
+ return sym.name
320
+ best = None
321
+ for s in (getattr(fs, "symbols", None) or []):
322
+ if s.start is None or s.end is None:
323
+ continue
324
+ if s.start <= line <= s.end and (best is None or s.start > best.start):
325
+ best = s
326
+ return best.name if best else None
symgrep/classfile.py ADDED
@@ -0,0 +1,221 @@
1
+ """Symbols from a compiled JVM `.class` file — the constant pool read directly, no JDK.
2
+
3
+ WHY THIS IS DIFFERENT FROM EVERY OTHER SYMBOL PATH. Source (.java, .py, …) is TEXT: ripgrep matches it
4
+ and tree-sitter/ast gives exact spans. A `.class` is BINARY bytecode — ripgrep, symgrep's whole Tier 1,
5
+ cannot search it, and there is no source text to grep. But the JVM classfile format (JVMS §4) is fixed
6
+ and self-describing: the constant pool holds the class name, every method name + type descriptor, and
7
+ every field name. Extracting them is reading a documented binary layout — PARSING, exactly like
8
+ tree-sitter reads a grammar, so it carries no model call and its `basis` is "classfile" (honest: this
9
+ came from bytecode, not source).
10
+
11
+ WHAT IT CAN AND CANNOT KNOW (stated, never blurred):
12
+ · CAN: class name, method names + descriptors (overloads disambiguated), field names — the public
13
+ shape a `file_symbols` consumer wants when the source is a dependency it does not have.
14
+ · Line spans: ONLY if the class was compiled with debug info (the optional Code→LineNumberTable
15
+ attribute). Present → real spans; absent → start=0, end=None, and the basis still says "classfile"
16
+ so a consumer never mistakes a missing span for line 0.
17
+ · CANNOT: local variable names, comments, or a trustworthy one-sentence "what it does" — bytecode is
18
+ a lossy witness of intent, which is why `.class` feeds the STRUCTURAL file_symbols path and is
19
+ deliberately kept OUT of the semantic (meaning) index. Describing a method from its bytecode would
20
+ be a confident-wrong-answer; if you have the source, index the source.
21
+
22
+ This slots into enclosing.file_symbols so the existing `file_symbols(path)` MCP tool answers for a
23
+ `.class` with no new surface — a dependency jar's symbols become queryable without unzipping to source.
24
+ """
25
+ import struct
26
+ from typing import Dict, List, Optional, Tuple
27
+
28
+ # Constant-pool tag → how many bytes its body occupies after the 1-byte tag, and whether it eats an
29
+ # extra pool slot (Long/Double do — JVMS §4.4.5). This is the format, not a heuristic.
30
+ _CP_LAYOUT: Dict[int, Tuple[int, bool]] = {
31
+ 1: (-1, False), # Utf8: u2 length then <length> bytes (handled specially)
32
+ 3: (4, False), # Integer
33
+ 4: (4, False), # Float
34
+ 5: (8, True), # Long — takes TWO pool slots
35
+ 6: (8, True), # Double — takes TWO pool slots
36
+ 7: (2, False), # Class → name_index
37
+ 8: (2, False), # String
38
+ 9: (4, False), # Fieldref
39
+ 10: (4, False), # Methodref
40
+ 11: (4, False), # InterfaceMethodref
41
+ 12: (4, False), # NameAndType
42
+ 15: (3, False), # MethodHandle
43
+ 16: (2, False), # MethodType
44
+ 17: (4, False), # Dynamic
45
+ 18: (4, False), # InvokeDynamic
46
+ 19: (2, False), # Module
47
+ 20: (2, False), # Package
48
+ }
49
+
50
+ MAGIC = 0xCAFEBABE
51
+
52
+
53
+ class NotAClassFile(ValueError):
54
+ pass
55
+
56
+
57
+ class _Reader:
58
+ def __init__(self, data: bytes):
59
+ self.d = data
60
+ self.i = 0
61
+
62
+ def u1(self) -> int:
63
+ v = self.d[self.i]
64
+ self.i += 1
65
+ return v
66
+
67
+ def u2(self) -> int:
68
+ v = struct.unpack_from(">H", self.d, self.i)[0]
69
+ self.i += 2
70
+ return v
71
+
72
+ def u4(self) -> int:
73
+ v = struct.unpack_from(">I", self.d, self.i)[0]
74
+ self.i += 4
75
+ return v
76
+
77
+ def take(self, n: int) -> bytes:
78
+ v = self.d[self.i:self.i + n]
79
+ self.i += n
80
+ return v
81
+
82
+
83
+ def _mutf8(b: bytes) -> str:
84
+ """Decode JVM "modified UTF-8" (JVMS 4.4.7): U+0000 encoded as 0xC0 0x80, supplementary chars as
85
+ CESU-8 surrogate pairs. A plain utf-8 decode silently corrupts both. Fast path for the ASCII/BMP
86
+ names that dominate; the manual walk runs only when the modified encodings are actually present."""
87
+ if 0xC0 not in b and 0xED not in b:
88
+ try:
89
+ return b.decode("utf-8")
90
+ except UnicodeDecodeError:
91
+ pass
92
+ out = []
93
+ i, n = 0, len(b)
94
+ while i < n:
95
+ c = b[i]
96
+ if c == 0xC0 and i + 1 < n and b[i + 1] == 0x80:
97
+ out.append("\x00"); i += 2
98
+ elif c < 0x80:
99
+ out.append(chr(c)); i += 1
100
+ elif c == 0xED and i + 5 < n and (b[i + 1] & 0xF0) == 0xA0: # 6-byte CESU-8 surrogate pair
101
+ hi = 0xD000 | ((b[i + 1] & 0x3F) << 6) | (b[i + 2] & 0x3F)
102
+ lo = 0xD000 | ((b[i + 4] & 0x3F) << 6) | (b[i + 5] & 0x3F)
103
+ out.append(chr(0x10000 + (((hi - 0xD800) << 10) | (lo - 0xDC00)))); i += 6
104
+ elif 0xC0 <= c < 0xE0 and i + 1 < n:
105
+ out.append(chr(((c & 0x1F) << 6) | (b[i + 1] & 0x3F))); i += 2
106
+ elif 0xE0 <= c < 0xF0 and i + 2 < n:
107
+ out.append(chr(((c & 0x0F) << 12) | ((b[i + 1] & 0x3F) << 6) | (b[i + 2] & 0x3F))); i += 3
108
+ else:
109
+ out.append("�"); i += 1
110
+ return "".join(out)
111
+
112
+
113
+ def _parse_constant_pool(r: _Reader) -> Dict[int, object]:
114
+ count = r.u2()
115
+ pool: Dict[int, object] = {}
116
+ idx = 1
117
+ while idx < count:
118
+ tag = r.u1()
119
+ if tag == 1: # Utf8
120
+ length = r.u2()
121
+ pool[idx] = ("Utf8", _mutf8(r.take(length)))
122
+ elif tag == 7: # Class → name_index
123
+ pool[idx] = ("Class", r.u2())
124
+ elif tag in _CP_LAYOUT:
125
+ size, wide = _CP_LAYOUT[tag]
126
+ body = r.take(size)
127
+ pool[idx] = (tag, body)
128
+ if wide:
129
+ idx += 1 # Long/Double occupy this slot AND the next
130
+ else:
131
+ raise NotAClassFile(f"unknown constant-pool tag {tag} at index {idx}")
132
+ idx += 1
133
+ return pool
134
+
135
+
136
+ def _utf8(pool: Dict[int, object], index: int) -> Optional[str]:
137
+ v = pool.get(index)
138
+ return v[1] if isinstance(v, tuple) and v[0] == "Utf8" else None
139
+
140
+
141
+ def _class_name(pool: Dict[int, object], class_index: int) -> Optional[str]:
142
+ v = pool.get(class_index)
143
+ if isinstance(v, tuple) and v[0] == "Class":
144
+ raw = _utf8(pool, v[1])
145
+ return raw.replace("/", ".") if raw else None # internal "a/b/C" → "a.b.C"
146
+ return None
147
+
148
+
149
+ def _member_symbols(r: _Reader, pool: Dict[int, object], class_name: str, kind: str) -> List[dict]:
150
+ out: List[dict] = []
151
+ count = r.u2()
152
+ for _ in range(count):
153
+ r.u2() # access_flags
154
+ name = _utf8(pool, r.u2())
155
+ desc = _utf8(pool, r.u2())
156
+ span: Optional[Tuple[int, int]] = None
157
+ attr_count = r.u2()
158
+ for _a in range(attr_count):
159
+ attr_name = _utf8(pool, r.u2())
160
+ attr_len = r.u4()
161
+ body_start = r.i
162
+ if kind == "method" and attr_name == "Code":
163
+ span = _code_line_span(r, pool)
164
+ r.i = body_start + attr_len # always skip by declared length (robust)
165
+ out.append({"name": name, "desc": desc, "kind": kind, "span": span})
166
+ return out
167
+
168
+
169
+ def _code_line_span(r: _Reader, pool: Dict[int, object]) -> Optional[Tuple[int, int]]:
170
+ """Read a Code attribute far enough to find its LineNumberTable (optional debug info)."""
171
+ r.u2(); r.u2() # max_stack, max_locals
172
+ code_len = r.u4()
173
+ r.i += code_len # skip bytecode
174
+ exc_len = r.u2()
175
+ r.i += exc_len * 8 # exception table
176
+ lines: List[int] = []
177
+ sub_count = r.u2()
178
+ for _ in range(sub_count):
179
+ sub_name = _utf8(pool, r.u2())
180
+ sub_len = r.u4()
181
+ start = r.i
182
+ if sub_name == "LineNumberTable":
183
+ n = r.u2()
184
+ for _e in range(n):
185
+ r.u2() # start_pc
186
+ lines.append(r.u2()) # line_number
187
+ r.i = start + sub_len
188
+ return (min(lines), max(lines)) if lines else None
189
+
190
+
191
+ def classfile_symbols(data: bytes):
192
+ """-> (class_name, [Symbol]) via enclosing.Symbol. Raises NotAClassFile on a non-class buffer."""
193
+ from .enclosing import FileSymbols, Symbol
194
+ r = _Reader(data)
195
+ if r.u4() != MAGIC:
196
+ raise NotAClassFile("bad magic (not 0xCAFEBABE)")
197
+ r.u2(); r.u2() # minor, major version
198
+ pool = _parse_constant_pool(r)
199
+ r.u2() # access_flags
200
+ class_name = _class_name(pool, r.u2()) or "(unknown)"
201
+ r.u2() # super_class
202
+ n_ifaces = r.u2()
203
+ r.i += n_ifaces * 2 # interfaces[]
204
+ fields = _member_symbols(r, pool, class_name, "field")
205
+ methods = _member_symbols(r, pool, class_name, "method")
206
+
207
+ # Disambiguate overloaded methods by appending the descriptor — keeps names UNIQUE (symgrep's ethos)
208
+ # only where they would collide, so the common case stays clean `Class.method`.
209
+ seen: Dict[str, int] = {}
210
+ for m in methods:
211
+ seen[m["name"]] = seen.get(m["name"], 0) + 1
212
+ syms: List[Symbol] = []
213
+ for m in methods:
214
+ base = f"{class_name}.{m['name']}"
215
+ name = f"{base}{m['desc']}" if seen[m["name"]] > 1 else base
216
+ start, end = (m["span"] if m["span"] else (0, None))
217
+ syms.append(Symbol(name=name, kind="method", start=start, end=end, basis="classfile"))
218
+ for f in fields:
219
+ syms.append(Symbol(name=f"{class_name}.{f['name']}", kind="field",
220
+ start=0, end=None, basis="classfile"))
221
+ return class_name, FileSymbols(symbols=syms, basis="classfile")