python-delphi-lsp 3.6.0__cp310-abi3-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- delphi_lsp/__init__.py +184 -0
- delphi_lsp/_version.py +1 -0
- delphi_lsp/agent_cache.py +1574 -0
- delphi_lsp/agent_cli.py +902 -0
- delphi_lsp/agent_context.py +3907 -0
- delphi_lsp/agent_cpg.py +349 -0
- delphi_lsp/agent_cpg_builder.py +666 -0
- delphi_lsp/agent_layers.py +1410 -0
- delphi_lsp/agent_metrics.py +526 -0
- delphi_lsp/agent_protocol.py +445 -0
- delphi_lsp/agent_relations.py +1881 -0
- delphi_lsp/agent_templates.py +614 -0
- delphi_lsp/agent_wiki.py +2853 -0
- delphi_lsp/agent_workspace.py +860 -0
- delphi_lsp/ast_serialize.py +126 -0
- delphi_lsp/binary.py +250 -0
- delphi_lsp/comment_builder.py +28 -0
- delphi_lsp/consts.py +345 -0
- delphi_lsp/delphi_lsp_native.pyd +0 -0
- delphi_lsp/delphiast_lexer.py +399 -0
- delphi_lsp/delphiast_parser.py +2189 -0
- delphi_lsp/delphiast_tokens.py +41 -0
- delphi_lsp/grammar.py +459 -0
- delphi_lsp/incremental.py +1139 -0
- delphi_lsp/lark_builder.py +2684 -0
- delphi_lsp/lark_tokens.py +236 -0
- delphi_lsp/lexical_scanner.py +30 -0
- delphi_lsp/lsp_server.py +6698 -0
- delphi_lsp/metrics.py +1452 -0
- delphi_lsp/native_bridge.py +300 -0
- delphi_lsp/navigation_cache.py +189 -0
- delphi_lsp/nodes.py +462 -0
- delphi_lsp/parallel_outline.py +662 -0
- delphi_lsp/parse_artifacts.py +969 -0
- delphi_lsp/parser.py +342 -0
- delphi_lsp/parser_backend.py +139 -0
- delphi_lsp/preprocessor.py +1122 -0
- delphi_lsp/progress.py +26 -0
- delphi_lsp/project_config.py +285 -0
- delphi_lsp/project_discovery.py +932 -0
- delphi_lsp/project_indexer.py +470 -0
- delphi_lsp/py.typed +1 -0
- delphi_lsp/response_cache.py +465 -0
- delphi_lsp/semantic.py +394 -0
- delphi_lsp/semantic_builder.py +2302 -0
- delphi_lsp/source_reader.py +17 -0
- delphi_lsp/wiki_external_sort.py +392 -0
- delphi_lsp/wiki_memory.py +548 -0
- delphi_lsp/workspace.py +83 -0
- delphi_lsp/writer.py +73 -0
- python_delphi_lsp-3.6.0.dist-info/METADATA +766 -0
- python_delphi_lsp-3.6.0.dist-info/RECORD +56 -0
- python_delphi_lsp-3.6.0.dist-info/WHEEL +4 -0
- python_delphi_lsp-3.6.0.dist-info/entry_points.txt +3 -0
- python_delphi_lsp-3.6.0.dist-info/licenses/LICENSE +373 -0
- python_delphi_lsp-3.6.0.dist-info/sboms/python-delphi-lsp.cyclonedx.json +770 -0
|
@@ -0,0 +1,3907 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import gc
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
import sys
|
|
8
|
+
import time
|
|
9
|
+
import unicodedata
|
|
10
|
+
from bisect import bisect_left, bisect_right
|
|
11
|
+
from collections import OrderedDict
|
|
12
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
13
|
+
from dataclasses import dataclass, field, replace
|
|
14
|
+
from functools import lru_cache
|
|
15
|
+
from heapq import heappop, heappush
|
|
16
|
+
from pathlib import Path, PureWindowsPath
|
|
17
|
+
|
|
18
|
+
from .agent_cpg import CpgSubgraph, CpgTarget
|
|
19
|
+
from .agent_cpg_builder import build_cpg_subgraph
|
|
20
|
+
from .agent_metrics import (
|
|
21
|
+
build_workspace_metrics,
|
|
22
|
+
project_metric_item,
|
|
23
|
+
unit_metric_item,
|
|
24
|
+
)
|
|
25
|
+
from .agent_protocol import (
|
|
26
|
+
AgentProtocolError,
|
|
27
|
+
AgentRequest,
|
|
28
|
+
AgentResponse,
|
|
29
|
+
ContextBudget,
|
|
30
|
+
Focus,
|
|
31
|
+
make_target_id,
|
|
32
|
+
paginate_items,
|
|
33
|
+
)
|
|
34
|
+
from .agent_relations import ProjectRelationIndex, RelationTarget
|
|
35
|
+
from .agent_workspace import (
|
|
36
|
+
AgentUnit,
|
|
37
|
+
AgentWorkspace,
|
|
38
|
+
unit_display_path,
|
|
39
|
+
unit_source_path,
|
|
40
|
+
unit_target_id,
|
|
41
|
+
)
|
|
42
|
+
from .consts import AttributeName, SyntaxNodeType
|
|
43
|
+
from .lsp_server import build_outline_semantic_model, multiline_string_block_end
|
|
44
|
+
from .metrics import ProjectMetrics
|
|
45
|
+
from .navigation_cache import NavigationShardStore, navigation_cache_key
|
|
46
|
+
from .nodes import CompoundSyntaxNode, SyntaxNode
|
|
47
|
+
from .parallel_outline import (
|
|
48
|
+
ParallelBuildStats,
|
|
49
|
+
ParallelOutlineError,
|
|
50
|
+
run_outline_tasks,
|
|
51
|
+
)
|
|
52
|
+
from .parser import DelphiParser
|
|
53
|
+
from .parser_backend import ParserMode
|
|
54
|
+
from .project_config import ProjectPathConfig, workspace_include_loader
|
|
55
|
+
from .semantic import (
|
|
56
|
+
Scope,
|
|
57
|
+
ScopeKind,
|
|
58
|
+
Symbol,
|
|
59
|
+
SymbolKind,
|
|
60
|
+
Visibility,
|
|
61
|
+
)
|
|
62
|
+
from .source_reader import read_source_text
|
|
63
|
+
|
|
64
|
+
_ROUTINE_KINDS = frozenset(
|
|
65
|
+
{
|
|
66
|
+
SymbolKind.METHOD,
|
|
67
|
+
SymbolKind.FUNCTION,
|
|
68
|
+
SymbolKind.PROCEDURE,
|
|
69
|
+
SymbolKind.CONSTRUCTOR,
|
|
70
|
+
SymbolKind.DESTRUCTOR,
|
|
71
|
+
}
|
|
72
|
+
)
|
|
73
|
+
_TYPE_KINDS = frozenset(
|
|
74
|
+
{
|
|
75
|
+
SymbolKind.TYPE,
|
|
76
|
+
SymbolKind.CLASS,
|
|
77
|
+
SymbolKind.RECORD,
|
|
78
|
+
SymbolKind.INTERFACE,
|
|
79
|
+
SymbolKind.ENUM,
|
|
80
|
+
}
|
|
81
|
+
)
|
|
82
|
+
_ROUTINE_WORDS = frozenset(
|
|
83
|
+
{"procedure", "function", "constructor", "destructor", "operator"}
|
|
84
|
+
)
|
|
85
|
+
_BLOCK_WORDS = frozenset({"begin", "case", "try", "asm"})
|
|
86
|
+
_STRUCTURED_TYPE_WORDS = frozenset(
|
|
87
|
+
{"class", "record", "object", "interface", "dispinterface"}
|
|
88
|
+
)
|
|
89
|
+
_NO_BODY_DIRECTIVES = frozenset({"abstract", "external", "forward"})
|
|
90
|
+
_ROUTINE_DIRECTIVES = frozenset(
|
|
91
|
+
{
|
|
92
|
+
"abstract",
|
|
93
|
+
"assembler",
|
|
94
|
+
"cdecl",
|
|
95
|
+
"deprecated",
|
|
96
|
+
"dispid",
|
|
97
|
+
"dynamic",
|
|
98
|
+
"experimental",
|
|
99
|
+
"export",
|
|
100
|
+
"external",
|
|
101
|
+
"final",
|
|
102
|
+
"forward",
|
|
103
|
+
"inline",
|
|
104
|
+
"message",
|
|
105
|
+
"noreturn",
|
|
106
|
+
"overload",
|
|
107
|
+
"override",
|
|
108
|
+
"pascal",
|
|
109
|
+
"platform",
|
|
110
|
+
"register",
|
|
111
|
+
"reintroduce",
|
|
112
|
+
"safecall",
|
|
113
|
+
"static",
|
|
114
|
+
"stdcall",
|
|
115
|
+
"unsafe",
|
|
116
|
+
"varargs",
|
|
117
|
+
"virtual",
|
|
118
|
+
"winapi",
|
|
119
|
+
}
|
|
120
|
+
)
|
|
121
|
+
_CALLING_CONVENTIONS = frozenset(
|
|
122
|
+
{"cdecl", "pascal", "register", "safecall", "stdcall", "winapi"}
|
|
123
|
+
)
|
|
124
|
+
_SOURCE_CHUNK_CHARS = 6000
|
|
125
|
+
_RANKED_QUERY_CACHE_SIZE = 16
|
|
126
|
+
_RANKED_QUERY_CACHE_MAX_ENTRIES = 50_000
|
|
127
|
+
_PREPARED_RESPONSE_CACHE_SIZE = 8
|
|
128
|
+
_DEFAULT_NAVIGATION_CACHE_MAX_BYTES = 512 * 1024**2
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass(frozen=True, slots=True)
|
|
132
|
+
class _Token:
|
|
133
|
+
value: str
|
|
134
|
+
start: int
|
|
135
|
+
end: int
|
|
136
|
+
word: bool = False
|
|
137
|
+
escaped: bool = False
|
|
138
|
+
directive: bool = False
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class _SourceDocument:
|
|
142
|
+
def __init__(
|
|
143
|
+
self,
|
|
144
|
+
source_path: Path,
|
|
145
|
+
display_path: str,
|
|
146
|
+
text: str,
|
|
147
|
+
*,
|
|
148
|
+
defines: tuple[str, ...] = (),
|
|
149
|
+
include_paths: tuple[str, ...] = (),
|
|
150
|
+
project_config: ProjectPathConfig | None = None,
|
|
151
|
+
) -> None:
|
|
152
|
+
self.source_path = source_path
|
|
153
|
+
self.display_path = display_path
|
|
154
|
+
self.text = text
|
|
155
|
+
self.defines = defines
|
|
156
|
+
self.include_paths = include_paths
|
|
157
|
+
self.project_config = project_config
|
|
158
|
+
self.line_starts = _line_starts(text)
|
|
159
|
+
self.tokens = tuple(_lex_delphi(text))
|
|
160
|
+
self.token_starts = tuple(token.start for token in self.tokens)
|
|
161
|
+
self.directive_starts = tuple(token.start for token in self.tokens if token.directive)
|
|
162
|
+
self.declaration_section_indexes = [0]
|
|
163
|
+
self.declaration_section_checkpoints: dict[int, tuple[str, int, int, int]] = {
|
|
164
|
+
0: ("", 0, 0, 0)
|
|
165
|
+
}
|
|
166
|
+
words = [token for token in self.tokens if token.word and not token.escaped]
|
|
167
|
+
self.unit_kind = next(
|
|
168
|
+
(token.value for token in words if token.value in {"unit", "program", "library", "package"}),
|
|
169
|
+
"",
|
|
170
|
+
)
|
|
171
|
+
implementation = next((token for token in words if token.value == "implementation"), None)
|
|
172
|
+
self.implementation_line = self.line_col(implementation.start)[0] if implementation else 0
|
|
173
|
+
self.routine_spans: dict[int, tuple[int, int] | None] = {}
|
|
174
|
+
self.routine_token_spans: dict[int, tuple[int, int, int] | None] = {}
|
|
175
|
+
self.parser_spans: dict[str, tuple[int, int] | None] = {}
|
|
176
|
+
self._full_parse_attempted = False
|
|
177
|
+
self._full_parse_result: object | None = None
|
|
178
|
+
self.retained_bytes = (
|
|
179
|
+
sys.getsizeof(self)
|
|
180
|
+
+ sys.getsizeof(self.text)
|
|
181
|
+
+ sys.getsizeof(self.line_starts)
|
|
182
|
+
+ len(self.line_starts) * 32
|
|
183
|
+
+ sys.getsizeof(self.tokens)
|
|
184
|
+
+ len(self.tokens) * 160
|
|
185
|
+
+ sys.getsizeof(self.token_starts)
|
|
186
|
+
+ len(self.token_starts) * 28
|
|
187
|
+
+ sys.getsizeof(self.directive_starts)
|
|
188
|
+
+ len(self.directive_starts) * 28
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
def offset(self, line: int, column: int = 1) -> int:
|
|
192
|
+
if not self.line_starts:
|
|
193
|
+
return 0
|
|
194
|
+
line_index = min(max(line - 1, 0), len(self.line_starts) - 1)
|
|
195
|
+
start = self.line_starts[line_index]
|
|
196
|
+
end = self.line_starts[line_index + 1] if line_index + 1 < len(self.line_starts) else len(self.text)
|
|
197
|
+
return min(start + max(column - 1, 0), end)
|
|
198
|
+
|
|
199
|
+
def line_start(self, line: int) -> int:
|
|
200
|
+
return self.offset(line, 1)
|
|
201
|
+
|
|
202
|
+
def line_end(self, line: int, *, include_newline: bool = False) -> int:
|
|
203
|
+
if line < len(self.line_starts):
|
|
204
|
+
end = self.line_starts[max(line, 0)]
|
|
205
|
+
else:
|
|
206
|
+
end = len(self.text)
|
|
207
|
+
if include_newline:
|
|
208
|
+
return end
|
|
209
|
+
while end > 0 and self.text[end - 1] in {"\r", "\n"}:
|
|
210
|
+
end -= 1
|
|
211
|
+
return end
|
|
212
|
+
|
|
213
|
+
def line_col(self, offset: int) -> tuple[int, int]:
|
|
214
|
+
clamped = min(max(offset, 0), len(self.text))
|
|
215
|
+
line_index = max(0, bisect_right(self.line_starts, clamped) - 1)
|
|
216
|
+
return line_index + 1, clamped - self.line_starts[line_index] + 1
|
|
217
|
+
|
|
218
|
+
def first_token_index(self, offset: int) -> int:
|
|
219
|
+
return bisect_left(self.token_starts, offset)
|
|
220
|
+
|
|
221
|
+
def contains_directive(self, start: int, end: int) -> bool:
|
|
222
|
+
index = bisect_left(self.directive_starts, start)
|
|
223
|
+
return index < len(self.directive_starts) and self.directive_starts[index] < end
|
|
224
|
+
|
|
225
|
+
def full_parse(self):
|
|
226
|
+
if not self._full_parse_attempted:
|
|
227
|
+
self._full_parse_attempted = True
|
|
228
|
+
try:
|
|
229
|
+
self._full_parse_result = DelphiParser(
|
|
230
|
+
defines=self.defines,
|
|
231
|
+
include_paths=self.include_paths,
|
|
232
|
+
include_loader=workspace_include_loader(
|
|
233
|
+
self.project_config,
|
|
234
|
+
self.include_paths,
|
|
235
|
+
),
|
|
236
|
+
).parse(self.text, str(self.source_path), build_semantic=False)
|
|
237
|
+
except Exception:
|
|
238
|
+
self._full_parse_result = None
|
|
239
|
+
return self._full_parse_result
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@dataclass(frozen=True, slots=True)
|
|
243
|
+
class _SourceSpec:
|
|
244
|
+
source_path: Path
|
|
245
|
+
display_path: str
|
|
246
|
+
defines: tuple[str, ...]
|
|
247
|
+
include_paths: tuple[str, ...]
|
|
248
|
+
project_config: ProjectPathConfig | None = None
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
class _SourceStore:
|
|
252
|
+
"""Load expensive tokenized source documents only when source evidence is requested."""
|
|
253
|
+
|
|
254
|
+
def __init__(
|
|
255
|
+
self,
|
|
256
|
+
specs: Mapping[Path, _SourceSpec],
|
|
257
|
+
*,
|
|
258
|
+
max_loaded_bytes: int,
|
|
259
|
+
) -> None:
|
|
260
|
+
self._specs = dict(specs)
|
|
261
|
+
self._loaded: OrderedDict[Path, _SourceDocument] = OrderedDict()
|
|
262
|
+
self._loaded_bytes = 0
|
|
263
|
+
self._max_loaded_bytes = max(0, max_loaded_bytes)
|
|
264
|
+
|
|
265
|
+
def __getitem__(self, source_path: Path) -> _SourceDocument:
|
|
266
|
+
cached = self._loaded.pop(source_path, None)
|
|
267
|
+
if cached is not None:
|
|
268
|
+
self._loaded[source_path] = cached
|
|
269
|
+
return cached
|
|
270
|
+
|
|
271
|
+
spec = self._specs[source_path]
|
|
272
|
+
document = _SourceDocument(
|
|
273
|
+
spec.source_path,
|
|
274
|
+
spec.display_path,
|
|
275
|
+
read_source_text(spec.source_path),
|
|
276
|
+
defines=spec.defines,
|
|
277
|
+
include_paths=spec.include_paths,
|
|
278
|
+
project_config=spec.project_config,
|
|
279
|
+
)
|
|
280
|
+
if document.retained_bytes > self._max_loaded_bytes:
|
|
281
|
+
return document
|
|
282
|
+
while (
|
|
283
|
+
self._loaded
|
|
284
|
+
and self._loaded_bytes + document.retained_bytes > self._max_loaded_bytes
|
|
285
|
+
):
|
|
286
|
+
_, evicted = self._loaded.popitem(last=False)
|
|
287
|
+
self._loaded_bytes -= evicted.retained_bytes
|
|
288
|
+
self._loaded[source_path] = document
|
|
289
|
+
self._loaded_bytes += document.retained_bytes
|
|
290
|
+
return document
|
|
291
|
+
|
|
292
|
+
def add_spec(self, spec: _SourceSpec) -> None:
|
|
293
|
+
self._specs.setdefault(spec.source_path, spec)
|
|
294
|
+
|
|
295
|
+
@property
|
|
296
|
+
def loaded_count(self) -> int:
|
|
297
|
+
return len(self._loaded)
|
|
298
|
+
|
|
299
|
+
@property
|
|
300
|
+
def retained_bytes(self) -> int:
|
|
301
|
+
return self._loaded_bytes
|
|
302
|
+
|
|
303
|
+
@property
|
|
304
|
+
def metadata_bytes(self) -> int:
|
|
305
|
+
return 512 + sum(
|
|
306
|
+
256
|
|
307
|
+
+ sys.getsizeof(path)
|
|
308
|
+
+ sys.getsizeof(spec)
|
|
309
|
+
+ sys.getsizeof(spec.display_path)
|
|
310
|
+
for path, spec in self._specs.items()
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
def clear_loaded(self) -> None:
|
|
314
|
+
self._loaded.clear()
|
|
315
|
+
self._loaded_bytes = 0
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
@dataclass(frozen=True, slots=True)
|
|
319
|
+
class _RawSymbol:
|
|
320
|
+
name: str
|
|
321
|
+
kind: SymbolKind
|
|
322
|
+
line: int
|
|
323
|
+
column: int
|
|
324
|
+
visibility: Visibility
|
|
325
|
+
type_name: str
|
|
326
|
+
source_path: Path
|
|
327
|
+
path: str
|
|
328
|
+
unit_id: str
|
|
329
|
+
unit_name: str
|
|
330
|
+
qualified_name: str
|
|
331
|
+
owner: str
|
|
332
|
+
parent_qualified_name: str
|
|
333
|
+
signature: str
|
|
334
|
+
context_ambiguous: bool = False
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
@dataclass(frozen=True, slots=True)
|
|
338
|
+
class _NavigationTask:
|
|
339
|
+
ordinal: int
|
|
340
|
+
source_path: str
|
|
341
|
+
display_path: str
|
|
342
|
+
unit_name: str
|
|
343
|
+
unit_path: str
|
|
344
|
+
unit_id: str
|
|
345
|
+
unit_has_error: bool
|
|
346
|
+
defines: tuple[str, ...]
|
|
347
|
+
include_paths: tuple[str, ...]
|
|
348
|
+
project_config: ProjectPathConfig | None = None
|
|
349
|
+
exact_name: str = ""
|
|
350
|
+
cache_key: str = ""
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
@dataclass(frozen=True, slots=True)
|
|
354
|
+
class _NavigationResult:
|
|
355
|
+
ordinal: int
|
|
356
|
+
source_path: str
|
|
357
|
+
text: str
|
|
358
|
+
model: None
|
|
359
|
+
lines_processed: int
|
|
360
|
+
symbols_discovered: int
|
|
361
|
+
read_error: str
|
|
362
|
+
raw_symbols: tuple[_RawSymbol, ...]
|
|
363
|
+
cache_key: str = ""
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
_THIN_UNIT_WRAPPER_RE = re.compile(
|
|
367
|
+
r"\A\s*unit\s+[\w.&]+\s*;\s*(?P<directives>(?:\{\$[^}]+\}\s*)+)\Z",
|
|
368
|
+
re.IGNORECASE,
|
|
369
|
+
)
|
|
370
|
+
_WRAPPER_DIRECTIVE_RE = re.compile(r"\{\$\s*(\w+)\s*([^}]*)\}", re.IGNORECASE)
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _thin_unit_include(
|
|
374
|
+
task: _NavigationTask,
|
|
375
|
+
text: str,
|
|
376
|
+
) -> tuple[Path, str, str, tuple[str, ...]] | None:
|
|
377
|
+
match = _THIN_UNIT_WRAPPER_RE.fullmatch(text)
|
|
378
|
+
if match is None:
|
|
379
|
+
return None
|
|
380
|
+
directives = _WRAPPER_DIRECTIVE_RE.findall(match.group("directives"))
|
|
381
|
+
includes = [
|
|
382
|
+
param.strip().strip("'\"")
|
|
383
|
+
for name, param in directives
|
|
384
|
+
if name.casefold() in {"i", "include"}
|
|
385
|
+
]
|
|
386
|
+
if len(includes) != 1 or not includes[0]:
|
|
387
|
+
return None
|
|
388
|
+
include_name = includes[0]
|
|
389
|
+
search_roots = (
|
|
390
|
+
Path(task.source_path).expanduser().resolve().parent,
|
|
391
|
+
*(Path(path).expanduser().resolve() for path in task.include_paths),
|
|
392
|
+
)
|
|
393
|
+
include_path = Path(include_name.replace("\\", "/"))
|
|
394
|
+
for root in search_roots:
|
|
395
|
+
candidate = (root / include_path).resolve()
|
|
396
|
+
if (
|
|
397
|
+
task.project_config is not None
|
|
398
|
+
and task.project_config.excludes_workspace_path(candidate)
|
|
399
|
+
):
|
|
400
|
+
continue
|
|
401
|
+
try:
|
|
402
|
+
include_text = read_source_text(candidate)
|
|
403
|
+
except (OSError, UnicodeError):
|
|
404
|
+
continue
|
|
405
|
+
define_directives = (
|
|
406
|
+
param.strip().split(None, 1)[0]
|
|
407
|
+
for name, param in directives
|
|
408
|
+
if name.casefold() == "define" and param.strip()
|
|
409
|
+
)
|
|
410
|
+
include_defines = tuple(dict.fromkeys((*task.defines, *define_directives)))
|
|
411
|
+
return (
|
|
412
|
+
candidate,
|
|
413
|
+
_related_display_path(task, candidate),
|
|
414
|
+
include_text,
|
|
415
|
+
include_defines,
|
|
416
|
+
)
|
|
417
|
+
return None
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _related_display_path(task: _NavigationTask, source_path: Path) -> str:
|
|
421
|
+
display_parts = Path(task.display_path).parts
|
|
422
|
+
root = Path(task.source_path).expanduser().resolve()
|
|
423
|
+
for _part in display_parts:
|
|
424
|
+
root = root.parent
|
|
425
|
+
try:
|
|
426
|
+
return source_path.relative_to(root).as_posix()
|
|
427
|
+
except ValueError:
|
|
428
|
+
return source_path.name
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _navigation_cache_text(
|
|
432
|
+
text: str,
|
|
433
|
+
wrapper: tuple[Path, str, str, tuple[str, ...]] | None,
|
|
434
|
+
) -> str:
|
|
435
|
+
if wrapper is None:
|
|
436
|
+
return text
|
|
437
|
+
_include_path, include_display_path, include_text, _include_defines = wrapper
|
|
438
|
+
return (
|
|
439
|
+
text
|
|
440
|
+
+ "\0thin-unit-include\0"
|
|
441
|
+
+ include_display_path
|
|
442
|
+
+ "\0"
|
|
443
|
+
+ include_text
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _parse_navigation_task(task: _NavigationTask) -> _NavigationResult:
|
|
448
|
+
source_path = Path(task.source_path)
|
|
449
|
+
try:
|
|
450
|
+
text = read_source_text(source_path)
|
|
451
|
+
except (OSError, UnicodeError) as error:
|
|
452
|
+
return _NavigationResult(
|
|
453
|
+
task.ordinal,
|
|
454
|
+
task.source_path,
|
|
455
|
+
"",
|
|
456
|
+
None,
|
|
457
|
+
0,
|
|
458
|
+
0,
|
|
459
|
+
str(error),
|
|
460
|
+
(),
|
|
461
|
+
"",
|
|
462
|
+
)
|
|
463
|
+
try:
|
|
464
|
+
wrapper = _thin_unit_include(task, text)
|
|
465
|
+
model = build_outline_semantic_model(
|
|
466
|
+
text,
|
|
467
|
+
task.source_path,
|
|
468
|
+
defines=task.defines,
|
|
469
|
+
_expand_thin_wrapper=wrapper is None,
|
|
470
|
+
)
|
|
471
|
+
document = _SourceDocument(
|
|
472
|
+
source_path,
|
|
473
|
+
task.display_path,
|
|
474
|
+
text,
|
|
475
|
+
defines=task.defines,
|
|
476
|
+
include_paths=task.include_paths,
|
|
477
|
+
project_config=task.project_config,
|
|
478
|
+
)
|
|
479
|
+
unit = AgentUnit(
|
|
480
|
+
unit_id=task.unit_id,
|
|
481
|
+
name=task.unit_name,
|
|
482
|
+
path=task.unit_path,
|
|
483
|
+
has_error=task.unit_has_error,
|
|
484
|
+
)
|
|
485
|
+
symbols = _collect_raw_symbols(model.unit_scope, unit, source_path, document)
|
|
486
|
+
raw_symbols = list(_exclude_routine_locals(symbols, document))
|
|
487
|
+
cache_text = text
|
|
488
|
+
if wrapper is not None:
|
|
489
|
+
include_path, include_display_path, include_text, include_defines = wrapper
|
|
490
|
+
include_model = build_outline_semantic_model(
|
|
491
|
+
include_text,
|
|
492
|
+
str(include_path),
|
|
493
|
+
defines=include_defines,
|
|
494
|
+
)
|
|
495
|
+
include_document = _SourceDocument(
|
|
496
|
+
include_path,
|
|
497
|
+
include_display_path,
|
|
498
|
+
include_text,
|
|
499
|
+
defines=include_defines,
|
|
500
|
+
include_paths=task.include_paths,
|
|
501
|
+
project_config=task.project_config,
|
|
502
|
+
)
|
|
503
|
+
included = _collect_raw_symbols(
|
|
504
|
+
include_model.unit_scope,
|
|
505
|
+
unit,
|
|
506
|
+
include_path,
|
|
507
|
+
include_document,
|
|
508
|
+
)
|
|
509
|
+
raw_symbols.extend(
|
|
510
|
+
replace(
|
|
511
|
+
_raw_symbol_with_unit_name(raw, task.unit_name),
|
|
512
|
+
unit_id=task.unit_id,
|
|
513
|
+
)
|
|
514
|
+
for raw in _exclude_routine_locals(included, include_document)
|
|
515
|
+
if raw.kind != SymbolKind.UNIT
|
|
516
|
+
)
|
|
517
|
+
cache_text = _navigation_cache_text(text, wrapper)
|
|
518
|
+
if task.exact_name and not any(
|
|
519
|
+
_normalized(symbol.name) == task.exact_name
|
|
520
|
+
for symbol in raw_symbols
|
|
521
|
+
):
|
|
522
|
+
return _NavigationResult(
|
|
523
|
+
task.ordinal,
|
|
524
|
+
task.source_path,
|
|
525
|
+
"",
|
|
526
|
+
None,
|
|
527
|
+
text.count("\n")
|
|
528
|
+
+ (0 if not text or text.endswith(("\n", "\r")) else 1),
|
|
529
|
+
0,
|
|
530
|
+
"",
|
|
531
|
+
(),
|
|
532
|
+
"",
|
|
533
|
+
)
|
|
534
|
+
except Exception as error:
|
|
535
|
+
raise ParallelOutlineError(
|
|
536
|
+
f"failed to build navigation shard for {task.source_path}: {error}"
|
|
537
|
+
) from error
|
|
538
|
+
return _NavigationResult(
|
|
539
|
+
task.ordinal,
|
|
540
|
+
task.source_path,
|
|
541
|
+
"",
|
|
542
|
+
None,
|
|
543
|
+
text.count("\n") + (0 if not text or text.endswith(("\n", "\r")) else 1),
|
|
544
|
+
len(raw_symbols),
|
|
545
|
+
"",
|
|
546
|
+
tuple(raw_symbols),
|
|
547
|
+
navigation_cache_key(cache_text, task.defines) if task.cache_key else "",
|
|
548
|
+
)
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def _navigation_shard_payload(result: _NavigationResult) -> dict[str, object]:
|
|
552
|
+
unit_name = result.raw_symbols[0].unit_name if result.raw_symbols else ""
|
|
553
|
+
symbols: list[dict[str, object]] = []
|
|
554
|
+
for raw in result.raw_symbols:
|
|
555
|
+
record: dict[str, object] = {
|
|
556
|
+
"name": raw.name,
|
|
557
|
+
"kind": raw.kind.value,
|
|
558
|
+
"source_path": str(raw.source_path),
|
|
559
|
+
"path": raw.path,
|
|
560
|
+
"line": raw.line,
|
|
561
|
+
"column": raw.column,
|
|
562
|
+
"visibility": raw.visibility.value,
|
|
563
|
+
"type": raw.type_name,
|
|
564
|
+
"qualified_name": raw.qualified_name,
|
|
565
|
+
"owner": raw.owner,
|
|
566
|
+
"parent_qualified_name": raw.parent_qualified_name,
|
|
567
|
+
"signature": raw.signature,
|
|
568
|
+
}
|
|
569
|
+
if raw.context_ambiguous:
|
|
570
|
+
record["context_ambiguous"] = True
|
|
571
|
+
symbols.append(record)
|
|
572
|
+
return {
|
|
573
|
+
"lines_processed": result.lines_processed,
|
|
574
|
+
"unit_name": unit_name,
|
|
575
|
+
"symbols": symbols,
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def _navigation_result_from_shard(
|
|
580
|
+
task: _NavigationTask,
|
|
581
|
+
payload: Mapping[str, object],
|
|
582
|
+
) -> _NavigationResult | None:
|
|
583
|
+
try:
|
|
584
|
+
lines_processed = _required_int(payload.get("lines_processed"))
|
|
585
|
+
unit_name = _required_string(payload.get("unit_name"))
|
|
586
|
+
records = payload.get("symbols")
|
|
587
|
+
if not isinstance(records, list):
|
|
588
|
+
return None
|
|
589
|
+
unit_id = make_target_id("unit", task.display_path, task.unit_name)
|
|
590
|
+
raw_symbols: list[_RawSymbol] = []
|
|
591
|
+
for value in records:
|
|
592
|
+
if not isinstance(value, Mapping):
|
|
593
|
+
return None
|
|
594
|
+
raw_symbols.append(
|
|
595
|
+
_RawSymbol(
|
|
596
|
+
name=_required_string(value.get("name")),
|
|
597
|
+
kind=SymbolKind(_required_string(value.get("kind"))),
|
|
598
|
+
line=_required_int(value.get("line")),
|
|
599
|
+
column=_required_int(value.get("column")),
|
|
600
|
+
visibility=Visibility(_required_string(value.get("visibility"))),
|
|
601
|
+
type_name=_required_string(value.get("type")),
|
|
602
|
+
source_path=Path(
|
|
603
|
+
_required_string(value.get("source_path"))
|
|
604
|
+
if value.get("source_path") is not None
|
|
605
|
+
else task.source_path
|
|
606
|
+
),
|
|
607
|
+
path=(
|
|
608
|
+
_required_string(value.get("path"))
|
|
609
|
+
if value.get("path") is not None
|
|
610
|
+
else task.display_path
|
|
611
|
+
),
|
|
612
|
+
unit_id=unit_id,
|
|
613
|
+
unit_name=unit_name or task.unit_name,
|
|
614
|
+
qualified_name=_required_string(value.get("qualified_name")),
|
|
615
|
+
owner=_required_string(value.get("owner")),
|
|
616
|
+
parent_qualified_name=_required_string(
|
|
617
|
+
value.get("parent_qualified_name")
|
|
618
|
+
),
|
|
619
|
+
signature=_required_string(value.get("signature")),
|
|
620
|
+
context_ambiguous=value.get("context_ambiguous") is True,
|
|
621
|
+
)
|
|
622
|
+
)
|
|
623
|
+
except (TypeError, ValueError):
|
|
624
|
+
return None
|
|
625
|
+
return _NavigationResult(
|
|
626
|
+
ordinal=task.ordinal,
|
|
627
|
+
source_path=task.source_path,
|
|
628
|
+
text="",
|
|
629
|
+
model=None,
|
|
630
|
+
lines_processed=lines_processed,
|
|
631
|
+
symbols_discovered=len(raw_symbols),
|
|
632
|
+
read_error="",
|
|
633
|
+
raw_symbols=tuple(raw_symbols),
|
|
634
|
+
)
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
def _required_string(value: object) -> str:
|
|
638
|
+
if not isinstance(value, str):
|
|
639
|
+
raise TypeError("navigation string is malformed")
|
|
640
|
+
return value
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def _required_int(value: object) -> int:
|
|
644
|
+
if isinstance(value, bool) or not isinstance(value, int):
|
|
645
|
+
raise TypeError("navigation integer is malformed")
|
|
646
|
+
return value
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
@dataclass(frozen=True, slots=True)
|
|
650
|
+
class _SymbolEntry:
|
|
651
|
+
name: str
|
|
652
|
+
kind: SymbolKind
|
|
653
|
+
line: int
|
|
654
|
+
column: int
|
|
655
|
+
visibility: Visibility
|
|
656
|
+
type_name: str
|
|
657
|
+
source_path: Path
|
|
658
|
+
path: str
|
|
659
|
+
unit_id: str
|
|
660
|
+
unit_name: str
|
|
661
|
+
qualified_name: str
|
|
662
|
+
normalized_name: str
|
|
663
|
+
normalized_qualified_name: str
|
|
664
|
+
relative_name_offset: int
|
|
665
|
+
owner: str
|
|
666
|
+
signature: str
|
|
667
|
+
ordinal: int
|
|
668
|
+
target_id: str
|
|
669
|
+
card_json_chars: int
|
|
670
|
+
card_json_upper_bound: int
|
|
671
|
+
parent_target_id: str = ""
|
|
672
|
+
context_ambiguous: bool = False
|
|
673
|
+
|
|
674
|
+
def card(self) -> dict[str, object]:
|
|
675
|
+
card: dict[str, object] = {
|
|
676
|
+
"target_id": self.target_id,
|
|
677
|
+
"unit_id": self.unit_id,
|
|
678
|
+
"name": self.name,
|
|
679
|
+
"qualified_name": self.qualified_name,
|
|
680
|
+
"kind": self.kind.value,
|
|
681
|
+
"path": self.path,
|
|
682
|
+
"line": self.line,
|
|
683
|
+
"column": self.column,
|
|
684
|
+
"visibility": self.visibility.value,
|
|
685
|
+
"owner": self.owner,
|
|
686
|
+
"type": self.type_name,
|
|
687
|
+
}
|
|
688
|
+
if self.context_ambiguous:
|
|
689
|
+
card["context_ambiguous"] = True
|
|
690
|
+
return card
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
class _SymbolCardSequence(Sequence[dict[str, object]]):
|
|
694
|
+
def __init__(
|
|
695
|
+
self,
|
|
696
|
+
entries: Sequence[_SymbolEntry],
|
|
697
|
+
max_chars: int,
|
|
698
|
+
) -> None:
|
|
699
|
+
self._entries = entries
|
|
700
|
+
self._max_chars = max_chars
|
|
701
|
+
self._chunk_payload_chars = _card_chunk_payload_chars(max_chars)
|
|
702
|
+
self._ends: list[int] = []
|
|
703
|
+
total = 0
|
|
704
|
+
for entry in entries:
|
|
705
|
+
if (
|
|
706
|
+
entry.card_json_upper_bound + 2 <= max_chars
|
|
707
|
+
or entry.card_json_chars + 2 <= max_chars
|
|
708
|
+
):
|
|
709
|
+
count = 1
|
|
710
|
+
else:
|
|
711
|
+
count = (
|
|
712
|
+
entry.card_json_chars + self._chunk_payload_chars - 1
|
|
713
|
+
) // self._chunk_payload_chars
|
|
714
|
+
total += count
|
|
715
|
+
self._ends.append(total)
|
|
716
|
+
|
|
717
|
+
def __len__(self) -> int:
|
|
718
|
+
return self._ends[-1] if self._ends else 0
|
|
719
|
+
|
|
720
|
+
def __getitem__(
|
|
721
|
+
self,
|
|
722
|
+
index: int | slice,
|
|
723
|
+
) -> dict[str, object] | list[dict[str, object]]:
|
|
724
|
+
if isinstance(index, slice):
|
|
725
|
+
return [self[item_index] for item_index in range(*index.indices(len(self)))]
|
|
726
|
+
if index < 0:
|
|
727
|
+
index += len(self)
|
|
728
|
+
if index < 0 or index >= len(self):
|
|
729
|
+
raise IndexError(index)
|
|
730
|
+
entry_index = bisect_right(self._ends, index)
|
|
731
|
+
entry = self._entries[entry_index]
|
|
732
|
+
previous_end = self._ends[entry_index - 1] if entry_index else 0
|
|
733
|
+
chunk_index = index - previous_end
|
|
734
|
+
if (
|
|
735
|
+
entry.card_json_upper_bound + 2 <= self._max_chars
|
|
736
|
+
or entry.card_json_chars + 2 <= self._max_chars
|
|
737
|
+
):
|
|
738
|
+
return entry.card()
|
|
739
|
+
serialized = _compact_json(entry.card())
|
|
740
|
+
chunk_count = self._ends[entry_index] - previous_end
|
|
741
|
+
start = chunk_index * self._chunk_payload_chars
|
|
742
|
+
return {
|
|
743
|
+
"item_type": "card_chunk",
|
|
744
|
+
"chunk_index": chunk_index,
|
|
745
|
+
"chunk_count": chunk_count,
|
|
746
|
+
"json": serialized[start:start + self._chunk_payload_chars],
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
def _card_chunk_payload_chars(max_chars: int) -> int:
|
|
751
|
+
wrapper_chars = len(
|
|
752
|
+
_compact_json(
|
|
753
|
+
{
|
|
754
|
+
"item_type": "card_chunk",
|
|
755
|
+
"chunk_index": 999999,
|
|
756
|
+
"chunk_count": 999999,
|
|
757
|
+
"json": "",
|
|
758
|
+
}
|
|
759
|
+
)
|
|
760
|
+
) + 2
|
|
761
|
+
payload_chars = (max_chars - wrapper_chars) // 2
|
|
762
|
+
if payload_chars < 1:
|
|
763
|
+
raise AgentProtocolError(
|
|
764
|
+
"item_too_large",
|
|
765
|
+
"max_chars is too small for a structured response chunk.",
|
|
766
|
+
)
|
|
767
|
+
return payload_chars
|
|
768
|
+
|
|
769
|
+
|
|
770
|
+
@dataclass(slots=True)
|
|
771
|
+
class _Registry:
|
|
772
|
+
project_id: str
|
|
773
|
+
revision: str
|
|
774
|
+
entries: tuple[_SymbolEntry, ...]
|
|
775
|
+
by_target: dict[str, _SymbolEntry]
|
|
776
|
+
sources: _SourceStore
|
|
777
|
+
ranked_queries: OrderedDict[str, tuple[_SymbolEntry, ...]]
|
|
778
|
+
static_retained_bytes: int
|
|
779
|
+
indexed_include_paths: set[Path] = field(default_factory=set)
|
|
780
|
+
include_queries: set[str] = field(default_factory=set)
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
class _CpgCandidates(Mapping[str, tuple[CpgTarget, ...]]):
|
|
784
|
+
def __init__(self, entries: Sequence[_SymbolEntry]) -> None:
|
|
785
|
+
self._entries = entries
|
|
786
|
+
self._resolved: dict[str, tuple[CpgTarget, ...]] = {}
|
|
787
|
+
|
|
788
|
+
def __getitem__(self, name: str) -> tuple[CpgTarget, ...]:
|
|
789
|
+
key = _normalized(name)
|
|
790
|
+
cached = self._resolved.get(key)
|
|
791
|
+
if cached is not None:
|
|
792
|
+
return cached
|
|
793
|
+
resolved = tuple(
|
|
794
|
+
_cpg_target(entry)
|
|
795
|
+
for entry in self._entries
|
|
796
|
+
if entry.kind in _ROUTINE_KINDS
|
|
797
|
+
and (
|
|
798
|
+
entry.normalized_name == key
|
|
799
|
+
or entry.normalized_qualified_name == key
|
|
800
|
+
)
|
|
801
|
+
)
|
|
802
|
+
self._resolved[key] = resolved
|
|
803
|
+
return resolved
|
|
804
|
+
|
|
805
|
+
def __iter__(self):
|
|
806
|
+
return iter(self._resolved)
|
|
807
|
+
|
|
808
|
+
def __len__(self) -> int:
|
|
809
|
+
return len(self._resolved)
|
|
810
|
+
|
|
811
|
+
def resolve_many(
|
|
812
|
+
self,
|
|
813
|
+
names: set[str],
|
|
814
|
+
) -> dict[str, tuple[CpgTarget, ...]]:
|
|
815
|
+
keys = {_normalized(name) for name in names}
|
|
816
|
+
missing = keys.difference(self._resolved)
|
|
817
|
+
if missing:
|
|
818
|
+
collected: dict[str, list[CpgTarget]] = {
|
|
819
|
+
name: [] for name in missing
|
|
820
|
+
}
|
|
821
|
+
for entry in self._entries:
|
|
822
|
+
if entry.kind not in _ROUTINE_KINDS:
|
|
823
|
+
continue
|
|
824
|
+
name_match = entry.normalized_name in missing
|
|
825
|
+
qualified_match = (
|
|
826
|
+
entry.normalized_qualified_name in missing
|
|
827
|
+
and entry.normalized_qualified_name
|
|
828
|
+
!= entry.normalized_name
|
|
829
|
+
)
|
|
830
|
+
if not name_match and not qualified_match:
|
|
831
|
+
continue
|
|
832
|
+
target = _cpg_target(entry)
|
|
833
|
+
if name_match:
|
|
834
|
+
collected[entry.normalized_name].append(target)
|
|
835
|
+
if qualified_match:
|
|
836
|
+
collected[entry.normalized_qualified_name].append(target)
|
|
837
|
+
self._resolved.update(
|
|
838
|
+
(name, tuple(matches))
|
|
839
|
+
for name, matches in collected.items()
|
|
840
|
+
)
|
|
841
|
+
return {
|
|
842
|
+
name: self._resolved.get(name, ())
|
|
843
|
+
for name in keys
|
|
844
|
+
}
|
|
845
|
+
|
|
846
|
+
|
|
847
|
+
def _cpg_target(entry: _SymbolEntry) -> CpgTarget:
|
|
848
|
+
return CpgTarget(
|
|
849
|
+
target_id=entry.target_id,
|
|
850
|
+
source_path=str(entry.source_path),
|
|
851
|
+
path=entry.path,
|
|
852
|
+
unit_id=entry.unit_id,
|
|
853
|
+
name=entry.name,
|
|
854
|
+
qualified_name=entry.qualified_name,
|
|
855
|
+
kind=entry.kind.value,
|
|
856
|
+
line=entry.line,
|
|
857
|
+
column=entry.column,
|
|
858
|
+
visibility=entry.visibility.value,
|
|
859
|
+
type_name=entry.type_name,
|
|
860
|
+
owner=entry.owner,
|
|
861
|
+
parent_target_id=entry.parent_target_id,
|
|
862
|
+
)
|
|
863
|
+
|
|
864
|
+
|
|
865
|
+
def _relation_target(entry: _SymbolEntry) -> RelationTarget:
|
|
866
|
+
return RelationTarget(
|
|
867
|
+
target_id=entry.target_id,
|
|
868
|
+
source_path=str(entry.source_path),
|
|
869
|
+
path=entry.path,
|
|
870
|
+
unit_id=entry.unit_id,
|
|
871
|
+
unit_name=entry.unit_name,
|
|
872
|
+
name=entry.name,
|
|
873
|
+
qualified_name=entry.qualified_name,
|
|
874
|
+
kind=entry.kind.value,
|
|
875
|
+
signature=entry.signature,
|
|
876
|
+
line=entry.line,
|
|
877
|
+
column=entry.column,
|
|
878
|
+
card={},
|
|
879
|
+
visibility=entry.visibility.value,
|
|
880
|
+
type_name=entry.type_name,
|
|
881
|
+
owner=entry.owner,
|
|
882
|
+
)
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
class AgentContext:
|
|
886
|
+
def __init__(
|
|
887
|
+
self,
|
|
888
|
+
workspace: AgentWorkspace,
|
|
889
|
+
*,
|
|
890
|
+
workers: int = 0,
|
|
891
|
+
worker_memory_budget_bytes: int | None = None,
|
|
892
|
+
revision_check_interval_seconds: float = 0.0,
|
|
893
|
+
navigation_cache_dir: str | Path | None = None,
|
|
894
|
+
navigation_cache_max_bytes: int = _DEFAULT_NAVIGATION_CACHE_MAX_BYTES,
|
|
895
|
+
) -> None:
|
|
896
|
+
self._workspace = workspace
|
|
897
|
+
self._workers = workers
|
|
898
|
+
self._worker_memory_budget_bytes = worker_memory_budget_bytes
|
|
899
|
+
self._revision_check_interval_seconds = max(
|
|
900
|
+
0.0,
|
|
901
|
+
revision_check_interval_seconds,
|
|
902
|
+
)
|
|
903
|
+
self._last_revision_check_at = (
|
|
904
|
+
time.monotonic() if self._revision_check_interval_seconds > 0.0 else 0.0
|
|
905
|
+
)
|
|
906
|
+
self._revision_epoch = 0
|
|
907
|
+
self._parallel_stats = ParallelBuildStats(0, 0, 0, 0.0, 0)
|
|
908
|
+
self._navigation_store = (
|
|
909
|
+
NavigationShardStore(navigation_cache_dir)
|
|
910
|
+
if navigation_cache_dir is not None
|
|
911
|
+
else None
|
|
912
|
+
)
|
|
913
|
+
self._navigation_cache_max_bytes = max(0, navigation_cache_max_bytes)
|
|
914
|
+
self._navigation_disk_hits = 0
|
|
915
|
+
self._navigation_disk_misses = 0
|
|
916
|
+
project_id = workspace.active_project_id
|
|
917
|
+
self._focus = Focus(project_id=project_id) if project_id else Focus()
|
|
918
|
+
self._last_revision = workspace.current_revision
|
|
919
|
+
self._registry: _Registry | None = None
|
|
920
|
+
self._relation_index: ProjectRelationIndex | None = None
|
|
921
|
+
self._metrics: ProjectMetrics | None = None
|
|
922
|
+
self._metrics_revision = ""
|
|
923
|
+
cpg_budget_base = (
|
|
924
|
+
worker_memory_budget_bytes
|
|
925
|
+
if worker_memory_budget_bytes is not None
|
|
926
|
+
else 512 * 1024**2
|
|
927
|
+
)
|
|
928
|
+
self._cpg_cache_limit = min(
|
|
929
|
+
128 * 1024**2,
|
|
930
|
+
max(1, cpg_budget_base // 5),
|
|
931
|
+
)
|
|
932
|
+
self._cpg_cache: OrderedDict[
|
|
933
|
+
tuple[str, str, str, str, int],
|
|
934
|
+
CpgSubgraph,
|
|
935
|
+
] = OrderedDict()
|
|
936
|
+
self._cpg_cache_bytes = 0
|
|
937
|
+
self._cpg_sources_parsed = 0
|
|
938
|
+
self._prepared_response_cache: OrderedDict[
|
|
939
|
+
tuple[str, str],
|
|
940
|
+
Sequence[dict[str, object]],
|
|
941
|
+
] = OrderedDict()
|
|
942
|
+
|
|
943
|
+
@classmethod
|
|
944
|
+
def open(
|
|
945
|
+
cls,
|
|
946
|
+
root: str | Path,
|
|
947
|
+
project_file: str | Path | None = None,
|
|
948
|
+
*,
|
|
949
|
+
workers: int = 0,
|
|
950
|
+
worker_memory_budget_bytes: int | None = None,
|
|
951
|
+
revision_check_interval_seconds: float = 0.0,
|
|
952
|
+
reference_query: str = "",
|
|
953
|
+
reference_path: str = "",
|
|
954
|
+
navigation_cache_dir: str | Path | None = None,
|
|
955
|
+
navigation_cache_max_bytes: int = _DEFAULT_NAVIGATION_CACHE_MAX_BYTES,
|
|
956
|
+
) -> AgentContext:
|
|
957
|
+
context = cls(
|
|
958
|
+
AgentWorkspace.open(root, project_file=project_file),
|
|
959
|
+
workers=workers,
|
|
960
|
+
worker_memory_budget_bytes=worker_memory_budget_bytes,
|
|
961
|
+
revision_check_interval_seconds=revision_check_interval_seconds,
|
|
962
|
+
navigation_cache_dir=navigation_cache_dir,
|
|
963
|
+
navigation_cache_max_bytes=navigation_cache_max_bytes,
|
|
964
|
+
)
|
|
965
|
+
if reference_query and len(context.workspace.units) >= 256:
|
|
966
|
+
context._prepare_reference_registry(
|
|
967
|
+
reference_query,
|
|
968
|
+
source_path=reference_path,
|
|
969
|
+
)
|
|
970
|
+
return context
|
|
971
|
+
|
|
972
|
+
@property
|
|
973
|
+
def workspace(self) -> AgentWorkspace:
|
|
974
|
+
return self._workspace
|
|
975
|
+
|
|
976
|
+
@property
|
|
977
|
+
def navigation_cache_is_warm(self) -> bool:
|
|
978
|
+
return self._registry is not None
|
|
979
|
+
|
|
980
|
+
@property
|
|
981
|
+
def parallel_stats(self) -> ParallelBuildStats:
|
|
982
|
+
return self._parallel_stats
|
|
983
|
+
|
|
984
|
+
@property
|
|
985
|
+
def navigation_disk_hits(self) -> int:
|
|
986
|
+
return self._navigation_disk_hits
|
|
987
|
+
|
|
988
|
+
@property
|
|
989
|
+
def navigation_disk_misses(self) -> int:
|
|
990
|
+
return self._navigation_disk_misses
|
|
991
|
+
|
|
992
|
+
@property
|
|
993
|
+
def cpg_cache_entries(self) -> int:
|
|
994
|
+
return len(self._cpg_cache)
|
|
995
|
+
|
|
996
|
+
@property
|
|
997
|
+
def cpg_cache_bytes(self) -> int:
|
|
998
|
+
return self._cpg_cache_bytes
|
|
999
|
+
|
|
1000
|
+
@property
|
|
1001
|
+
def cpg_sources_parsed(self) -> int:
|
|
1002
|
+
return self._cpg_sources_parsed
|
|
1003
|
+
|
|
1004
|
+
def cache_roots(self) -> tuple[object, ...]:
|
|
1005
|
+
return (
|
|
1006
|
+
*self._workspace.cache_roots(),
|
|
1007
|
+
self._registry,
|
|
1008
|
+
self._relation_index,
|
|
1009
|
+
self._metrics,
|
|
1010
|
+
tuple(self._cpg_cache.values()),
|
|
1011
|
+
tuple(self._prepared_response_cache.values()),
|
|
1012
|
+
)
|
|
1013
|
+
|
|
1014
|
+
@property
|
|
1015
|
+
def estimated_cache_bytes(self) -> int:
|
|
1016
|
+
retained = self._workspace.estimated_cache_bytes
|
|
1017
|
+
if self._registry is not None:
|
|
1018
|
+
retained += (
|
|
1019
|
+
self._registry.static_retained_bytes
|
|
1020
|
+
+ self._registry.sources.retained_bytes
|
|
1021
|
+
+ sum(
|
|
1022
|
+
sys.getsizeof(query)
|
|
1023
|
+
+ sys.getsizeof(ranked)
|
|
1024
|
+
+ len(ranked) * 8
|
|
1025
|
+
for query, ranked in self._registry.ranked_queries.items()
|
|
1026
|
+
)
|
|
1027
|
+
)
|
|
1028
|
+
if self._relation_index is not None:
|
|
1029
|
+
retained += self._relation_index.estimated_cache_bytes
|
|
1030
|
+
if self._metrics is not None:
|
|
1031
|
+
retained += 4096 + len(self._metrics.file_metrics) * 1024
|
|
1032
|
+
retained += self._cpg_cache_bytes
|
|
1033
|
+
retained += sum(
|
|
1034
|
+
sys.getsizeof(key)
|
|
1035
|
+
+ sys.getsizeof(items)
|
|
1036
|
+
+ len(items) * 8
|
|
1037
|
+
for key, items in self._prepared_response_cache.items()
|
|
1038
|
+
)
|
|
1039
|
+
return retained
|
|
1040
|
+
|
|
1041
|
+
def evict_auxiliary_caches(self) -> None:
|
|
1042
|
+
self._relation_index = None
|
|
1043
|
+
self._metrics = None
|
|
1044
|
+
self._metrics_revision = ""
|
|
1045
|
+
self._clear_cpg_cache()
|
|
1046
|
+
self._prepared_response_cache.clear()
|
|
1047
|
+
if self._registry is not None:
|
|
1048
|
+
self._registry.sources.clear_loaded()
|
|
1049
|
+
|
|
1050
|
+
def evict_navigation_caches(self) -> None:
|
|
1051
|
+
self.evict_auxiliary_caches()
|
|
1052
|
+
self._registry = None
|
|
1053
|
+
self._workspace.evict_recomputable_caches()
|
|
1054
|
+
|
|
1055
|
+
def prewarm_navigation(self) -> str:
|
|
1056
|
+
revision_epoch = self._revision_epoch
|
|
1057
|
+
revision = self._refresh_workspace("")
|
|
1058
|
+
self._require_registry(revision)
|
|
1059
|
+
if self._revision_epoch == revision_epoch:
|
|
1060
|
+
self._last_revision_check_at = time.monotonic()
|
|
1061
|
+
return revision
|
|
1062
|
+
|
|
1063
|
+
def _prepare_reference_registry(
|
|
1064
|
+
self,
|
|
1065
|
+
query: str,
|
|
1066
|
+
*,
|
|
1067
|
+
source_path: str = "",
|
|
1068
|
+
) -> None:
|
|
1069
|
+
project_id = self._require_selected_project()
|
|
1070
|
+
(
|
|
1071
|
+
self._registry,
|
|
1072
|
+
self._parallel_stats,
|
|
1073
|
+
self._navigation_disk_hits,
|
|
1074
|
+
self._navigation_disk_misses,
|
|
1075
|
+
) = _build_registry(
|
|
1076
|
+
self._workspace,
|
|
1077
|
+
project_id,
|
|
1078
|
+
self._last_revision,
|
|
1079
|
+
workers=self._workers,
|
|
1080
|
+
worker_memory_budget_bytes=self._worker_memory_budget_bytes,
|
|
1081
|
+
navigation_store=self._navigation_store,
|
|
1082
|
+
navigation_cache_max_bytes=self._navigation_cache_max_bytes,
|
|
1083
|
+
exact_query=query,
|
|
1084
|
+
exact_path=source_path,
|
|
1085
|
+
)
|
|
1086
|
+
|
|
1087
|
+
def invalidate_revision_cache(self) -> None:
|
|
1088
|
+
self._revision_epoch += 1
|
|
1089
|
+
self._last_revision_check_at = float("-inf")
|
|
1090
|
+
|
|
1091
|
+
def handle(
|
|
1092
|
+
self,
|
|
1093
|
+
request: AgentRequest | Mapping[str, object],
|
|
1094
|
+
*,
|
|
1095
|
+
cancel_check: Callable[[], None] | None = None,
|
|
1096
|
+
) -> AgentResponse:
|
|
1097
|
+
parsed = _validated_request(request)
|
|
1098
|
+
if cancel_check is not None:
|
|
1099
|
+
cancel_check()
|
|
1100
|
+
revision = self._refresh_workspace(parsed.project_id)
|
|
1101
|
+
|
|
1102
|
+
if parsed.action == "cpg":
|
|
1103
|
+
registry = self._require_registry(revision)
|
|
1104
|
+
entry = self._resolve_target(registry, parsed.target_id)
|
|
1105
|
+
graph = self._require_cpg_subgraph(registry, entry, parsed)
|
|
1106
|
+
return self._response(
|
|
1107
|
+
parsed,
|
|
1108
|
+
revision,
|
|
1109
|
+
graph.to_items(),
|
|
1110
|
+
target_id=entry.target_id,
|
|
1111
|
+
)
|
|
1112
|
+
if parsed.action == "trace":
|
|
1113
|
+
if cancel_check is not None:
|
|
1114
|
+
cancel_check()
|
|
1115
|
+
if parsed.relation is None:
|
|
1116
|
+
raise AgentProtocolError("relation_required", "Trace requires a relation.")
|
|
1117
|
+
targeted_references = (
|
|
1118
|
+
parsed.relation == "references"
|
|
1119
|
+
and len(self._workspace.units) >= 256
|
|
1120
|
+
)
|
|
1121
|
+
relation_index = (
|
|
1122
|
+
self._relation_index
|
|
1123
|
+
if (
|
|
1124
|
+
targeted_references
|
|
1125
|
+
and self._relation_index is not None
|
|
1126
|
+
and self._relation_index.project_id
|
|
1127
|
+
== self._workspace.active_project_id
|
|
1128
|
+
and self._relation_index.revision == revision
|
|
1129
|
+
and self._relation_index.has_target(parsed.target_id)
|
|
1130
|
+
)
|
|
1131
|
+
else None
|
|
1132
|
+
)
|
|
1133
|
+
resolved_target_id = parsed.target_id
|
|
1134
|
+
if relation_index is None:
|
|
1135
|
+
registry = self._require_registry(revision)
|
|
1136
|
+
entry = self._resolve_target(registry, parsed.target_id)
|
|
1137
|
+
resolved_target_id = entry.target_id
|
|
1138
|
+
if targeted_references:
|
|
1139
|
+
relation_index = self._require_reference_relation_index(
|
|
1140
|
+
registry,
|
|
1141
|
+
entry,
|
|
1142
|
+
)
|
|
1143
|
+
registry.sources.clear_loaded()
|
|
1144
|
+
self._registry = None
|
|
1145
|
+
del registry
|
|
1146
|
+
gc.collect()
|
|
1147
|
+
else:
|
|
1148
|
+
relation_index = self._require_relation_index(
|
|
1149
|
+
registry,
|
|
1150
|
+
parsed.relation,
|
|
1151
|
+
)
|
|
1152
|
+
items = relation_index.trace(
|
|
1153
|
+
resolved_target_id,
|
|
1154
|
+
parsed.relation,
|
|
1155
|
+
cancel_check=cancel_check,
|
|
1156
|
+
max_relations=parsed.max_items,
|
|
1157
|
+
)
|
|
1158
|
+
return self._response(
|
|
1159
|
+
parsed,
|
|
1160
|
+
revision,
|
|
1161
|
+
items,
|
|
1162
|
+
target_id=resolved_target_id,
|
|
1163
|
+
)
|
|
1164
|
+
if parsed.action == "open":
|
|
1165
|
+
items = self._open_items()
|
|
1166
|
+
return self._response(parsed, revision, items)
|
|
1167
|
+
if parsed.action == "problems":
|
|
1168
|
+
self._require_selected_project()
|
|
1169
|
+
items = self._problem_items()
|
|
1170
|
+
return self._response(parsed, revision, items)
|
|
1171
|
+
if parsed.action == "metrics":
|
|
1172
|
+
return self._handle_metrics(
|
|
1173
|
+
parsed,
|
|
1174
|
+
revision,
|
|
1175
|
+
cancel_check=cancel_check,
|
|
1176
|
+
)
|
|
1177
|
+
if parsed.action == "focus":
|
|
1178
|
+
return self._handle_focus(parsed, revision)
|
|
1179
|
+
if parsed.action == "find":
|
|
1180
|
+
registry = self._require_registry(revision)
|
|
1181
|
+
if parsed.query.strip() and not _has_exact_query_match(
|
|
1182
|
+
registry.entries,
|
|
1183
|
+
parsed.query,
|
|
1184
|
+
):
|
|
1185
|
+
registry_augmented = _augment_registry_with_include_matches(
|
|
1186
|
+
self._workspace,
|
|
1187
|
+
registry,
|
|
1188
|
+
parsed.query,
|
|
1189
|
+
)
|
|
1190
|
+
if registry_augmented:
|
|
1191
|
+
self._prepared_response_cache.clear()
|
|
1192
|
+
ranked = registry.ranked_queries.pop(parsed.query, None)
|
|
1193
|
+
if ranked is None:
|
|
1194
|
+
ranked = (
|
|
1195
|
+
registry.entries
|
|
1196
|
+
if not parsed.query.strip()
|
|
1197
|
+
else tuple(_ranked_entries(registry.entries, parsed.query))
|
|
1198
|
+
)
|
|
1199
|
+
if len(ranked) <= _RANKED_QUERY_CACHE_MAX_ENTRIES:
|
|
1200
|
+
registry.ranked_queries[parsed.query] = ranked
|
|
1201
|
+
while len(registry.ranked_queries) > _RANKED_QUERY_CACHE_SIZE:
|
|
1202
|
+
registry.ranked_queries.popitem(last=False)
|
|
1203
|
+
return self._response(
|
|
1204
|
+
parsed,
|
|
1205
|
+
revision,
|
|
1206
|
+
_SymbolCardSequence(ranked, parsed.max_chars),
|
|
1207
|
+
items_prepared=True,
|
|
1208
|
+
)
|
|
1209
|
+
if parsed.action == "inspect":
|
|
1210
|
+
registry = self._require_registry(revision)
|
|
1211
|
+
entry = self._resolve_target(registry, parsed.target_id)
|
|
1212
|
+
items = self._inspect_items(registry, entry, parsed)
|
|
1213
|
+
return self._response(parsed, revision, items, target_id=entry.target_id)
|
|
1214
|
+
raise AgentProtocolError("invalid_action", f"Unsupported action value: {parsed.action!r}.")
|
|
1215
|
+
|
|
1216
|
+
def _refresh_workspace(self, requested_project_id: str) -> str:
|
|
1217
|
+
previous_project_id = self._workspace.active_project_id
|
|
1218
|
+
selected_project_id = requested_project_id or previous_project_id
|
|
1219
|
+
now = time.monotonic()
|
|
1220
|
+
revision_epoch = self._revision_epoch
|
|
1221
|
+
selection_changed = bool(
|
|
1222
|
+
requested_project_id
|
|
1223
|
+
and requested_project_id != previous_project_id
|
|
1224
|
+
)
|
|
1225
|
+
revision_is_fresh = (
|
|
1226
|
+
not selection_changed
|
|
1227
|
+
and self._revision_check_interval_seconds > 0.0
|
|
1228
|
+
and now - self._last_revision_check_at
|
|
1229
|
+
< self._revision_check_interval_seconds
|
|
1230
|
+
)
|
|
1231
|
+
if revision_is_fresh:
|
|
1232
|
+
revision = self._last_revision
|
|
1233
|
+
elif selected_project_id:
|
|
1234
|
+
revision = self._workspace._select_project_with_revision(selected_project_id)
|
|
1235
|
+
else:
|
|
1236
|
+
revision = self._workspace.workspace_revision
|
|
1237
|
+
if not revision_is_fresh and self._revision_epoch == revision_epoch:
|
|
1238
|
+
self._last_revision_check_at = now
|
|
1239
|
+
current_project_id = self._workspace.active_project_id
|
|
1240
|
+
|
|
1241
|
+
if current_project_id != previous_project_id:
|
|
1242
|
+
self._registry = None
|
|
1243
|
+
self._relation_index = None
|
|
1244
|
+
self._metrics = None
|
|
1245
|
+
self._metrics_revision = ""
|
|
1246
|
+
self._clear_cpg_cache()
|
|
1247
|
+
self._prepared_response_cache.clear()
|
|
1248
|
+
self._focus = Focus(project_id=current_project_id) if current_project_id else Focus()
|
|
1249
|
+
elif revision != self._last_revision:
|
|
1250
|
+
self._registry = None
|
|
1251
|
+
self._relation_index = None
|
|
1252
|
+
self._metrics = None
|
|
1253
|
+
self._metrics_revision = ""
|
|
1254
|
+
self._clear_cpg_cache()
|
|
1255
|
+
self._prepared_response_cache.clear()
|
|
1256
|
+
elif self._focus.project_id != current_project_id:
|
|
1257
|
+
self._focus = Focus(project_id=current_project_id) if current_project_id else Focus()
|
|
1258
|
+
self._last_revision = revision
|
|
1259
|
+
return revision
|
|
1260
|
+
|
|
1261
|
+
def _clear_cpg_cache(self) -> None:
|
|
1262
|
+
self._cpg_cache.clear()
|
|
1263
|
+
self._cpg_cache_bytes = 0
|
|
1264
|
+
|
|
1265
|
+
def _require_cpg_subgraph(
|
|
1266
|
+
self,
|
|
1267
|
+
registry: _Registry,
|
|
1268
|
+
entry: _SymbolEntry,
|
|
1269
|
+
request: AgentRequest,
|
|
1270
|
+
) -> CpgSubgraph:
|
|
1271
|
+
if entry.kind not in _TYPE_KINDS | _ROUTINE_KINDS | {SymbolKind.UNIT}:
|
|
1272
|
+
raise AgentProtocolError(
|
|
1273
|
+
"cpg_not_applicable",
|
|
1274
|
+
f"CPG does not apply to {entry.kind.value} target {entry.target_id}.",
|
|
1275
|
+
)
|
|
1276
|
+
key = (
|
|
1277
|
+
registry.revision,
|
|
1278
|
+
entry.target_id,
|
|
1279
|
+
request.graph,
|
|
1280
|
+
request.direction,
|
|
1281
|
+
request.depth,
|
|
1282
|
+
)
|
|
1283
|
+
cached = self._cpg_cache.pop(key, None)
|
|
1284
|
+
if cached is not None:
|
|
1285
|
+
self._cpg_cache[key] = cached
|
|
1286
|
+
return cached
|
|
1287
|
+
try:
|
|
1288
|
+
document = registry.sources[entry.source_path]
|
|
1289
|
+
self._cpg_sources_parsed += 1
|
|
1290
|
+
parsed = DelphiParser(
|
|
1291
|
+
defines=document.defines,
|
|
1292
|
+
include_paths=document.include_paths,
|
|
1293
|
+
include_loader=workspace_include_loader(
|
|
1294
|
+
document.project_config,
|
|
1295
|
+
document.include_paths,
|
|
1296
|
+
),
|
|
1297
|
+
mode=(
|
|
1298
|
+
ParserMode.TOLERANT
|
|
1299
|
+
if len(self._workspace.units) >= 256
|
|
1300
|
+
else ParserMode.STRICT
|
|
1301
|
+
),
|
|
1302
|
+
).parse(
|
|
1303
|
+
document.text,
|
|
1304
|
+
str(entry.source_path),
|
|
1305
|
+
build_semantic=False,
|
|
1306
|
+
)
|
|
1307
|
+
graph = build_cpg_subgraph(
|
|
1308
|
+
target=_cpg_target(entry),
|
|
1309
|
+
syntax_root=parsed.root,
|
|
1310
|
+
candidates=_CpgCandidates(registry.entries),
|
|
1311
|
+
graph=request.graph,
|
|
1312
|
+
direction=request.direction,
|
|
1313
|
+
depth=request.depth,
|
|
1314
|
+
)
|
|
1315
|
+
except AgentProtocolError:
|
|
1316
|
+
raise
|
|
1317
|
+
except (OSError, UnicodeError, KeyError):
|
|
1318
|
+
raise AgentProtocolError(
|
|
1319
|
+
"source_unavailable",
|
|
1320
|
+
f"Could not read selected source {entry.path}.",
|
|
1321
|
+
) from None
|
|
1322
|
+
except Exception:
|
|
1323
|
+
raise AgentProtocolError(
|
|
1324
|
+
"cpg_build_failed",
|
|
1325
|
+
"Could not build the selected CPG subgraph.",
|
|
1326
|
+
) from None
|
|
1327
|
+
if graph.retained_bytes <= self._cpg_cache_limit:
|
|
1328
|
+
while (
|
|
1329
|
+
self._cpg_cache
|
|
1330
|
+
and self._cpg_cache_bytes + graph.retained_bytes
|
|
1331
|
+
> self._cpg_cache_limit
|
|
1332
|
+
):
|
|
1333
|
+
_, evicted = self._cpg_cache.popitem(last=False)
|
|
1334
|
+
self._cpg_cache_bytes -= evicted.retained_bytes
|
|
1335
|
+
self._cpg_cache[key] = graph
|
|
1336
|
+
self._cpg_cache_bytes += graph.retained_bytes
|
|
1337
|
+
return graph
|
|
1338
|
+
|
|
1339
|
+
def _open_items(self) -> list[dict[str, object]]:
|
|
1340
|
+
active_project_id = self._workspace.active_project_id
|
|
1341
|
+
items: list[dict[str, object]] = []
|
|
1342
|
+
for project in self._workspace.projects:
|
|
1343
|
+
project_item = _sanitize_workspace_mapping(
|
|
1344
|
+
project.to_mapping(),
|
|
1345
|
+
self._workspace.root,
|
|
1346
|
+
path_namespace="project",
|
|
1347
|
+
)
|
|
1348
|
+
items.append(
|
|
1349
|
+
{
|
|
1350
|
+
"item_type": "project",
|
|
1351
|
+
**project_item,
|
|
1352
|
+
"active": project.project_id == active_project_id,
|
|
1353
|
+
}
|
|
1354
|
+
)
|
|
1355
|
+
if not active_project_id:
|
|
1356
|
+
return items
|
|
1357
|
+
|
|
1358
|
+
for unit in self._workspace.units:
|
|
1359
|
+
display_path = unit_display_path(self._workspace.root, unit)
|
|
1360
|
+
items.append(
|
|
1361
|
+
{
|
|
1362
|
+
"item_type": "unit",
|
|
1363
|
+
"unit_id": unit_target_id(self._workspace.root, unit),
|
|
1364
|
+
"name": unit.name,
|
|
1365
|
+
"path": display_path,
|
|
1366
|
+
"has_error": unit.has_error,
|
|
1367
|
+
}
|
|
1368
|
+
)
|
|
1369
|
+
for include_file in self._workspace.include_files:
|
|
1370
|
+
items.append(
|
|
1371
|
+
{
|
|
1372
|
+
"item_type": "include_file",
|
|
1373
|
+
**_sanitize_workspace_mapping(
|
|
1374
|
+
include_file,
|
|
1375
|
+
self._workspace.root,
|
|
1376
|
+
path_namespace="include",
|
|
1377
|
+
),
|
|
1378
|
+
}
|
|
1379
|
+
)
|
|
1380
|
+
for entry in self._workspace.search_path_entries:
|
|
1381
|
+
items.append(
|
|
1382
|
+
{
|
|
1383
|
+
"item_type": "search_path",
|
|
1384
|
+
**_sanitize_workspace_mapping(
|
|
1385
|
+
entry,
|
|
1386
|
+
self._workspace.root,
|
|
1387
|
+
path_namespace="search-path",
|
|
1388
|
+
),
|
|
1389
|
+
}
|
|
1390
|
+
)
|
|
1391
|
+
for entry in self._workspace.include_path_entries:
|
|
1392
|
+
items.append(
|
|
1393
|
+
{
|
|
1394
|
+
"item_type": "include_path",
|
|
1395
|
+
**_sanitize_workspace_mapping(
|
|
1396
|
+
entry,
|
|
1397
|
+
self._workspace.root,
|
|
1398
|
+
path_namespace="include-path",
|
|
1399
|
+
),
|
|
1400
|
+
}
|
|
1401
|
+
)
|
|
1402
|
+
for entry in self._workspace.define_entries:
|
|
1403
|
+
items.append(
|
|
1404
|
+
{
|
|
1405
|
+
"item_type": "define",
|
|
1406
|
+
**_sanitize_workspace_mapping(
|
|
1407
|
+
entry,
|
|
1408
|
+
self._workspace.root,
|
|
1409
|
+
path_namespace="define",
|
|
1410
|
+
),
|
|
1411
|
+
}
|
|
1412
|
+
)
|
|
1413
|
+
items.extend(self._problem_items())
|
|
1414
|
+
return items
|
|
1415
|
+
|
|
1416
|
+
def _problem_items(self) -> list[dict[str, object]]:
|
|
1417
|
+
return [
|
|
1418
|
+
{
|
|
1419
|
+
"item_type": "problem",
|
|
1420
|
+
**_sanitize_workspace_mapping(
|
|
1421
|
+
problem,
|
|
1422
|
+
self._workspace.root,
|
|
1423
|
+
path_namespace="problem",
|
|
1424
|
+
),
|
|
1425
|
+
}
|
|
1426
|
+
for problem in self._workspace.problems
|
|
1427
|
+
]
|
|
1428
|
+
|
|
1429
|
+
def _handle_focus(self, request: AgentRequest, revision: str) -> AgentResponse:
|
|
1430
|
+
if request.target_id:
|
|
1431
|
+
registry = self._require_registry(revision)
|
|
1432
|
+
entry = self._resolve_target(registry, request.target_id, allow_focused=False)
|
|
1433
|
+
self._focus = Focus(
|
|
1434
|
+
project_id=registry.project_id,
|
|
1435
|
+
unit_id=entry.unit_id,
|
|
1436
|
+
target_id=entry.target_id,
|
|
1437
|
+
)
|
|
1438
|
+
elif request.project_id:
|
|
1439
|
+
self._focus = Focus(project_id=self._workspace.active_project_id)
|
|
1440
|
+
self._require_registry(revision)
|
|
1441
|
+
return self._response(request, revision, [self._focus.to_mapping()])
|
|
1442
|
+
|
|
1443
|
+
def _handle_metrics(
|
|
1444
|
+
self,
|
|
1445
|
+
request: AgentRequest,
|
|
1446
|
+
revision: str,
|
|
1447
|
+
*,
|
|
1448
|
+
cancel_check: Callable[[], None] | None = None,
|
|
1449
|
+
) -> AgentResponse:
|
|
1450
|
+
if request.detail not in {"summary", "members"}:
|
|
1451
|
+
raise AgentProtocolError(
|
|
1452
|
+
"invalid_detail",
|
|
1453
|
+
"Metrics supports only summary or members detail.",
|
|
1454
|
+
)
|
|
1455
|
+
metrics = self._require_metrics(revision, cancel_check=cancel_check)
|
|
1456
|
+
if cancel_check is not None:
|
|
1457
|
+
cancel_check()
|
|
1458
|
+
detail = request.detail == "members"
|
|
1459
|
+
if request.target_id:
|
|
1460
|
+
unit = next(
|
|
1461
|
+
(
|
|
1462
|
+
candidate
|
|
1463
|
+
for candidate in metrics.file_metrics
|
|
1464
|
+
if candidate.unit_id == request.target_id
|
|
1465
|
+
),
|
|
1466
|
+
None,
|
|
1467
|
+
)
|
|
1468
|
+
if unit is None:
|
|
1469
|
+
raise AgentProtocolError(
|
|
1470
|
+
"target_not_found",
|
|
1471
|
+
f"Target not found: {request.target_id}.",
|
|
1472
|
+
)
|
|
1473
|
+
return self._response(
|
|
1474
|
+
request,
|
|
1475
|
+
revision,
|
|
1476
|
+
[unit_metric_item(unit, detail=detail)],
|
|
1477
|
+
target_id=unit.unit_id,
|
|
1478
|
+
)
|
|
1479
|
+
|
|
1480
|
+
units = metrics.file_metrics
|
|
1481
|
+
if request.query:
|
|
1482
|
+
query = request.query.casefold()
|
|
1483
|
+
units = tuple(
|
|
1484
|
+
unit
|
|
1485
|
+
for unit in units
|
|
1486
|
+
if query in unit.name.casefold() or query in unit.path.casefold()
|
|
1487
|
+
)
|
|
1488
|
+
items = [unit_metric_item(unit, detail=detail) for unit in units]
|
|
1489
|
+
else:
|
|
1490
|
+
items = [
|
|
1491
|
+
project_metric_item(metrics),
|
|
1492
|
+
*(unit_metric_item(unit, detail=detail) for unit in units),
|
|
1493
|
+
]
|
|
1494
|
+
return self._response(request, revision, items)
|
|
1495
|
+
|
|
1496
|
+
def _require_metrics(
|
|
1497
|
+
self,
|
|
1498
|
+
revision: str,
|
|
1499
|
+
*,
|
|
1500
|
+
cancel_check: Callable[[], None] | None = None,
|
|
1501
|
+
) -> ProjectMetrics:
|
|
1502
|
+
self._require_selected_project()
|
|
1503
|
+
if self._metrics is not None and self._metrics_revision == revision:
|
|
1504
|
+
return self._metrics
|
|
1505
|
+
self._metrics = build_workspace_metrics(
|
|
1506
|
+
self._workspace,
|
|
1507
|
+
workers=max(1, self._workers),
|
|
1508
|
+
worker_memory_budget_bytes=self._worker_memory_budget_bytes,
|
|
1509
|
+
cancel_check=cancel_check,
|
|
1510
|
+
)
|
|
1511
|
+
self._metrics_revision = revision
|
|
1512
|
+
return self._metrics
|
|
1513
|
+
|
|
1514
|
+
def _require_registry(self, revision: str) -> _Registry:
|
|
1515
|
+
project_id = self._require_selected_project()
|
|
1516
|
+
if (
|
|
1517
|
+
self._registry is not None
|
|
1518
|
+
and self._registry.project_id == project_id
|
|
1519
|
+
and self._registry.revision == revision
|
|
1520
|
+
):
|
|
1521
|
+
return self._registry
|
|
1522
|
+
(
|
|
1523
|
+
self._registry,
|
|
1524
|
+
self._parallel_stats,
|
|
1525
|
+
self._navigation_disk_hits,
|
|
1526
|
+
self._navigation_disk_misses,
|
|
1527
|
+
) = _build_registry(
|
|
1528
|
+
self._workspace,
|
|
1529
|
+
project_id,
|
|
1530
|
+
revision,
|
|
1531
|
+
workers=self._workers,
|
|
1532
|
+
worker_memory_budget_bytes=self._worker_memory_budget_bytes,
|
|
1533
|
+
navigation_store=self._navigation_store,
|
|
1534
|
+
navigation_cache_max_bytes=self._navigation_cache_max_bytes,
|
|
1535
|
+
)
|
|
1536
|
+
if self._focus.target_id:
|
|
1537
|
+
focused_entry = self._registry.by_target.get(self._focus.target_id)
|
|
1538
|
+
if focused_entry is None:
|
|
1539
|
+
self._focus = Focus(project_id=project_id)
|
|
1540
|
+
else:
|
|
1541
|
+
self._focus = Focus(
|
|
1542
|
+
project_id=project_id,
|
|
1543
|
+
unit_id=focused_entry.unit_id,
|
|
1544
|
+
target_id=focused_entry.target_id,
|
|
1545
|
+
)
|
|
1546
|
+
return self._registry
|
|
1547
|
+
|
|
1548
|
+
def _require_relation_index(
|
|
1549
|
+
self,
|
|
1550
|
+
registry: _Registry,
|
|
1551
|
+
relation: str,
|
|
1552
|
+
) -> ProjectRelationIndex:
|
|
1553
|
+
requires_complete_targets = relation not in {"uses", "used_by"}
|
|
1554
|
+
if (
|
|
1555
|
+
self._relation_index is not None
|
|
1556
|
+
and self._relation_index.project_id == registry.project_id
|
|
1557
|
+
and self._relation_index.revision == registry.revision
|
|
1558
|
+
and (
|
|
1559
|
+
not requires_complete_targets
|
|
1560
|
+
or self._relation_index.targets_complete
|
|
1561
|
+
)
|
|
1562
|
+
and (
|
|
1563
|
+
relation not in {"uses", "used_by"}
|
|
1564
|
+
or self._relation_index.unit_targets_complete
|
|
1565
|
+
)
|
|
1566
|
+
):
|
|
1567
|
+
return self._relation_index
|
|
1568
|
+
targets = tuple(
|
|
1569
|
+
_relation_target(entry)
|
|
1570
|
+
for entry in registry.entries
|
|
1571
|
+
if requires_complete_targets or entry.kind == SymbolKind.UNIT
|
|
1572
|
+
)
|
|
1573
|
+
self._relation_index = ProjectRelationIndex(
|
|
1574
|
+
self._workspace,
|
|
1575
|
+
registry.project_id,
|
|
1576
|
+
registry.revision,
|
|
1577
|
+
targets,
|
|
1578
|
+
targets_complete=requires_complete_targets,
|
|
1579
|
+
unit_targets_complete=True,
|
|
1580
|
+
)
|
|
1581
|
+
return self._relation_index
|
|
1582
|
+
|
|
1583
|
+
def _require_reference_relation_index(
|
|
1584
|
+
self,
|
|
1585
|
+
registry: _Registry,
|
|
1586
|
+
entry: _SymbolEntry,
|
|
1587
|
+
) -> ProjectRelationIndex:
|
|
1588
|
+
if (
|
|
1589
|
+
self._relation_index is not None
|
|
1590
|
+
and self._relation_index.project_id == registry.project_id
|
|
1591
|
+
and self._relation_index.revision == registry.revision
|
|
1592
|
+
and self._relation_index.has_target(entry.target_id)
|
|
1593
|
+
):
|
|
1594
|
+
return self._relation_index
|
|
1595
|
+
targets = tuple(
|
|
1596
|
+
_relation_target(candidate)
|
|
1597
|
+
for candidate in registry.entries
|
|
1598
|
+
if candidate.source_path == entry.source_path
|
|
1599
|
+
)
|
|
1600
|
+
self._relation_index = ProjectRelationIndex(
|
|
1601
|
+
self._workspace,
|
|
1602
|
+
registry.project_id,
|
|
1603
|
+
registry.revision,
|
|
1604
|
+
targets,
|
|
1605
|
+
targets_complete=False,
|
|
1606
|
+
unit_targets_complete=False,
|
|
1607
|
+
)
|
|
1608
|
+
return self._relation_index
|
|
1609
|
+
|
|
1610
|
+
def _require_selected_project(self) -> str:
|
|
1611
|
+
project_id = self._workspace.active_project_id
|
|
1612
|
+
if not project_id:
|
|
1613
|
+
raise AgentProtocolError(
|
|
1614
|
+
"project_required",
|
|
1615
|
+
"Multiple projects were found. Run 'query open', then select one "
|
|
1616
|
+
"with 'query focus --project-id PROJECT_ID'.",
|
|
1617
|
+
)
|
|
1618
|
+
return project_id
|
|
1619
|
+
|
|
1620
|
+
def _resolve_target(
|
|
1621
|
+
self,
|
|
1622
|
+
registry: _Registry,
|
|
1623
|
+
target_id: str,
|
|
1624
|
+
*,
|
|
1625
|
+
allow_focused: bool = True,
|
|
1626
|
+
) -> _SymbolEntry:
|
|
1627
|
+
resolved_id = target_id
|
|
1628
|
+
if not resolved_id and allow_focused:
|
|
1629
|
+
resolved_id = self._focus.target_id
|
|
1630
|
+
if not resolved_id:
|
|
1631
|
+
raise AgentProtocolError("target_required", "A target_id or focused target is required.")
|
|
1632
|
+
entry = registry.by_target.get(resolved_id)
|
|
1633
|
+
if entry is None:
|
|
1634
|
+
raise AgentProtocolError("target_not_found", f"Target not found: {resolved_id}.")
|
|
1635
|
+
return entry
|
|
1636
|
+
|
|
1637
|
+
def _inspect_items(
|
|
1638
|
+
self,
|
|
1639
|
+
registry: _Registry,
|
|
1640
|
+
entry: _SymbolEntry,
|
|
1641
|
+
request: AgentRequest,
|
|
1642
|
+
) -> list[dict[str, object]]:
|
|
1643
|
+
if request.detail == "summary":
|
|
1644
|
+
return [entry.card()]
|
|
1645
|
+
if request.detail == "members":
|
|
1646
|
+
return [
|
|
1647
|
+
candidate.card()
|
|
1648
|
+
for candidate in registry.entries
|
|
1649
|
+
if candidate.parent_target_id == entry.target_id
|
|
1650
|
+
]
|
|
1651
|
+
|
|
1652
|
+
document = registry.sources[entry.source_path]
|
|
1653
|
+
if request.detail == "declaration":
|
|
1654
|
+
start, end = _declaration_span(document, entry)
|
|
1655
|
+
return _source_items(
|
|
1656
|
+
document,
|
|
1657
|
+
start,
|
|
1658
|
+
end,
|
|
1659
|
+
request.max_chars,
|
|
1660
|
+
role="declaration",
|
|
1661
|
+
target_id=entry.target_id,
|
|
1662
|
+
)
|
|
1663
|
+
if request.detail == "context":
|
|
1664
|
+
declaration_start, declaration_end = _declaration_span(document, entry)
|
|
1665
|
+
start_line = max(1, document.line_col(declaration_start)[0] - 3)
|
|
1666
|
+
end_line = document.line_col(declaration_end)[0] + 5
|
|
1667
|
+
start = document.line_start(start_line)
|
|
1668
|
+
end = document.line_end(end_line, include_newline=True)
|
|
1669
|
+
return [entry.card(), *_source_items(
|
|
1670
|
+
document,
|
|
1671
|
+
start,
|
|
1672
|
+
end,
|
|
1673
|
+
request.max_chars,
|
|
1674
|
+
role="context",
|
|
1675
|
+
target_id=entry.target_id,
|
|
1676
|
+
)]
|
|
1677
|
+
if request.detail == "body":
|
|
1678
|
+
body_entry, span = _body_entry_and_span(registry, entry)
|
|
1679
|
+
if span is None:
|
|
1680
|
+
raise AgentProtocolError(
|
|
1681
|
+
"body_unavailable",
|
|
1682
|
+
f"No routine or type body is available for target: {entry.target_id}.",
|
|
1683
|
+
)
|
|
1684
|
+
body_document = registry.sources[body_entry.source_path]
|
|
1685
|
+
return _source_items(
|
|
1686
|
+
body_document,
|
|
1687
|
+
span[0],
|
|
1688
|
+
span[1],
|
|
1689
|
+
request.max_chars,
|
|
1690
|
+
role="body",
|
|
1691
|
+
target_id=body_entry.target_id,
|
|
1692
|
+
)
|
|
1693
|
+
if request.detail == "implementations":
|
|
1694
|
+
items: list[dict[str, object]] = []
|
|
1695
|
+
for counterpart in _matching_counterparts(registry, entry):
|
|
1696
|
+
card = counterpart.card()
|
|
1697
|
+
card["item_type"] = "counterpart"
|
|
1698
|
+
items.append(card)
|
|
1699
|
+
counterpart_document = registry.sources[counterpart.source_path]
|
|
1700
|
+
start, end = _declaration_span(counterpart_document, counterpart)
|
|
1701
|
+
items.extend(
|
|
1702
|
+
_source_items(
|
|
1703
|
+
counterpart_document,
|
|
1704
|
+
start,
|
|
1705
|
+
end,
|
|
1706
|
+
request.max_chars,
|
|
1707
|
+
role="counterpart_declaration",
|
|
1708
|
+
target_id=counterpart.target_id,
|
|
1709
|
+
)
|
|
1710
|
+
)
|
|
1711
|
+
return items
|
|
1712
|
+
raise AgentProtocolError(
|
|
1713
|
+
"invalid_detail",
|
|
1714
|
+
f"Unsupported detail value: {request.detail!r}.",
|
|
1715
|
+
)
|
|
1716
|
+
|
|
1717
|
+
def _response(
|
|
1718
|
+
self,
|
|
1719
|
+
request: AgentRequest,
|
|
1720
|
+
revision: str,
|
|
1721
|
+
items: Sequence[dict[str, object]],
|
|
1722
|
+
*,
|
|
1723
|
+
target_id: str = "",
|
|
1724
|
+
items_prepared: bool = False,
|
|
1725
|
+
) -> AgentResponse:
|
|
1726
|
+
fingerprint = _request_fingerprint(
|
|
1727
|
+
request,
|
|
1728
|
+
project_id=self._workspace.active_project_id,
|
|
1729
|
+
target_id=target_id or request.target_id,
|
|
1730
|
+
)
|
|
1731
|
+
cache_key = (revision, fingerprint)
|
|
1732
|
+
prepared = self._prepared_response_cache.pop(cache_key, None)
|
|
1733
|
+
if prepared is None:
|
|
1734
|
+
prepared = (
|
|
1735
|
+
items
|
|
1736
|
+
if items_prepared
|
|
1737
|
+
else _prepare_items(items, request.max_chars)
|
|
1738
|
+
)
|
|
1739
|
+
self._prepared_response_cache[cache_key] = prepared
|
|
1740
|
+
while len(self._prepared_response_cache) > _PREPARED_RESPONSE_CACHE_SIZE:
|
|
1741
|
+
self._prepared_response_cache.popitem(last=False)
|
|
1742
|
+
page, selected = paginate_items(
|
|
1743
|
+
prepared,
|
|
1744
|
+
revision,
|
|
1745
|
+
fingerprint,
|
|
1746
|
+
request.max_items,
|
|
1747
|
+
request.max_chars,
|
|
1748
|
+
request.cursor,
|
|
1749
|
+
)
|
|
1750
|
+
context_chars = len(_compact_json(selected))
|
|
1751
|
+
return AgentResponse(
|
|
1752
|
+
workspace_revision=revision,
|
|
1753
|
+
focus=self._focus,
|
|
1754
|
+
result=selected,
|
|
1755
|
+
page=page,
|
|
1756
|
+
context=ContextBudget(chars=context_chars),
|
|
1757
|
+
)
|
|
1758
|
+
|
|
1759
|
+
|
|
1760
|
+
def _validated_request(request: AgentRequest | Mapping[str, object]) -> AgentRequest:
|
|
1761
|
+
if isinstance(request, AgentRequest):
|
|
1762
|
+
return AgentRequest.from_mapping(request.to_mapping())
|
|
1763
|
+
return AgentRequest.from_mapping(request)
|
|
1764
|
+
|
|
1765
|
+
|
|
1766
|
+
def _build_registry(
|
|
1767
|
+
workspace: AgentWorkspace,
|
|
1768
|
+
project_id: str,
|
|
1769
|
+
revision: str,
|
|
1770
|
+
*,
|
|
1771
|
+
workers: int = 0,
|
|
1772
|
+
worker_memory_budget_bytes: int | None = None,
|
|
1773
|
+
navigation_store: NavigationShardStore | None = None,
|
|
1774
|
+
navigation_cache_max_bytes: int = _DEFAULT_NAVIGATION_CACHE_MAX_BYTES,
|
|
1775
|
+
exact_query: str = "",
|
|
1776
|
+
exact_path: str = "",
|
|
1777
|
+
) -> tuple[_Registry, ParallelBuildStats, int, int]:
|
|
1778
|
+
build_started = time.perf_counter()
|
|
1779
|
+
raw_symbols: list[_RawSymbol] = []
|
|
1780
|
+
units = tuple(workspace.units)
|
|
1781
|
+
exact_needle = _normalized(exact_query.strip())
|
|
1782
|
+
exact_path_needle = exact_path.strip().replace("\\", "/").casefold()
|
|
1783
|
+
include_files = workspace.include_files
|
|
1784
|
+
selected_include_paths = tuple(
|
|
1785
|
+
include["path"]
|
|
1786
|
+
for include in include_files
|
|
1787
|
+
if exact_path_needle
|
|
1788
|
+
and include["path"].replace("\\", "/").casefold()
|
|
1789
|
+
== exact_path_needle
|
|
1790
|
+
)
|
|
1791
|
+
name_needle = exact_needle.rsplit('.', 1)[-1] if exact_needle else ""
|
|
1792
|
+
outline_name_needle = name_needle if "." not in exact_needle else ""
|
|
1793
|
+
tasks = (
|
|
1794
|
+
()
|
|
1795
|
+
if selected_include_paths
|
|
1796
|
+
else tuple(
|
|
1797
|
+
_NavigationTask(
|
|
1798
|
+
ordinal,
|
|
1799
|
+
str(unit_source_path(workspace.root, unit)),
|
|
1800
|
+
unit_display_path(workspace.root, unit),
|
|
1801
|
+
unit.name,
|
|
1802
|
+
unit.path,
|
|
1803
|
+
unit.unit_id,
|
|
1804
|
+
unit.has_error,
|
|
1805
|
+
workspace.defines,
|
|
1806
|
+
workspace.include_paths,
|
|
1807
|
+
workspace.project_config,
|
|
1808
|
+
outline_name_needle,
|
|
1809
|
+
)
|
|
1810
|
+
for ordinal, unit in enumerate(units)
|
|
1811
|
+
)
|
|
1812
|
+
)
|
|
1813
|
+
|
|
1814
|
+
if exact_path_needle:
|
|
1815
|
+
selected_tasks: tuple[_NavigationTask, ...]
|
|
1816
|
+
if selected_include_paths:
|
|
1817
|
+
selected_tasks = ()
|
|
1818
|
+
else:
|
|
1819
|
+
selected_tasks = tuple(
|
|
1820
|
+
task
|
|
1821
|
+
for task in tasks
|
|
1822
|
+
if task.display_path.replace("\\", "/").casefold()
|
|
1823
|
+
== exact_path_needle
|
|
1824
|
+
)
|
|
1825
|
+
if not selected_tasks:
|
|
1826
|
+
selected_tasks = tuple(
|
|
1827
|
+
task
|
|
1828
|
+
for task in tasks
|
|
1829
|
+
if _source_may_include_selected_path(
|
|
1830
|
+
task.source_path,
|
|
1831
|
+
exact_path_needle,
|
|
1832
|
+
)
|
|
1833
|
+
)
|
|
1834
|
+
tasks = selected_tasks
|
|
1835
|
+
elif exact_needle:
|
|
1836
|
+
tasks = tuple(
|
|
1837
|
+
task
|
|
1838
|
+
for task in tasks
|
|
1839
|
+
if _source_may_contain_exact_name(task.source_path, name_needle)
|
|
1840
|
+
)
|
|
1841
|
+
source_specs = {
|
|
1842
|
+
Path(task.source_path): _SourceSpec(
|
|
1843
|
+
Path(task.source_path),
|
|
1844
|
+
task.display_path,
|
|
1845
|
+
workspace.defines,
|
|
1846
|
+
workspace.include_paths,
|
|
1847
|
+
workspace.project_config,
|
|
1848
|
+
)
|
|
1849
|
+
for task in tasks
|
|
1850
|
+
}
|
|
1851
|
+
|
|
1852
|
+
def consume_result(result: _NavigationResult) -> None:
|
|
1853
|
+
if result.read_error:
|
|
1854
|
+
unit = units[result.ordinal]
|
|
1855
|
+
display_path = unit_display_path(workspace.root, unit)
|
|
1856
|
+
raise AgentProtocolError(
|
|
1857
|
+
"source_unavailable",
|
|
1858
|
+
f"Could not read selected source {display_path}.",
|
|
1859
|
+
)
|
|
1860
|
+
if navigation_store is not None and result.cache_key:
|
|
1861
|
+
try:
|
|
1862
|
+
navigation_store.store(
|
|
1863
|
+
result.cache_key,
|
|
1864
|
+
_navigation_shard_payload(result),
|
|
1865
|
+
)
|
|
1866
|
+
except (OSError, TypeError, ValueError):
|
|
1867
|
+
pass
|
|
1868
|
+
if exact_needle and not any(
|
|
1869
|
+
_normalized(raw.name) == exact_needle
|
|
1870
|
+
or _normalized(raw.qualified_name) == exact_needle
|
|
1871
|
+
for raw in result.raw_symbols
|
|
1872
|
+
):
|
|
1873
|
+
return
|
|
1874
|
+
raw_symbols.extend(result.raw_symbols)
|
|
1875
|
+
|
|
1876
|
+
disk_hits = 0
|
|
1877
|
+
disk_misses = 0
|
|
1878
|
+
live_navigation_keys: set[str] = set()
|
|
1879
|
+
pending_tasks: list[_NavigationTask] = []
|
|
1880
|
+
if navigation_store is None:
|
|
1881
|
+
pending_tasks.extend(tasks)
|
|
1882
|
+
else:
|
|
1883
|
+
for task in tasks:
|
|
1884
|
+
try:
|
|
1885
|
+
text = read_source_text(Path(task.source_path))
|
|
1886
|
+
cache_key = navigation_cache_key(
|
|
1887
|
+
_navigation_cache_text(
|
|
1888
|
+
text,
|
|
1889
|
+
_thin_unit_include(task, text),
|
|
1890
|
+
),
|
|
1891
|
+
task.defines,
|
|
1892
|
+
)
|
|
1893
|
+
live_navigation_keys.add(cache_key)
|
|
1894
|
+
payload = navigation_store.load(cache_key)
|
|
1895
|
+
except (OSError, UnicodeError, ValueError):
|
|
1896
|
+
cache_key = ""
|
|
1897
|
+
payload = None
|
|
1898
|
+
cached = (
|
|
1899
|
+
_navigation_result_from_shard(task, payload)
|
|
1900
|
+
if payload is not None
|
|
1901
|
+
else None
|
|
1902
|
+
)
|
|
1903
|
+
if cached is None:
|
|
1904
|
+
disk_misses += 1
|
|
1905
|
+
pending_tasks.append(replace(task, cache_key=cache_key or "miss"))
|
|
1906
|
+
continue
|
|
1907
|
+
disk_hits += 1
|
|
1908
|
+
consume_result(cached)
|
|
1909
|
+
|
|
1910
|
+
outline_batch = run_outline_tasks(
|
|
1911
|
+
pending_tasks,
|
|
1912
|
+
configured_workers=workers,
|
|
1913
|
+
memory_budget_bytes=worker_memory_budget_bytes,
|
|
1914
|
+
on_complete=consume_result,
|
|
1915
|
+
retain_results=False,
|
|
1916
|
+
task_runner=_parse_navigation_task,
|
|
1917
|
+
)
|
|
1918
|
+
if navigation_store is not None:
|
|
1919
|
+
try:
|
|
1920
|
+
navigation_store.prune(
|
|
1921
|
+
live_navigation_keys,
|
|
1922
|
+
navigation_cache_max_bytes,
|
|
1923
|
+
)
|
|
1924
|
+
except (OSError, ValueError):
|
|
1925
|
+
pass
|
|
1926
|
+
parallel_stats = replace(
|
|
1927
|
+
outline_batch.stats,
|
|
1928
|
+
files_completed=len(tasks) + len(selected_include_paths),
|
|
1929
|
+
elapsed_seconds=time.perf_counter() - build_started,
|
|
1930
|
+
)
|
|
1931
|
+
|
|
1932
|
+
entries_tuple = _entries_from_raw_symbols(raw_symbols)
|
|
1933
|
+
for raw in raw_symbols:
|
|
1934
|
+
source_specs.setdefault(
|
|
1935
|
+
raw.source_path,
|
|
1936
|
+
_SourceSpec(
|
|
1937
|
+
raw.source_path,
|
|
1938
|
+
raw.path,
|
|
1939
|
+
workspace.defines,
|
|
1940
|
+
workspace.include_paths,
|
|
1941
|
+
workspace.project_config,
|
|
1942
|
+
),
|
|
1943
|
+
)
|
|
1944
|
+
source_cache_bytes = _source_cache_budget(worker_memory_budget_bytes)
|
|
1945
|
+
sources = _SourceStore(source_specs, max_loaded_bytes=source_cache_bytes)
|
|
1946
|
+
static_retained_bytes = _estimate_registry_bytes(entries_tuple, sources)
|
|
1947
|
+
registry = _Registry(
|
|
1948
|
+
project_id=project_id,
|
|
1949
|
+
revision=revision,
|
|
1950
|
+
entries=entries_tuple,
|
|
1951
|
+
by_target={entry.target_id: entry for entry in entries_tuple},
|
|
1952
|
+
sources=sources,
|
|
1953
|
+
ranked_queries=OrderedDict(),
|
|
1954
|
+
static_retained_bytes=static_retained_bytes,
|
|
1955
|
+
)
|
|
1956
|
+
if selected_include_paths:
|
|
1957
|
+
_augment_registry_with_include_matches(
|
|
1958
|
+
workspace,
|
|
1959
|
+
registry,
|
|
1960
|
+
exact_query,
|
|
1961
|
+
exact_path=exact_path,
|
|
1962
|
+
)
|
|
1963
|
+
return (
|
|
1964
|
+
registry,
|
|
1965
|
+
parallel_stats,
|
|
1966
|
+
disk_hits,
|
|
1967
|
+
disk_misses,
|
|
1968
|
+
)
|
|
1969
|
+
|
|
1970
|
+
|
|
1971
|
+
def _entries_from_raw_symbols(
|
|
1972
|
+
raw_symbols: Sequence[_RawSymbol],
|
|
1973
|
+
) -> tuple[_SymbolEntry, ...]:
|
|
1974
|
+
ordered = sorted(raw_symbols, key=_raw_sort_key)
|
|
1975
|
+
overload_groups: dict[tuple[str, str, str, str], list[_RawSymbol]] = {}
|
|
1976
|
+
for raw in raw_symbols:
|
|
1977
|
+
identity = (
|
|
1978
|
+
raw.kind.value.casefold(),
|
|
1979
|
+
raw.path.casefold(),
|
|
1980
|
+
_normalized(raw.qualified_name),
|
|
1981
|
+
_normalized(raw.signature),
|
|
1982
|
+
)
|
|
1983
|
+
overload_groups.setdefault(identity, []).append(raw)
|
|
1984
|
+
ordinals: dict[int, int] = {}
|
|
1985
|
+
for group in overload_groups.values():
|
|
1986
|
+
overload_order = sorted(
|
|
1987
|
+
group,
|
|
1988
|
+
key=lambda raw: (
|
|
1989
|
+
raw.line,
|
|
1990
|
+
raw.column,
|
|
1991
|
+
_raw_sort_key(raw),
|
|
1992
|
+
),
|
|
1993
|
+
)
|
|
1994
|
+
for ordinal, raw in enumerate(overload_order):
|
|
1995
|
+
ordinals[id(raw)] = ordinal
|
|
1996
|
+
|
|
1997
|
+
shared_strings: dict[str, str] = {}
|
|
1998
|
+
|
|
1999
|
+
def shared(value: str) -> str:
|
|
2000
|
+
return shared_strings.setdefault(value, value)
|
|
2001
|
+
|
|
2002
|
+
target_ids = {
|
|
2003
|
+
id(raw): make_target_id(
|
|
2004
|
+
raw.kind.value,
|
|
2005
|
+
raw.path,
|
|
2006
|
+
_target_identity_name(raw),
|
|
2007
|
+
ordinals[id(raw)],
|
|
2008
|
+
)
|
|
2009
|
+
for raw in raw_symbols
|
|
2010
|
+
}
|
|
2011
|
+
parent_ids: dict[tuple[Path, str], str] = {}
|
|
2012
|
+
for raw in raw_symbols:
|
|
2013
|
+
if raw.kind in {
|
|
2014
|
+
SymbolKind.CLASS,
|
|
2015
|
+
SymbolKind.RECORD,
|
|
2016
|
+
SymbolKind.INTERFACE,
|
|
2017
|
+
SymbolKind.TYPE,
|
|
2018
|
+
}:
|
|
2019
|
+
parent_ids.setdefault(
|
|
2020
|
+
(raw.source_path, _normalized(raw.qualified_name)),
|
|
2021
|
+
target_ids[id(raw)],
|
|
2022
|
+
)
|
|
2023
|
+
|
|
2024
|
+
entries: list[_SymbolEntry] = []
|
|
2025
|
+
for raw in ordered:
|
|
2026
|
+
ordinal = ordinals[id(raw)]
|
|
2027
|
+
(
|
|
2028
|
+
normalized_name,
|
|
2029
|
+
normalized_qualified_name,
|
|
2030
|
+
relative_name_offset,
|
|
2031
|
+
) = _normalized_search_fields(
|
|
2032
|
+
raw.name,
|
|
2033
|
+
raw.qualified_name,
|
|
2034
|
+
raw.unit_name,
|
|
2035
|
+
)
|
|
2036
|
+
entry = _SymbolEntry(
|
|
2037
|
+
name=shared(raw.name),
|
|
2038
|
+
kind=raw.kind,
|
|
2039
|
+
line=raw.line,
|
|
2040
|
+
column=raw.column,
|
|
2041
|
+
visibility=raw.visibility,
|
|
2042
|
+
type_name=shared(raw.type_name),
|
|
2043
|
+
source_path=raw.source_path,
|
|
2044
|
+
path=shared(raw.path),
|
|
2045
|
+
unit_id=shared(raw.unit_id),
|
|
2046
|
+
unit_name=shared(raw.unit_name),
|
|
2047
|
+
qualified_name=shared(raw.qualified_name),
|
|
2048
|
+
normalized_name=shared(normalized_name),
|
|
2049
|
+
normalized_qualified_name=shared(normalized_qualified_name),
|
|
2050
|
+
relative_name_offset=relative_name_offset,
|
|
2051
|
+
owner=shared(raw.owner),
|
|
2052
|
+
signature=shared(raw.signature),
|
|
2053
|
+
ordinal=ordinal,
|
|
2054
|
+
target_id=target_ids[id(raw)],
|
|
2055
|
+
card_json_chars=0,
|
|
2056
|
+
card_json_upper_bound=0,
|
|
2057
|
+
parent_target_id=parent_ids.get(
|
|
2058
|
+
(
|
|
2059
|
+
raw.source_path,
|
|
2060
|
+
_normalized(raw.parent_qualified_name),
|
|
2061
|
+
),
|
|
2062
|
+
"",
|
|
2063
|
+
),
|
|
2064
|
+
context_ambiguous=raw.context_ambiguous,
|
|
2065
|
+
)
|
|
2066
|
+
entries.append(
|
|
2067
|
+
replace(
|
|
2068
|
+
entry,
|
|
2069
|
+
card_json_chars=_symbol_card_json_chars(entry),
|
|
2070
|
+
card_json_upper_bound=_symbol_card_json_upper_bound(entry),
|
|
2071
|
+
)
|
|
2072
|
+
)
|
|
2073
|
+
return tuple(sorted(entries, key=_entry_sort_key))
|
|
2074
|
+
|
|
2075
|
+
|
|
2076
|
+
def _augment_registry_with_include_matches(
|
|
2077
|
+
workspace: AgentWorkspace,
|
|
2078
|
+
registry: _Registry,
|
|
2079
|
+
query: str,
|
|
2080
|
+
*,
|
|
2081
|
+
exact_path: str = "",
|
|
2082
|
+
) -> bool:
|
|
2083
|
+
normalized_query = unicodedata.normalize('NFC', query.strip()).casefold()
|
|
2084
|
+
if normalized_query in registry.include_queries:
|
|
2085
|
+
return False
|
|
2086
|
+
registry.include_queries.add(normalized_query)
|
|
2087
|
+
query_tail = normalized_query.rsplit('.', 1)[-1]
|
|
2088
|
+
if not query_tail:
|
|
2089
|
+
return False
|
|
2090
|
+
exact_path_needle = exact_path.strip().replace("\\", "/").casefold()
|
|
2091
|
+
|
|
2092
|
+
raw_symbols: list[_RawSymbol] = []
|
|
2093
|
+
for include in workspace.include_files:
|
|
2094
|
+
recorded_path = include['path']
|
|
2095
|
+
display_path = recorded_path
|
|
2096
|
+
if (
|
|
2097
|
+
exact_path_needle
|
|
2098
|
+
and display_path.replace("\\", "/").casefold()
|
|
2099
|
+
!= exact_path_needle
|
|
2100
|
+
):
|
|
2101
|
+
continue
|
|
2102
|
+
source_path = workspace.include_source_path(recorded_path)
|
|
2103
|
+
if source_path in registry.indexed_include_paths:
|
|
2104
|
+
continue
|
|
2105
|
+
try:
|
|
2106
|
+
source = read_source_text(source_path)
|
|
2107
|
+
except (OSError, UnicodeError):
|
|
2108
|
+
continue
|
|
2109
|
+
if query_tail not in source.casefold():
|
|
2110
|
+
continue
|
|
2111
|
+
|
|
2112
|
+
registry.indexed_include_paths.add(source_path)
|
|
2113
|
+
include_name = (
|
|
2114
|
+
query.strip().split('.', 1)[0]
|
|
2115
|
+
if exact_path_needle and '.' in query
|
|
2116
|
+
else Path(include['name']).stem
|
|
2117
|
+
)
|
|
2118
|
+
unit_id = make_target_id('unit', display_path, include_name)
|
|
2119
|
+
try:
|
|
2120
|
+
result = _parse_navigation_task(
|
|
2121
|
+
_NavigationTask(
|
|
2122
|
+
0,
|
|
2123
|
+
str(source_path),
|
|
2124
|
+
display_path,
|
|
2125
|
+
include_name,
|
|
2126
|
+
display_path,
|
|
2127
|
+
unit_id,
|
|
2128
|
+
False,
|
|
2129
|
+
workspace.defines,
|
|
2130
|
+
workspace.include_paths,
|
|
2131
|
+
workspace.project_config,
|
|
2132
|
+
)
|
|
2133
|
+
)
|
|
2134
|
+
except ParallelOutlineError:
|
|
2135
|
+
continue
|
|
2136
|
+
if result.read_error:
|
|
2137
|
+
continue
|
|
2138
|
+
raw_symbols.extend(
|
|
2139
|
+
_raw_symbol_with_unit_name(raw, include_name)
|
|
2140
|
+
for raw in result.raw_symbols
|
|
2141
|
+
)
|
|
2142
|
+
registry.sources.add_spec(
|
|
2143
|
+
_SourceSpec(
|
|
2144
|
+
source_path,
|
|
2145
|
+
display_path,
|
|
2146
|
+
workspace.defines,
|
|
2147
|
+
workspace.include_paths,
|
|
2148
|
+
workspace.project_config,
|
|
2149
|
+
)
|
|
2150
|
+
)
|
|
2151
|
+
|
|
2152
|
+
if not raw_symbols:
|
|
2153
|
+
return False
|
|
2154
|
+
new_entries = _entries_from_raw_symbols(raw_symbols)
|
|
2155
|
+
registry.entries = tuple(
|
|
2156
|
+
sorted((*registry.entries, *new_entries), key=_entry_sort_key)
|
|
2157
|
+
)
|
|
2158
|
+
registry.by_target.update(
|
|
2159
|
+
(entry.target_id, entry) for entry in new_entries
|
|
2160
|
+
)
|
|
2161
|
+
registry.ranked_queries.clear()
|
|
2162
|
+
registry.static_retained_bytes = _estimate_registry_bytes(
|
|
2163
|
+
registry.entries,
|
|
2164
|
+
registry.sources,
|
|
2165
|
+
)
|
|
2166
|
+
return True
|
|
2167
|
+
|
|
2168
|
+
|
|
2169
|
+
def _raw_symbol_with_unit_name(
|
|
2170
|
+
raw: _RawSymbol,
|
|
2171
|
+
unit_name: str,
|
|
2172
|
+
) -> _RawSymbol:
|
|
2173
|
+
old_unit_name = raw.unit_name
|
|
2174
|
+
|
|
2175
|
+
def rebased(value: str) -> str:
|
|
2176
|
+
if _normalized(value) == _normalized(old_unit_name):
|
|
2177
|
+
return unit_name
|
|
2178
|
+
prefix = f"{old_unit_name}."
|
|
2179
|
+
if value.casefold().startswith(prefix.casefold()):
|
|
2180
|
+
return f"{unit_name}{value[len(old_unit_name):]}"
|
|
2181
|
+
return value
|
|
2182
|
+
|
|
2183
|
+
return replace(
|
|
2184
|
+
raw,
|
|
2185
|
+
name=(
|
|
2186
|
+
unit_name
|
|
2187
|
+
if raw.kind == SymbolKind.UNIT
|
|
2188
|
+
and _normalized(raw.name) == _normalized(old_unit_name)
|
|
2189
|
+
else raw.name
|
|
2190
|
+
),
|
|
2191
|
+
unit_name=unit_name,
|
|
2192
|
+
qualified_name=rebased(raw.qualified_name),
|
|
2193
|
+
owner=rebased(raw.owner),
|
|
2194
|
+
parent_qualified_name=rebased(raw.parent_qualified_name),
|
|
2195
|
+
)
|
|
2196
|
+
|
|
2197
|
+
|
|
2198
|
+
def _has_exact_query_match(
|
|
2199
|
+
entries: Sequence[_SymbolEntry],
|
|
2200
|
+
query: str,
|
|
2201
|
+
) -> bool:
|
|
2202
|
+
needle = unicodedata.normalize('NFC', query.strip()).casefold()
|
|
2203
|
+
return any(
|
|
2204
|
+
entry.normalized_name == needle
|
|
2205
|
+
or entry.normalized_qualified_name == needle
|
|
2206
|
+
for entry in entries
|
|
2207
|
+
)
|
|
2208
|
+
|
|
2209
|
+
|
|
2210
|
+
def _source_may_contain_exact_name(source_path: str, needle: str) -> bool:
|
|
2211
|
+
try:
|
|
2212
|
+
source = read_source_text(Path(source_path))
|
|
2213
|
+
except (OSError, UnicodeError):
|
|
2214
|
+
# Preserve the generic registry's source-unavailable error contract.
|
|
2215
|
+
return True
|
|
2216
|
+
if not needle.isascii():
|
|
2217
|
+
source = unicodedata.normalize('NFC', source)
|
|
2218
|
+
return needle in source.casefold()
|
|
2219
|
+
|
|
2220
|
+
|
|
2221
|
+
def _source_may_include_selected_path(
|
|
2222
|
+
source_path: str,
|
|
2223
|
+
selected_path: str,
|
|
2224
|
+
) -> bool:
|
|
2225
|
+
include_name = Path(selected_path.replace("\\", "/")).name.casefold()
|
|
2226
|
+
if not include_name:
|
|
2227
|
+
return False
|
|
2228
|
+
try:
|
|
2229
|
+
source = read_source_text(Path(source_path))
|
|
2230
|
+
except (OSError, UnicodeError):
|
|
2231
|
+
return True
|
|
2232
|
+
return include_name in source.casefold()
|
|
2233
|
+
|
|
2234
|
+
|
|
2235
|
+
def _source_cache_budget(total_budget_bytes: int | None) -> int:
|
|
2236
|
+
if total_budget_bytes is None:
|
|
2237
|
+
return 64 * 1024**2
|
|
2238
|
+
return max(0, min(128 * 1024**2, total_budget_bytes // 4))
|
|
2239
|
+
|
|
2240
|
+
|
|
2241
|
+
def _estimate_registry_bytes(
|
|
2242
|
+
entries: tuple[_SymbolEntry, ...],
|
|
2243
|
+
sources: _SourceStore,
|
|
2244
|
+
) -> int:
|
|
2245
|
+
retained = 4096 + sources.metadata_bytes + sys.getsizeof(entries)
|
|
2246
|
+
seen_values: set[int] = set()
|
|
2247
|
+
for entry in entries:
|
|
2248
|
+
retained += 384 + sys.getsizeof(entry)
|
|
2249
|
+
for value in (
|
|
2250
|
+
entry.source_path,
|
|
2251
|
+
entry.name,
|
|
2252
|
+
entry.type_name,
|
|
2253
|
+
entry.path,
|
|
2254
|
+
entry.unit_id,
|
|
2255
|
+
entry.unit_name,
|
|
2256
|
+
entry.qualified_name,
|
|
2257
|
+
entry.normalized_name,
|
|
2258
|
+
entry.normalized_qualified_name,
|
|
2259
|
+
entry.owner,
|
|
2260
|
+
entry.signature,
|
|
2261
|
+
entry.target_id,
|
|
2262
|
+
entry.parent_target_id,
|
|
2263
|
+
):
|
|
2264
|
+
identifier = id(value)
|
|
2265
|
+
if identifier in seen_values:
|
|
2266
|
+
continue
|
|
2267
|
+
seen_values.add(identifier)
|
|
2268
|
+
retained += sys.getsizeof(value)
|
|
2269
|
+
retained += len(entries) * 96
|
|
2270
|
+
return retained
|
|
2271
|
+
|
|
2272
|
+
|
|
2273
|
+
def _symbol_card_json_upper_bound(entry: _SymbolEntry) -> int:
|
|
2274
|
+
strings = (
|
|
2275
|
+
entry.target_id,
|
|
2276
|
+
entry.unit_id,
|
|
2277
|
+
entry.name,
|
|
2278
|
+
entry.qualified_name,
|
|
2279
|
+
entry.kind.value,
|
|
2280
|
+
entry.path,
|
|
2281
|
+
entry.visibility.value,
|
|
2282
|
+
entry.owner,
|
|
2283
|
+
entry.type_name,
|
|
2284
|
+
)
|
|
2285
|
+
numeric_chars = len(str(entry.line)) + len(str(entry.column))
|
|
2286
|
+
# JSON string escaping expands one input character to at most six characters.
|
|
2287
|
+
# The fixed allowance covers keys, quotes, separators, brackets, and numbers.
|
|
2288
|
+
ambiguity_chars = (
|
|
2289
|
+
_CONTEXT_AMBIGUOUS_JSON_CHARS if entry.context_ambiguous else 0
|
|
2290
|
+
)
|
|
2291
|
+
return (
|
|
2292
|
+
512
|
|
2293
|
+
+ 6 * sum(len(value) for value in strings)
|
|
2294
|
+
+ numeric_chars
|
|
2295
|
+
+ ambiguity_chars
|
|
2296
|
+
)
|
|
2297
|
+
|
|
2298
|
+
|
|
2299
|
+
_SYMBOL_CARD_JSON_KEYS = (
|
|
2300
|
+
"target_id",
|
|
2301
|
+
"unit_id",
|
|
2302
|
+
"name",
|
|
2303
|
+
"qualified_name",
|
|
2304
|
+
"kind",
|
|
2305
|
+
"path",
|
|
2306
|
+
"line",
|
|
2307
|
+
"column",
|
|
2308
|
+
"visibility",
|
|
2309
|
+
"owner",
|
|
2310
|
+
"type",
|
|
2311
|
+
)
|
|
2312
|
+
_SYMBOL_CARD_JSON_FIXED_CHARS = (
|
|
2313
|
+
2
|
|
2314
|
+
+ len(_SYMBOL_CARD_JSON_KEYS) - 1
|
|
2315
|
+
+ sum(len(key) + 3 for key in _SYMBOL_CARD_JSON_KEYS)
|
|
2316
|
+
)
|
|
2317
|
+
_CONTEXT_AMBIGUOUS_JSON_CHARS = len(',"context_ambiguous":true')
|
|
2318
|
+
|
|
2319
|
+
|
|
2320
|
+
def _symbol_card_json_chars(entry: _SymbolEntry) -> int:
|
|
2321
|
+
strings = (
|
|
2322
|
+
entry.target_id,
|
|
2323
|
+
entry.unit_id,
|
|
2324
|
+
entry.name,
|
|
2325
|
+
entry.qualified_name,
|
|
2326
|
+
entry.kind.value,
|
|
2327
|
+
entry.path,
|
|
2328
|
+
entry.visibility.value,
|
|
2329
|
+
entry.owner,
|
|
2330
|
+
entry.type_name,
|
|
2331
|
+
)
|
|
2332
|
+
return (
|
|
2333
|
+
_SYMBOL_CARD_JSON_FIXED_CHARS
|
|
2334
|
+
+ sum(_json_string_chars(value) for value in strings)
|
|
2335
|
+
+ len(str(entry.line))
|
|
2336
|
+
+ len(str(entry.column))
|
|
2337
|
+
+ (
|
|
2338
|
+
_CONTEXT_AMBIGUOUS_JSON_CHARS
|
|
2339
|
+
if entry.context_ambiguous
|
|
2340
|
+
else 0
|
|
2341
|
+
)
|
|
2342
|
+
)
|
|
2343
|
+
|
|
2344
|
+
|
|
2345
|
+
@lru_cache(maxsize=16_384)
|
|
2346
|
+
def _json_string_chars(value: str) -> int:
|
|
2347
|
+
if (
|
|
2348
|
+
value.isascii()
|
|
2349
|
+
and value.isprintable()
|
|
2350
|
+
and '"' not in value
|
|
2351
|
+
and "\\" not in value
|
|
2352
|
+
):
|
|
2353
|
+
return len(value) + 2
|
|
2354
|
+
return len(json.dumps(value, ensure_ascii=False))
|
|
2355
|
+
|
|
2356
|
+
|
|
2357
|
+
def _stable_path_component(value: str) -> str:
|
|
2358
|
+
normalized = unicodedata.normalize("NFC", value).replace("\\", "_").replace("/", "_")
|
|
2359
|
+
return normalized or "unknown"
|
|
2360
|
+
|
|
2361
|
+
|
|
2362
|
+
def _sanitize_workspace_mapping(
|
|
2363
|
+
value: Mapping[str, object],
|
|
2364
|
+
root: Path,
|
|
2365
|
+
*,
|
|
2366
|
+
path_namespace: str,
|
|
2367
|
+
) -> dict[str, object]:
|
|
2368
|
+
sanitized = dict(value)
|
|
2369
|
+
message = sanitized.get("message")
|
|
2370
|
+
path = sanitized.get("path")
|
|
2371
|
+
if isinstance(path, str):
|
|
2372
|
+
safe_path = _sanitize_workspace_path(path, root, path_namespace)
|
|
2373
|
+
sanitized["path"] = safe_path
|
|
2374
|
+
if isinstance(message, str):
|
|
2375
|
+
message = _replace_path_in_message(message, path, safe_path)
|
|
2376
|
+
origin = sanitized.get("origin")
|
|
2377
|
+
if isinstance(origin, str):
|
|
2378
|
+
safe_origin = _sanitize_workspace_path(origin, root, "origin")
|
|
2379
|
+
sanitized["origin"] = safe_origin
|
|
2380
|
+
if isinstance(message, str):
|
|
2381
|
+
message = _replace_path_in_message(message, origin, safe_origin)
|
|
2382
|
+
origins = sanitized.get("origins")
|
|
2383
|
+
if isinstance(origins, (list, tuple)):
|
|
2384
|
+
sanitized["origins"] = [
|
|
2385
|
+
_sanitize_workspace_path(item, root, "origin") if isinstance(item, str) else item
|
|
2386
|
+
for item in origins
|
|
2387
|
+
]
|
|
2388
|
+
if isinstance(message, str):
|
|
2389
|
+
sanitized["message"] = message
|
|
2390
|
+
return sanitized
|
|
2391
|
+
|
|
2392
|
+
|
|
2393
|
+
def _replace_path_in_message(message: str, path: str, replacement: str) -> str:
|
|
2394
|
+
variants = {
|
|
2395
|
+
path,
|
|
2396
|
+
path.replace("\\", "/"),
|
|
2397
|
+
path.replace("/", "\\"),
|
|
2398
|
+
}
|
|
2399
|
+
for variant in sorted(variants, key=len, reverse=True):
|
|
2400
|
+
message = message.replace(variant, replacement)
|
|
2401
|
+
return message
|
|
2402
|
+
|
|
2403
|
+
|
|
2404
|
+
def _sanitize_workspace_path(value: str, root: Path, namespace: str) -> str:
|
|
2405
|
+
normalized = unicodedata.normalize("NFC", value).replace("\\", "/")
|
|
2406
|
+
native_path = Path(value).expanduser()
|
|
2407
|
+
if native_path.is_absolute():
|
|
2408
|
+
resolved = native_path.resolve()
|
|
2409
|
+
try:
|
|
2410
|
+
return resolved.relative_to(root).as_posix()
|
|
2411
|
+
except ValueError:
|
|
2412
|
+
component = resolved.name
|
|
2413
|
+
else:
|
|
2414
|
+
windows_path = PureWindowsPath(value)
|
|
2415
|
+
if not windows_path.is_absolute():
|
|
2416
|
+
return normalized
|
|
2417
|
+
component = windows_path.name or windows_path.drive.rstrip(":\\/")
|
|
2418
|
+
return f"@external/{namespace}/{_stable_path_component(component)}"
|
|
2419
|
+
|
|
2420
|
+
|
|
2421
|
+
def _target_identity_name(raw: _RawSymbol) -> str:
|
|
2422
|
+
if raw.kind in _ROUTINE_KINDS:
|
|
2423
|
+
return f"{raw.qualified_name}\x1f{_normalized(raw.signature)}"
|
|
2424
|
+
return raw.qualified_name
|
|
2425
|
+
|
|
2426
|
+
|
|
2427
|
+
def _collect_raw_symbols(
|
|
2428
|
+
scope: Scope,
|
|
2429
|
+
unit: AgentUnit,
|
|
2430
|
+
source_path: Path,
|
|
2431
|
+
document: _SourceDocument,
|
|
2432
|
+
) -> list[_RawSymbol]:
|
|
2433
|
+
unit_name = scope.name or unit.name
|
|
2434
|
+
unit_id = make_target_id("unit", document.display_path, unit.name)
|
|
2435
|
+
collected: list[_RawSymbol] = []
|
|
2436
|
+
symbols = [symbol for group in scope.symbols.values() for symbol in group]
|
|
2437
|
+
symbols.sort(key=_symbol_sort_key)
|
|
2438
|
+
for symbol in symbols:
|
|
2439
|
+
if symbol.scope.kind != ScopeKind.UNIT:
|
|
2440
|
+
continue
|
|
2441
|
+
_correct_outline_symbol_kind(document, symbol)
|
|
2442
|
+
if symbol.kind == SymbolKind.UNIT:
|
|
2443
|
+
qualified_name = unit_name
|
|
2444
|
+
owner = ""
|
|
2445
|
+
else:
|
|
2446
|
+
declared_name = _declared_symbol_name(document, symbol).strip(".")
|
|
2447
|
+
qualified_name = f"{unit_name}.{declared_name}"
|
|
2448
|
+
owner = qualified_name.rsplit(".", 1)[0]
|
|
2449
|
+
collected.append(
|
|
2450
|
+
_RawSymbol(
|
|
2451
|
+
name=symbol.name,
|
|
2452
|
+
kind=symbol.kind,
|
|
2453
|
+
line=symbol.decl_range.start_line,
|
|
2454
|
+
column=symbol.decl_range.start_col,
|
|
2455
|
+
visibility=symbol.visibility,
|
|
2456
|
+
type_name=symbol.type_ref.display_name(),
|
|
2457
|
+
source_path=source_path,
|
|
2458
|
+
path=document.display_path,
|
|
2459
|
+
unit_id=unit_id,
|
|
2460
|
+
unit_name=unit_name,
|
|
2461
|
+
qualified_name=qualified_name,
|
|
2462
|
+
owner=owner,
|
|
2463
|
+
parent_qualified_name="",
|
|
2464
|
+
signature=_symbol_signature(document, symbol),
|
|
2465
|
+
context_ambiguous=(
|
|
2466
|
+
symbol.attributes.get("context_ambiguous") == "true"
|
|
2467
|
+
),
|
|
2468
|
+
)
|
|
2469
|
+
)
|
|
2470
|
+
member_scope = symbol.member_scope
|
|
2471
|
+
if member_scope is None or member_scope.kind != ScopeKind.TYPE:
|
|
2472
|
+
continue
|
|
2473
|
+
members = [member for group in member_scope.symbols.values() for member in group]
|
|
2474
|
+
members.sort(key=_symbol_sort_key)
|
|
2475
|
+
for member in members:
|
|
2476
|
+
member_name = _declared_symbol_name(document, member).strip(".")
|
|
2477
|
+
member_qualified_name = f"{qualified_name}.{member_name}"
|
|
2478
|
+
collected.append(
|
|
2479
|
+
_RawSymbol(
|
|
2480
|
+
name=member.name,
|
|
2481
|
+
kind=member.kind,
|
|
2482
|
+
line=member.decl_range.start_line,
|
|
2483
|
+
column=member.decl_range.start_col,
|
|
2484
|
+
visibility=member.visibility,
|
|
2485
|
+
type_name=member.type_ref.display_name(),
|
|
2486
|
+
source_path=source_path,
|
|
2487
|
+
path=document.display_path,
|
|
2488
|
+
unit_id=unit_id,
|
|
2489
|
+
unit_name=unit_name,
|
|
2490
|
+
qualified_name=member_qualified_name,
|
|
2491
|
+
owner=qualified_name,
|
|
2492
|
+
parent_qualified_name=qualified_name,
|
|
2493
|
+
signature=_symbol_signature(document, member),
|
|
2494
|
+
context_ambiguous=(
|
|
2495
|
+
member.attributes.get("context_ambiguous") == "true"
|
|
2496
|
+
),
|
|
2497
|
+
)
|
|
2498
|
+
)
|
|
2499
|
+
return collected
|
|
2500
|
+
|
|
2501
|
+
|
|
2502
|
+
def _correct_outline_symbol_kind(document: _SourceDocument, symbol: Symbol) -> None:
|
|
2503
|
+
if symbol.kind != SymbolKind.CONSTANT:
|
|
2504
|
+
return
|
|
2505
|
+
start = _declaration_start(document, symbol.decl_range.start_line)
|
|
2506
|
+
if _declaration_section(document, start) != "type":
|
|
2507
|
+
return
|
|
2508
|
+
token_index = document.first_token_index(start)
|
|
2509
|
+
equals_index = _next_token_value(document.tokens, token_index, "=")
|
|
2510
|
+
if equals_index is None or equals_index + 1 >= len(document.tokens):
|
|
2511
|
+
return
|
|
2512
|
+
rhs = document.tokens[equals_index + 1:]
|
|
2513
|
+
index = 0
|
|
2514
|
+
if rhs and rhs[0].value == "packed":
|
|
2515
|
+
index = 1
|
|
2516
|
+
if index >= len(rhs):
|
|
2517
|
+
return
|
|
2518
|
+
value = rhs[index].value
|
|
2519
|
+
if value == "class" and not (
|
|
2520
|
+
index + 1 < len(rhs) and rhs[index + 1].value == "of"
|
|
2521
|
+
):
|
|
2522
|
+
symbol.kind = SymbolKind.CLASS
|
|
2523
|
+
elif value == "record":
|
|
2524
|
+
symbol.kind = SymbolKind.RECORD
|
|
2525
|
+
elif value == "interface":
|
|
2526
|
+
symbol.kind = SymbolKind.INTERFACE
|
|
2527
|
+
elif value == "(":
|
|
2528
|
+
symbol.kind = SymbolKind.ENUM
|
|
2529
|
+
else:
|
|
2530
|
+
symbol.kind = SymbolKind.TYPE
|
|
2531
|
+
|
|
2532
|
+
|
|
2533
|
+
def _advance_declaration_section(
|
|
2534
|
+
state: tuple[str, int, int, int],
|
|
2535
|
+
token: _Token,
|
|
2536
|
+
) -> tuple[str, int, int, int]:
|
|
2537
|
+
section, parentheses, brackets, angles = state
|
|
2538
|
+
if token.directive:
|
|
2539
|
+
return state
|
|
2540
|
+
if token.value == "(":
|
|
2541
|
+
parentheses += 1
|
|
2542
|
+
elif token.value == ")":
|
|
2543
|
+
parentheses = max(0, parentheses - 1)
|
|
2544
|
+
elif token.value == "[":
|
|
2545
|
+
brackets += 1
|
|
2546
|
+
elif token.value == "]":
|
|
2547
|
+
brackets = max(0, brackets - 1)
|
|
2548
|
+
elif token.value == "<":
|
|
2549
|
+
angles += 1
|
|
2550
|
+
elif token.value == ">":
|
|
2551
|
+
angles = max(0, angles - 1)
|
|
2552
|
+
elif not parentheses and not brackets and not angles and token.word:
|
|
2553
|
+
if token.value in {"const", "resourcestring", "threadvar", "type", "var"}:
|
|
2554
|
+
section = token.value
|
|
2555
|
+
elif token.value in {"implementation", "initialization", "finalization"}:
|
|
2556
|
+
section = ""
|
|
2557
|
+
return section, parentheses, brackets, angles
|
|
2558
|
+
|
|
2559
|
+
|
|
2560
|
+
def _declaration_section(document: _SourceDocument, offset: int) -> str:
|
|
2561
|
+
target_index = document.first_token_index(offset)
|
|
2562
|
+
cached = document.declaration_section_checkpoints.get(target_index)
|
|
2563
|
+
if cached is not None:
|
|
2564
|
+
return cached[0]
|
|
2565
|
+
checkpoint_position = bisect_right(document.declaration_section_indexes, target_index) - 1
|
|
2566
|
+
checkpoint_index = document.declaration_section_indexes[checkpoint_position]
|
|
2567
|
+
state = document.declaration_section_checkpoints[checkpoint_index]
|
|
2568
|
+
for token in document.tokens[checkpoint_index:target_index]:
|
|
2569
|
+
state = _advance_declaration_section(state, token)
|
|
2570
|
+
insert_at = bisect_left(document.declaration_section_indexes, target_index)
|
|
2571
|
+
document.declaration_section_indexes.insert(insert_at, target_index)
|
|
2572
|
+
document.declaration_section_checkpoints[target_index] = state
|
|
2573
|
+
return state[0]
|
|
2574
|
+
|
|
2575
|
+
|
|
2576
|
+
def _declared_symbol_name(document: _SourceDocument, symbol: Symbol) -> str:
|
|
2577
|
+
if symbol.kind in _ROUTINE_KINDS:
|
|
2578
|
+
return _routine_declared_name(document, symbol) or symbol.name
|
|
2579
|
+
if symbol.kind in _TYPE_KINDS:
|
|
2580
|
+
return _type_declared_name(document, symbol) or symbol.name
|
|
2581
|
+
return symbol.name
|
|
2582
|
+
|
|
2583
|
+
|
|
2584
|
+
def _routine_declared_name(document: _SourceDocument, symbol: Symbol) -> str:
|
|
2585
|
+
start = _declaration_start(document, symbol.decl_range.start_line)
|
|
2586
|
+
token_index = document.first_token_index(start)
|
|
2587
|
+
routine_index = _routine_keyword_index(document.tokens, token_index)
|
|
2588
|
+
if routine_index is None:
|
|
2589
|
+
return ""
|
|
2590
|
+
heading_end = _heading_semicolon_index(document.tokens, routine_index)
|
|
2591
|
+
if heading_end is None:
|
|
2592
|
+
return ""
|
|
2593
|
+
name_span = _routine_name_token_span(document.tokens, routine_index, heading_end)
|
|
2594
|
+
if name_span is None:
|
|
2595
|
+
return ""
|
|
2596
|
+
return _join_source_tokens(document, document.tokens[name_span[0]:name_span[1]])
|
|
2597
|
+
|
|
2598
|
+
|
|
2599
|
+
def _type_declared_name(document: _SourceDocument, symbol: Symbol) -> str:
|
|
2600
|
+
start = _declaration_start(document, symbol.decl_range.start_line)
|
|
2601
|
+
token_index = document.first_token_index(start)
|
|
2602
|
+
equals_index = next(
|
|
2603
|
+
(
|
|
2604
|
+
index
|
|
2605
|
+
for index in range(token_index, len(document.tokens))
|
|
2606
|
+
if document.tokens[index].value in {"=", ";"}
|
|
2607
|
+
),
|
|
2608
|
+
None,
|
|
2609
|
+
)
|
|
2610
|
+
if equals_index is None or document.tokens[equals_index].value != "=":
|
|
2611
|
+
return ""
|
|
2612
|
+
name_index = next(
|
|
2613
|
+
(
|
|
2614
|
+
index
|
|
2615
|
+
for index in range(token_index, equals_index)
|
|
2616
|
+
if document.tokens[index].word
|
|
2617
|
+
and _normalized(document.tokens[index].value) == _normalized(symbol.name)
|
|
2618
|
+
),
|
|
2619
|
+
token_index,
|
|
2620
|
+
)
|
|
2621
|
+
return _join_source_tokens(document, document.tokens[name_index:equals_index])
|
|
2622
|
+
|
|
2623
|
+
|
|
2624
|
+
def _join_source_tokens(document: _SourceDocument, tokens: tuple[_Token, ...]) -> str:
|
|
2625
|
+
return unicodedata.normalize(
|
|
2626
|
+
"NFC",
|
|
2627
|
+
"".join(document.text[token.start:token.end] for token in tokens),
|
|
2628
|
+
)
|
|
2629
|
+
|
|
2630
|
+
|
|
2631
|
+
def _symbol_signature(document: _SourceDocument, symbol: Symbol) -> str:
|
|
2632
|
+
if symbol.kind not in _ROUTINE_KINDS:
|
|
2633
|
+
return ""
|
|
2634
|
+
start = _declaration_start(document, symbol.decl_range.start_line)
|
|
2635
|
+
start_index = document.first_token_index(start)
|
|
2636
|
+
routine_index = _routine_keyword_index(document.tokens, start_index)
|
|
2637
|
+
if routine_index is None:
|
|
2638
|
+
return ""
|
|
2639
|
+
heading_end = _heading_semicolon_index(document.tokens, routine_index)
|
|
2640
|
+
if heading_end is None:
|
|
2641
|
+
return ""
|
|
2642
|
+
name_span = _routine_name_token_span(document.tokens, routine_index, heading_end)
|
|
2643
|
+
if name_span is None:
|
|
2644
|
+
return ""
|
|
2645
|
+
signature_tokens = document.tokens[name_span[1]:heading_end]
|
|
2646
|
+
declaration_end = _routine_declaration_end_index(document.tokens, routine_index)
|
|
2647
|
+
calling_conventions = ""
|
|
2648
|
+
if declaration_end is not None:
|
|
2649
|
+
conventions = sorted(
|
|
2650
|
+
{
|
|
2651
|
+
token.value
|
|
2652
|
+
for token in document.tokens[heading_end + 1:declaration_end]
|
|
2653
|
+
if token.word and not token.escaped and token.value in _CALLING_CONVENTIONS
|
|
2654
|
+
}
|
|
2655
|
+
)
|
|
2656
|
+
if conventions:
|
|
2657
|
+
calling_conventions = f"|cc:{','.join(conventions)}"
|
|
2658
|
+
return _normalized(f"{_normalize_routine_signature(signature_tokens)}{calling_conventions}")
|
|
2659
|
+
|
|
2660
|
+
|
|
2661
|
+
def _routine_name_token_span(
|
|
2662
|
+
tokens: tuple[_Token, ...],
|
|
2663
|
+
routine_index: int,
|
|
2664
|
+
heading_end: int,
|
|
2665
|
+
) -> tuple[int, int] | None:
|
|
2666
|
+
start = routine_index + 1
|
|
2667
|
+
if start >= heading_end:
|
|
2668
|
+
return None
|
|
2669
|
+
end = start + 1
|
|
2670
|
+
while end < heading_end:
|
|
2671
|
+
if (
|
|
2672
|
+
tokens[end].value == "."
|
|
2673
|
+
and end + 1 < heading_end
|
|
2674
|
+
and (tokens[end + 1].word or tokens[routine_index].value == "operator")
|
|
2675
|
+
):
|
|
2676
|
+
end += 2
|
|
2677
|
+
continue
|
|
2678
|
+
if tokens[end].value == "<":
|
|
2679
|
+
generic_end = _matching_token_index(tokens, end, "<", ">")
|
|
2680
|
+
if (
|
|
2681
|
+
generic_end is not None
|
|
2682
|
+
and generic_end + 1 < heading_end
|
|
2683
|
+
and tokens[generic_end + 1].value == "."
|
|
2684
|
+
):
|
|
2685
|
+
end = generic_end + 1
|
|
2686
|
+
continue
|
|
2687
|
+
break
|
|
2688
|
+
return start, end
|
|
2689
|
+
|
|
2690
|
+
|
|
2691
|
+
def _normalize_routine_signature(tokens: tuple[_Token, ...]) -> str:
|
|
2692
|
+
open_index = next((index for index, token in enumerate(tokens) if token.value == "("), None)
|
|
2693
|
+
if open_index is None:
|
|
2694
|
+
colon = _top_level_token_index(tokens, ":")
|
|
2695
|
+
if colon is None:
|
|
2696
|
+
return f"{_join_token_values(tokens)}()"
|
|
2697
|
+
generic = _join_token_values(tokens[:colon])
|
|
2698
|
+
return f"{generic}():{_normalize_type_tokens(tokens[colon + 1:])}"
|
|
2699
|
+
|
|
2700
|
+
close_index = _matching_token_index(tokens, open_index, "(", ")")
|
|
2701
|
+
if close_index is None:
|
|
2702
|
+
return _join_token_values(tokens)
|
|
2703
|
+
generic = _join_token_values(tokens[:open_index])
|
|
2704
|
+
parameters = _normalize_parameters(tokens[open_index + 1:close_index])
|
|
2705
|
+
result_type = _normalize_result_type(tokens[close_index + 1:])
|
|
2706
|
+
return f"{generic}({parameters}){result_type}"
|
|
2707
|
+
|
|
2708
|
+
|
|
2709
|
+
def _normalize_parameters(tokens: tuple[_Token, ...]) -> str:
|
|
2710
|
+
groups = _split_tokens_at_top_level(tokens, ";")
|
|
2711
|
+
normalized: list[str] = []
|
|
2712
|
+
modes = {"const", "constref", "out", "var"}
|
|
2713
|
+
for group in groups:
|
|
2714
|
+
if not group:
|
|
2715
|
+
continue
|
|
2716
|
+
colon = _top_level_token_index(group, ":")
|
|
2717
|
+
if colon is None:
|
|
2718
|
+
normalized.append(_join_token_values(group))
|
|
2719
|
+
continue
|
|
2720
|
+
names = list(group[:colon])
|
|
2721
|
+
mode = ""
|
|
2722
|
+
if names and names[0].word and names[0].value in modes:
|
|
2723
|
+
mode = names.pop(0).value
|
|
2724
|
+
parameter_count = 1 + sum(token.value == "," for token in names)
|
|
2725
|
+
type_tokens = group[colon + 1:]
|
|
2726
|
+
default = _top_level_token_index(type_tokens, "=")
|
|
2727
|
+
if default is not None:
|
|
2728
|
+
type_tokens = type_tokens[:default]
|
|
2729
|
+
normalized.append(f"{mode}#{parameter_count}:{_normalize_type_tokens(type_tokens)}")
|
|
2730
|
+
return ";".join(normalized)
|
|
2731
|
+
|
|
2732
|
+
|
|
2733
|
+
def _normalize_result_type(tokens: tuple[_Token, ...]) -> str:
|
|
2734
|
+
if not tokens or tokens[0].value != ":":
|
|
2735
|
+
return ""
|
|
2736
|
+
return f":{_normalize_type_tokens(tokens[1:])}"
|
|
2737
|
+
|
|
2738
|
+
|
|
2739
|
+
def _normalize_type_tokens(tokens: tuple[_Token, ...]) -> str:
|
|
2740
|
+
parts: list[str] = []
|
|
2741
|
+
index = 0
|
|
2742
|
+
while index < len(tokens):
|
|
2743
|
+
token = tokens[index]
|
|
2744
|
+
parts.append(_normalized(token.value))
|
|
2745
|
+
if (
|
|
2746
|
+
token.word
|
|
2747
|
+
and token.value in {"procedure", "function"}
|
|
2748
|
+
and index + 1 < len(tokens)
|
|
2749
|
+
and tokens[index + 1].value == "("
|
|
2750
|
+
):
|
|
2751
|
+
close = _matching_token_index(tokens, index + 1, "(", ")")
|
|
2752
|
+
if close is None:
|
|
2753
|
+
index += 1
|
|
2754
|
+
continue
|
|
2755
|
+
parts.append("(")
|
|
2756
|
+
parts.append(_normalize_parameters(tokens[index + 2:close]))
|
|
2757
|
+
parts.append(")")
|
|
2758
|
+
index = close + 1
|
|
2759
|
+
continue
|
|
2760
|
+
index += 1
|
|
2761
|
+
return unicodedata.normalize("NFC", "".join(parts)).casefold()
|
|
2762
|
+
|
|
2763
|
+
|
|
2764
|
+
def _split_tokens_at_top_level(
|
|
2765
|
+
tokens: tuple[_Token, ...],
|
|
2766
|
+
separator: str,
|
|
2767
|
+
) -> list[tuple[_Token, ...]]:
|
|
2768
|
+
groups: list[tuple[_Token, ...]] = []
|
|
2769
|
+
start = 0
|
|
2770
|
+
depths = {"(": 0, "[": 0, "<": 0}
|
|
2771
|
+
closing = {")": "(", "]": "[", ">": "<"}
|
|
2772
|
+
for index, token in enumerate(tokens):
|
|
2773
|
+
if token.value in depths:
|
|
2774
|
+
depths[token.value] += 1
|
|
2775
|
+
elif token.value in closing:
|
|
2776
|
+
opener = closing[token.value]
|
|
2777
|
+
depths[opener] = max(0, depths[opener] - 1)
|
|
2778
|
+
elif token.value == separator and not any(depths.values()):
|
|
2779
|
+
groups.append(tokens[start:index])
|
|
2780
|
+
start = index + 1
|
|
2781
|
+
groups.append(tokens[start:])
|
|
2782
|
+
return groups
|
|
2783
|
+
|
|
2784
|
+
|
|
2785
|
+
def _top_level_token_index(tokens: tuple[_Token, ...], value: str) -> int | None:
|
|
2786
|
+
depths = {"(": 0, "[": 0, "<": 0}
|
|
2787
|
+
closing = {")": "(", "]": "[", ">": "<"}
|
|
2788
|
+
for index, token in enumerate(tokens):
|
|
2789
|
+
if token.value in depths:
|
|
2790
|
+
depths[token.value] += 1
|
|
2791
|
+
elif token.value in closing:
|
|
2792
|
+
opener = closing[token.value]
|
|
2793
|
+
depths[opener] = max(0, depths[opener] - 1)
|
|
2794
|
+
elif token.value == value and not any(depths.values()):
|
|
2795
|
+
return index
|
|
2796
|
+
return None
|
|
2797
|
+
|
|
2798
|
+
|
|
2799
|
+
def _matching_token_index(
|
|
2800
|
+
tokens: tuple[_Token, ...],
|
|
2801
|
+
start: int,
|
|
2802
|
+
opener: str,
|
|
2803
|
+
closer: str,
|
|
2804
|
+
) -> int | None:
|
|
2805
|
+
depth = 0
|
|
2806
|
+
for index in range(start, len(tokens)):
|
|
2807
|
+
if tokens[index].value == opener:
|
|
2808
|
+
depth += 1
|
|
2809
|
+
elif tokens[index].value == closer:
|
|
2810
|
+
depth -= 1
|
|
2811
|
+
if depth == 0:
|
|
2812
|
+
return index
|
|
2813
|
+
return None
|
|
2814
|
+
|
|
2815
|
+
|
|
2816
|
+
def _join_token_values(tokens: tuple[_Token, ...]) -> str:
|
|
2817
|
+
return unicodedata.normalize("NFC", "".join(token.value for token in tokens)).casefold()
|
|
2818
|
+
|
|
2819
|
+
|
|
2820
|
+
def _exclude_routine_locals(
|
|
2821
|
+
symbols: list[_RawSymbol],
|
|
2822
|
+
document: _SourceDocument,
|
|
2823
|
+
) -> list[_RawSymbol]:
|
|
2824
|
+
containers: list[tuple[int, int, _RawSymbol]] = []
|
|
2825
|
+
for raw in symbols:
|
|
2826
|
+
if raw.parent_qualified_name or raw.kind not in _ROUTINE_KINDS:
|
|
2827
|
+
continue
|
|
2828
|
+
span = _raw_routine_span(raw, document)
|
|
2829
|
+
if span is not None:
|
|
2830
|
+
containers.append((span[0], span[1], raw))
|
|
2831
|
+
if not containers:
|
|
2832
|
+
return symbols
|
|
2833
|
+
|
|
2834
|
+
containers.sort(key=lambda item: (item[0], item[1]))
|
|
2835
|
+
positioned = sorted(
|
|
2836
|
+
(
|
|
2837
|
+
document.offset(
|
|
2838
|
+
raw.line,
|
|
2839
|
+
raw.column,
|
|
2840
|
+
),
|
|
2841
|
+
order,
|
|
2842
|
+
raw,
|
|
2843
|
+
)
|
|
2844
|
+
for order, raw in enumerate(symbols)
|
|
2845
|
+
)
|
|
2846
|
+
active_ends: list[tuple[int, int, int]] = []
|
|
2847
|
+
active_ids: set[int] = set()
|
|
2848
|
+
excluded_orders: set[int] = set()
|
|
2849
|
+
container_index = 0
|
|
2850
|
+
for offset, order, raw in positioned:
|
|
2851
|
+
while (
|
|
2852
|
+
container_index < len(containers)
|
|
2853
|
+
and containers[container_index][0] < offset
|
|
2854
|
+
):
|
|
2855
|
+
_, end, container = containers[container_index]
|
|
2856
|
+
container_id = id(container)
|
|
2857
|
+
heappush(active_ends, (end, container_index, container_id))
|
|
2858
|
+
active_ids.add(container_id)
|
|
2859
|
+
container_index += 1
|
|
2860
|
+
while active_ends and active_ends[0][0] <= offset:
|
|
2861
|
+
_, _, container_id = heappop(active_ends)
|
|
2862
|
+
active_ids.discard(container_id)
|
|
2863
|
+
raw_id = id(raw)
|
|
2864
|
+
if active_ids and (raw_id not in active_ids or len(active_ids) > 1):
|
|
2865
|
+
excluded_orders.add(order)
|
|
2866
|
+
|
|
2867
|
+
return [
|
|
2868
|
+
raw
|
|
2869
|
+
for order, raw in enumerate(symbols)
|
|
2870
|
+
if order not in excluded_orders
|
|
2871
|
+
]
|
|
2872
|
+
|
|
2873
|
+
|
|
2874
|
+
def _raw_routine_span(
|
|
2875
|
+
raw: _RawSymbol,
|
|
2876
|
+
document: _SourceDocument,
|
|
2877
|
+
) -> tuple[int, int] | None:
|
|
2878
|
+
if raw.kind not in _ROUTINE_KINDS or raw.parent_qualified_name:
|
|
2879
|
+
return None
|
|
2880
|
+
line = raw.line
|
|
2881
|
+
if document.unit_kind == "unit" and (
|
|
2882
|
+
not document.implementation_line or line < document.implementation_line
|
|
2883
|
+
):
|
|
2884
|
+
return None
|
|
2885
|
+
start = _declaration_start(document, line)
|
|
2886
|
+
return _routine_span(document, start)
|
|
2887
|
+
|
|
2888
|
+
|
|
2889
|
+
def _body_entry_and_span(
|
|
2890
|
+
registry: _Registry,
|
|
2891
|
+
entry: _SymbolEntry,
|
|
2892
|
+
) -> tuple[_SymbolEntry, tuple[int, int] | None]:
|
|
2893
|
+
candidates = [entry]
|
|
2894
|
+
if entry.kind in _ROUTINE_KINDS:
|
|
2895
|
+
candidates.extend(_matching_counterparts(registry, entry))
|
|
2896
|
+
for candidate in candidates:
|
|
2897
|
+
document = registry.sources[candidate.source_path]
|
|
2898
|
+
span = _entry_body_span(candidate, document)
|
|
2899
|
+
if span is not None:
|
|
2900
|
+
return candidate, span
|
|
2901
|
+
return entry, None
|
|
2902
|
+
|
|
2903
|
+
|
|
2904
|
+
def _matching_counterparts(
|
|
2905
|
+
registry: _Registry,
|
|
2906
|
+
entry: _SymbolEntry,
|
|
2907
|
+
) -> list[_SymbolEntry]:
|
|
2908
|
+
return [
|
|
2909
|
+
candidate
|
|
2910
|
+
for candidate in registry.entries
|
|
2911
|
+
if candidate.target_id != entry.target_id
|
|
2912
|
+
and candidate.kind == entry.kind
|
|
2913
|
+
and _normalized(candidate.qualified_name) == _normalized(entry.qualified_name)
|
|
2914
|
+
and candidate.signature == entry.signature
|
|
2915
|
+
]
|
|
2916
|
+
|
|
2917
|
+
|
|
2918
|
+
def _entry_body_span(
|
|
2919
|
+
entry: _SymbolEntry,
|
|
2920
|
+
document: _SourceDocument,
|
|
2921
|
+
) -> tuple[int, int] | None:
|
|
2922
|
+
if entry.kind in _TYPE_KINDS:
|
|
2923
|
+
if _is_forward_type(document, entry):
|
|
2924
|
+
return None
|
|
2925
|
+
span = _type_span(document, entry)
|
|
2926
|
+
if span is not None and not document.contains_directive(*span):
|
|
2927
|
+
return span
|
|
2928
|
+
return _full_parser_span(document, entry)
|
|
2929
|
+
if entry.kind not in _ROUTINE_KINDS or entry.parent_target_id:
|
|
2930
|
+
return None
|
|
2931
|
+
line = entry.line
|
|
2932
|
+
if document.unit_kind == "unit" and (
|
|
2933
|
+
not document.implementation_line or line < document.implementation_line
|
|
2934
|
+
):
|
|
2935
|
+
return None
|
|
2936
|
+
start = _declaration_start(document, line)
|
|
2937
|
+
span = _routine_span(document, start)
|
|
2938
|
+
if span is not None and not document.contains_directive(*span):
|
|
2939
|
+
return span
|
|
2940
|
+
return _full_parser_span(document, entry)
|
|
2941
|
+
|
|
2942
|
+
|
|
2943
|
+
def _full_parser_span(
|
|
2944
|
+
document: _SourceDocument,
|
|
2945
|
+
entry: _SymbolEntry,
|
|
2946
|
+
) -> tuple[int, int] | None:
|
|
2947
|
+
if entry.target_id in document.parser_spans:
|
|
2948
|
+
return document.parser_spans[entry.target_id]
|
|
2949
|
+
result = document.full_parse()
|
|
2950
|
+
if result is None:
|
|
2951
|
+
document.parser_spans[entry.target_id] = None
|
|
2952
|
+
return None
|
|
2953
|
+
|
|
2954
|
+
expected_type = (
|
|
2955
|
+
SyntaxNodeType.ntMethod
|
|
2956
|
+
if entry.kind in _ROUTINE_KINDS
|
|
2957
|
+
else SyntaxNodeType.ntTypeDecl
|
|
2958
|
+
)
|
|
2959
|
+
expected_line = entry.line
|
|
2960
|
+
candidates: list[tuple[int, int]] = []
|
|
2961
|
+
for node in _walk_syntax_nodes(result.root):
|
|
2962
|
+
if node.typ != expected_type or not isinstance(node, CompoundSyntaxNode):
|
|
2963
|
+
continue
|
|
2964
|
+
mapped = _mapped_tree_span(document, result, node)
|
|
2965
|
+
if mapped is None or document.line_col(mapped[0])[0] != expected_line:
|
|
2966
|
+
continue
|
|
2967
|
+
if expected_type == SyntaxNodeType.ntMethod and not _syntax_method_has_body(node):
|
|
2968
|
+
continue
|
|
2969
|
+
candidates.append(mapped)
|
|
2970
|
+
|
|
2971
|
+
span = candidates[0] if len(candidates) == 1 else None
|
|
2972
|
+
document.parser_spans[entry.target_id] = span
|
|
2973
|
+
return span
|
|
2974
|
+
|
|
2975
|
+
|
|
2976
|
+
def _walk_syntax_nodes(root: SyntaxNode):
|
|
2977
|
+
stack = [root]
|
|
2978
|
+
while stack:
|
|
2979
|
+
node = stack.pop()
|
|
2980
|
+
yield node
|
|
2981
|
+
stack.extend(reversed(node.child_nodes))
|
|
2982
|
+
|
|
2983
|
+
|
|
2984
|
+
def _syntax_method_has_body(node: SyntaxNode) -> bool:
|
|
2985
|
+
if any(
|
|
2986
|
+
node.has_attribute(attribute)
|
|
2987
|
+
for attribute in (
|
|
2988
|
+
AttributeName.anAbstract,
|
|
2989
|
+
AttributeName.anExternal,
|
|
2990
|
+
AttributeName.anForwarded,
|
|
2991
|
+
)
|
|
2992
|
+
):
|
|
2993
|
+
return False
|
|
2994
|
+
return any(
|
|
2995
|
+
candidate.typ == SyntaxNodeType.ntStatements
|
|
2996
|
+
for candidate in _walk_syntax_nodes(node)
|
|
2997
|
+
if candidate is not node
|
|
2998
|
+
)
|
|
2999
|
+
|
|
3000
|
+
|
|
3001
|
+
def _mapped_tree_span(document: _SourceDocument, result, node: SyntaxNode) -> tuple[int, int] | None:
|
|
3002
|
+
end_line, end_col = _syntax_tree_end(node)
|
|
3003
|
+
start_file, start_line, _ = result.preprocessed.map_position(node.line, node.col)
|
|
3004
|
+
end_file, mapped_end_line, mapped_end_col = result.preprocessed.map_position(end_line, end_col)
|
|
3005
|
+
if not start_file or not end_file:
|
|
3006
|
+
return None
|
|
3007
|
+
try:
|
|
3008
|
+
start_path = Path(start_file).expanduser().resolve()
|
|
3009
|
+
end_path = Path(end_file).expanduser().resolve()
|
|
3010
|
+
except OSError:
|
|
3011
|
+
return None
|
|
3012
|
+
if start_path != document.source_path or end_path != document.source_path:
|
|
3013
|
+
return None
|
|
3014
|
+
|
|
3015
|
+
previous_line = start_line - 1
|
|
3016
|
+
for preprocessed_line in range(node.line, end_line + 1):
|
|
3017
|
+
mapped_file, mapped_line, _ = result.preprocessed.map_position(preprocessed_line, 1)
|
|
3018
|
+
if not mapped_file or Path(mapped_file).expanduser().resolve() != document.source_path:
|
|
3019
|
+
return None
|
|
3020
|
+
if mapped_line != previous_line + 1:
|
|
3021
|
+
return None
|
|
3022
|
+
previous_line = mapped_line
|
|
3023
|
+
|
|
3024
|
+
start = _declaration_start(document, start_line)
|
|
3025
|
+
end = document.offset(mapped_end_line, mapped_end_col)
|
|
3026
|
+
if end < len(document.text) and not document.text[end].isspace():
|
|
3027
|
+
end += 1
|
|
3028
|
+
if end <= start:
|
|
3029
|
+
return None
|
|
3030
|
+
return start, end
|
|
3031
|
+
|
|
3032
|
+
|
|
3033
|
+
def _syntax_tree_end(node: SyntaxNode) -> tuple[int, int]:
|
|
3034
|
+
end = (
|
|
3035
|
+
(node.end_line, node.end_col)
|
|
3036
|
+
if isinstance(node, CompoundSyntaxNode)
|
|
3037
|
+
else (node.line, node.col)
|
|
3038
|
+
)
|
|
3039
|
+
for child in node.child_nodes:
|
|
3040
|
+
end = max(end, _syntax_tree_end(child))
|
|
3041
|
+
return end
|
|
3042
|
+
|
|
3043
|
+
|
|
3044
|
+
def _declaration_span(
|
|
3045
|
+
document: _SourceDocument,
|
|
3046
|
+
entry: _SymbolEntry,
|
|
3047
|
+
) -> tuple[int, int]:
|
|
3048
|
+
line = entry.line
|
|
3049
|
+
start = _declaration_start(document, line)
|
|
3050
|
+
if entry.kind in _TYPE_KINDS:
|
|
3051
|
+
full_span, direct_structured = _type_declaration_layout(document, entry)
|
|
3052
|
+
if direct_structured:
|
|
3053
|
+
return start, document.line_end(line)
|
|
3054
|
+
if full_span is not None:
|
|
3055
|
+
return full_span
|
|
3056
|
+
return start, document.line_end(line)
|
|
3057
|
+
|
|
3058
|
+
token_index = document.first_token_index(start)
|
|
3059
|
+
if entry.kind in _ROUTINE_KINDS:
|
|
3060
|
+
routine_index = _routine_keyword_index(document.tokens, token_index)
|
|
3061
|
+
if routine_index is not None:
|
|
3062
|
+
declaration_end = _routine_declaration_end_index(document.tokens, routine_index)
|
|
3063
|
+
if declaration_end is not None:
|
|
3064
|
+
adjacent_routine = _top_level_routine_boundary_index(
|
|
3065
|
+
document.tokens,
|
|
3066
|
+
routine_index + 1,
|
|
3067
|
+
declaration_end,
|
|
3068
|
+
)
|
|
3069
|
+
if adjacent_routine is not None:
|
|
3070
|
+
boundary_line = document.line_col(
|
|
3071
|
+
document.tokens[adjacent_routine].start
|
|
3072
|
+
)[0]
|
|
3073
|
+
return start, document.line_end(max(line, boundary_line - 1))
|
|
3074
|
+
return start, document.tokens[declaration_end].end
|
|
3075
|
+
for token in document.tokens[token_index:]:
|
|
3076
|
+
if token.value == ";":
|
|
3077
|
+
return start, token.end
|
|
3078
|
+
return start, document.line_end(line)
|
|
3079
|
+
|
|
3080
|
+
|
|
3081
|
+
def _declaration_start(document: _SourceDocument, line: int) -> int:
|
|
3082
|
+
start = document.line_start(line)
|
|
3083
|
+
end = document.line_end(line)
|
|
3084
|
+
while start < end and document.text[start] in {" ", "\t"}:
|
|
3085
|
+
start += 1
|
|
3086
|
+
return start
|
|
3087
|
+
|
|
3088
|
+
|
|
3089
|
+
def _type_span(
|
|
3090
|
+
document: _SourceDocument,
|
|
3091
|
+
entry: _SymbolEntry,
|
|
3092
|
+
) -> tuple[int, int] | None:
|
|
3093
|
+
declaration, _ = _type_declaration_layout(document, entry)
|
|
3094
|
+
return declaration
|
|
3095
|
+
|
|
3096
|
+
|
|
3097
|
+
def _is_forward_type(document: _SourceDocument, entry: _SymbolEntry) -> bool:
|
|
3098
|
+
start = _declaration_start(document, entry.line)
|
|
3099
|
+
token_index = document.first_token_index(start)
|
|
3100
|
+
equals_index = _next_token_value(document.tokens, token_index, "=")
|
|
3101
|
+
if equals_index is None:
|
|
3102
|
+
return False
|
|
3103
|
+
structure_index = _structured_type_index(document.tokens, equals_index + 1)
|
|
3104
|
+
return (
|
|
3105
|
+
structure_index is not None
|
|
3106
|
+
and structure_index + 1 < len(document.tokens)
|
|
3107
|
+
and document.tokens[structure_index + 1].value == ";"
|
|
3108
|
+
)
|
|
3109
|
+
|
|
3110
|
+
|
|
3111
|
+
def _type_declaration_layout(
|
|
3112
|
+
document: _SourceDocument,
|
|
3113
|
+
entry: _SymbolEntry,
|
|
3114
|
+
) -> tuple[tuple[int, int] | None, bool]:
|
|
3115
|
+
start = _declaration_start(document, entry.line)
|
|
3116
|
+
token_index = document.first_token_index(start)
|
|
3117
|
+
equals_index = _next_token_value(document.tokens, token_index, "=")
|
|
3118
|
+
if equals_index is None:
|
|
3119
|
+
return None, False
|
|
3120
|
+
structure_index = _structured_type_index(document.tokens, equals_index + 1)
|
|
3121
|
+
if structure_index is not None:
|
|
3122
|
+
direct_structured = all(
|
|
3123
|
+
token.value in {"packed"}
|
|
3124
|
+
for token in document.tokens[equals_index + 1:structure_index]
|
|
3125
|
+
)
|
|
3126
|
+
if (
|
|
3127
|
+
structure_index + 1 < len(document.tokens)
|
|
3128
|
+
and document.tokens[structure_index + 1].value == ";"
|
|
3129
|
+
):
|
|
3130
|
+
return (start, document.tokens[structure_index + 1].end), True
|
|
3131
|
+
end = _match_end_terminated_block(document.tokens, structure_index)
|
|
3132
|
+
return ((start, end) if end is not None else None), direct_structured
|
|
3133
|
+
|
|
3134
|
+
declaration_end = _top_level_semicolon_index(document.tokens, equals_index + 1)
|
|
3135
|
+
if declaration_end is None:
|
|
3136
|
+
return None, False
|
|
3137
|
+
return (start, document.tokens[declaration_end].end), False
|
|
3138
|
+
|
|
3139
|
+
|
|
3140
|
+
def _next_token_value(
|
|
3141
|
+
tokens: tuple[_Token, ...],
|
|
3142
|
+
start_index: int,
|
|
3143
|
+
value: str,
|
|
3144
|
+
) -> int | None:
|
|
3145
|
+
for index in range(start_index, len(tokens)):
|
|
3146
|
+
if tokens[index].value == value:
|
|
3147
|
+
return index
|
|
3148
|
+
if tokens[index].value == ";":
|
|
3149
|
+
return None
|
|
3150
|
+
return None
|
|
3151
|
+
|
|
3152
|
+
|
|
3153
|
+
def _structured_type_index(
|
|
3154
|
+
tokens: tuple[_Token, ...],
|
|
3155
|
+
start_index: int,
|
|
3156
|
+
) -> int | None:
|
|
3157
|
+
for index in range(start_index, len(tokens)):
|
|
3158
|
+
token = tokens[index]
|
|
3159
|
+
if token.value == ";":
|
|
3160
|
+
return None
|
|
3161
|
+
if (
|
|
3162
|
+
token.word
|
|
3163
|
+
and not token.escaped
|
|
3164
|
+
and token.value in _STRUCTURED_TYPE_WORDS
|
|
3165
|
+
and _is_structured_type_opener(tokens, index)
|
|
3166
|
+
):
|
|
3167
|
+
return index
|
|
3168
|
+
return None
|
|
3169
|
+
|
|
3170
|
+
|
|
3171
|
+
def _routine_span(
|
|
3172
|
+
document: _SourceDocument,
|
|
3173
|
+
start: int,
|
|
3174
|
+
) -> tuple[int, int] | None:
|
|
3175
|
+
cached = document.routine_spans.get(start, ...)
|
|
3176
|
+
if cached is not ...:
|
|
3177
|
+
return cached
|
|
3178
|
+
token_index = document.first_token_index(start)
|
|
3179
|
+
found = _find_routine_token_span(
|
|
3180
|
+
document.tokens,
|
|
3181
|
+
document.token_starts,
|
|
3182
|
+
token_index,
|
|
3183
|
+
cache=document.routine_token_spans,
|
|
3184
|
+
)
|
|
3185
|
+
span = (start, found[1]) if found is not None else None
|
|
3186
|
+
document.routine_spans[start] = span
|
|
3187
|
+
return span
|
|
3188
|
+
|
|
3189
|
+
|
|
3190
|
+
def _find_routine_token_span(
|
|
3191
|
+
tokens: tuple[_Token, ...],
|
|
3192
|
+
token_starts: tuple[int, ...],
|
|
3193
|
+
start_index: int,
|
|
3194
|
+
*,
|
|
3195
|
+
cache: dict[int, tuple[int, int, int] | None] | None = None,
|
|
3196
|
+
depth: int = 0,
|
|
3197
|
+
) -> tuple[int, int, int] | None:
|
|
3198
|
+
if depth > 64:
|
|
3199
|
+
return None
|
|
3200
|
+
routine_index = _routine_keyword_index(tokens, start_index)
|
|
3201
|
+
if routine_index is None:
|
|
3202
|
+
return None
|
|
3203
|
+
spans = cache if cache is not None else {}
|
|
3204
|
+
missing = object()
|
|
3205
|
+
cached = spans.get(routine_index, missing)
|
|
3206
|
+
if cached is not missing:
|
|
3207
|
+
if cached is None:
|
|
3208
|
+
return None
|
|
3209
|
+
return tokens[start_index].start, cached[1], cached[2]
|
|
3210
|
+
|
|
3211
|
+
heading_end = _heading_semicolon_index(tokens, routine_index)
|
|
3212
|
+
if heading_end is None:
|
|
3213
|
+
spans[routine_index] = None
|
|
3214
|
+
return None
|
|
3215
|
+
|
|
3216
|
+
frames: list[list[int]] = [[routine_index, heading_end + 1]]
|
|
3217
|
+
|
|
3218
|
+
def reject_active_frames() -> None:
|
|
3219
|
+
for active_routine_index, _ in frames:
|
|
3220
|
+
spans[active_routine_index] = None
|
|
3221
|
+
frames.clear()
|
|
3222
|
+
|
|
3223
|
+
while frames:
|
|
3224
|
+
frame = frames[-1]
|
|
3225
|
+
frame_routine_index, index = frame
|
|
3226
|
+
if index >= len(tokens):
|
|
3227
|
+
reject_active_frames()
|
|
3228
|
+
continue
|
|
3229
|
+
|
|
3230
|
+
token = tokens[index]
|
|
3231
|
+
if token.directive:
|
|
3232
|
+
reject_active_frames()
|
|
3233
|
+
continue
|
|
3234
|
+
if token.word and not token.escaped:
|
|
3235
|
+
if token.value in _NO_BODY_DIRECTIVES:
|
|
3236
|
+
spans[frame_routine_index] = None
|
|
3237
|
+
frames.pop()
|
|
3238
|
+
continue
|
|
3239
|
+
if token.value in {"implementation", "initialization", "finalization"}:
|
|
3240
|
+
reject_active_frames()
|
|
3241
|
+
continue
|
|
3242
|
+
if (
|
|
3243
|
+
token.value in _STRUCTURED_TYPE_WORDS
|
|
3244
|
+
and _is_structured_type_opener(tokens, index)
|
|
3245
|
+
):
|
|
3246
|
+
if index + 1 < len(tokens) and tokens[index + 1].value == ";":
|
|
3247
|
+
frame[1] = index + 2
|
|
3248
|
+
continue
|
|
3249
|
+
structured_end = _match_end_terminated_block(tokens, index)
|
|
3250
|
+
if structured_end is None:
|
|
3251
|
+
reject_active_frames()
|
|
3252
|
+
continue
|
|
3253
|
+
frame[1] = bisect_left(token_starts, structured_end)
|
|
3254
|
+
continue
|
|
3255
|
+
if token.value == "end":
|
|
3256
|
+
reject_active_frames()
|
|
3257
|
+
continue
|
|
3258
|
+
if token.value in {"begin", "asm"}:
|
|
3259
|
+
end = _match_end_terminated_block(tokens, index)
|
|
3260
|
+
if end is None:
|
|
3261
|
+
reject_active_frames()
|
|
3262
|
+
continue
|
|
3263
|
+
end_index = bisect_left(token_starts, end)
|
|
3264
|
+
spans[frame_routine_index] = (
|
|
3265
|
+
tokens[frame_routine_index].start,
|
|
3266
|
+
end,
|
|
3267
|
+
end_index,
|
|
3268
|
+
)
|
|
3269
|
+
frames.pop()
|
|
3270
|
+
continue
|
|
3271
|
+
if token.value in _ROUTINE_WORDS and _is_nested_routine_declaration(tokens, index):
|
|
3272
|
+
nested_routine_index = _routine_keyword_index(tokens, index)
|
|
3273
|
+
if nested_routine_index is None:
|
|
3274
|
+
frame[1] = index + 1
|
|
3275
|
+
continue
|
|
3276
|
+
nested = spans.get(nested_routine_index, missing)
|
|
3277
|
+
if nested is missing:
|
|
3278
|
+
nested_heading_end = _heading_semicolon_index(tokens, nested_routine_index)
|
|
3279
|
+
if nested_heading_end is None:
|
|
3280
|
+
spans[nested_routine_index] = None
|
|
3281
|
+
continue
|
|
3282
|
+
frames.append([nested_routine_index, nested_heading_end + 1])
|
|
3283
|
+
continue
|
|
3284
|
+
if nested is not None:
|
|
3285
|
+
frame[1] = max(index + 1, nested[2])
|
|
3286
|
+
continue
|
|
3287
|
+
skipped = _routine_declaration_end_index(tokens, index)
|
|
3288
|
+
if skipped is not None:
|
|
3289
|
+
frame[1] = skipped + 1
|
|
3290
|
+
continue
|
|
3291
|
+
frame[1] = index + 1
|
|
3292
|
+
|
|
3293
|
+
result = spans.get(routine_index)
|
|
3294
|
+
if result is None:
|
|
3295
|
+
return None
|
|
3296
|
+
return tokens[start_index].start, result[1], result[2]
|
|
3297
|
+
|
|
3298
|
+
|
|
3299
|
+
def _routine_keyword_index(
|
|
3300
|
+
tokens: tuple[_Token, ...],
|
|
3301
|
+
start_index: int,
|
|
3302
|
+
) -> int | None:
|
|
3303
|
+
for index in range(start_index, min(len(tokens), start_index + 6)):
|
|
3304
|
+
token = tokens[index]
|
|
3305
|
+
if token.word and not token.escaped and token.value in _ROUTINE_WORDS:
|
|
3306
|
+
return index
|
|
3307
|
+
if token.value == ";":
|
|
3308
|
+
return None
|
|
3309
|
+
return None
|
|
3310
|
+
|
|
3311
|
+
|
|
3312
|
+
def _heading_semicolon_index(
|
|
3313
|
+
tokens: tuple[_Token, ...],
|
|
3314
|
+
routine_index: int,
|
|
3315
|
+
) -> int | None:
|
|
3316
|
+
return _top_level_semicolon_index(tokens, routine_index + 1)
|
|
3317
|
+
|
|
3318
|
+
|
|
3319
|
+
def _routine_declaration_end_index(
|
|
3320
|
+
tokens: tuple[_Token, ...],
|
|
3321
|
+
routine_index: int,
|
|
3322
|
+
) -> int | None:
|
|
3323
|
+
declaration_end = _heading_semicolon_index(tokens, routine_index)
|
|
3324
|
+
if declaration_end is None:
|
|
3325
|
+
return None
|
|
3326
|
+
cursor = declaration_end + 1
|
|
3327
|
+
while cursor < len(tokens):
|
|
3328
|
+
directive = tokens[cursor]
|
|
3329
|
+
if (
|
|
3330
|
+
not directive.word
|
|
3331
|
+
or directive.escaped
|
|
3332
|
+
or directive.value not in _ROUTINE_DIRECTIVES
|
|
3333
|
+
):
|
|
3334
|
+
break
|
|
3335
|
+
directive_end = _top_level_semicolon_index(tokens, cursor + 1)
|
|
3336
|
+
if directive_end is None:
|
|
3337
|
+
break
|
|
3338
|
+
declaration_end = directive_end
|
|
3339
|
+
cursor = directive_end + 1
|
|
3340
|
+
return declaration_end
|
|
3341
|
+
|
|
3342
|
+
|
|
3343
|
+
def _top_level_semicolon_index(
|
|
3344
|
+
tokens: tuple[_Token, ...],
|
|
3345
|
+
start_index: int,
|
|
3346
|
+
) -> int | None:
|
|
3347
|
+
parentheses = 0
|
|
3348
|
+
brackets = 0
|
|
3349
|
+
angles = 0
|
|
3350
|
+
for index in range(start_index, len(tokens)):
|
|
3351
|
+
value = tokens[index].value
|
|
3352
|
+
if value == "(":
|
|
3353
|
+
parentheses += 1
|
|
3354
|
+
elif value == ")":
|
|
3355
|
+
parentheses = max(0, parentheses - 1)
|
|
3356
|
+
elif value == "[":
|
|
3357
|
+
brackets += 1
|
|
3358
|
+
elif value == "]":
|
|
3359
|
+
brackets = max(0, brackets - 1)
|
|
3360
|
+
elif value == "<":
|
|
3361
|
+
angles += 1
|
|
3362
|
+
elif value == ">":
|
|
3363
|
+
angles = max(0, angles - 1)
|
|
3364
|
+
elif value == ";" and parentheses == 0 and brackets == 0 and angles == 0:
|
|
3365
|
+
return index
|
|
3366
|
+
return None
|
|
3367
|
+
|
|
3368
|
+
|
|
3369
|
+
def _top_level_routine_boundary_index(
|
|
3370
|
+
tokens: tuple[_Token, ...],
|
|
3371
|
+
start_index: int,
|
|
3372
|
+
stop_index: int,
|
|
3373
|
+
) -> int | None:
|
|
3374
|
+
parentheses = 0
|
|
3375
|
+
brackets = 0
|
|
3376
|
+
angles = 0
|
|
3377
|
+
for index in range(start_index, min(stop_index, len(tokens))):
|
|
3378
|
+
token = tokens[index]
|
|
3379
|
+
value = token.value
|
|
3380
|
+
if (
|
|
3381
|
+
parentheses == 0
|
|
3382
|
+
and brackets == 0
|
|
3383
|
+
and angles == 0
|
|
3384
|
+
and token.word
|
|
3385
|
+
and not token.escaped
|
|
3386
|
+
and value in _ROUTINE_WORDS
|
|
3387
|
+
and _is_nested_routine_declaration(tokens, index)
|
|
3388
|
+
):
|
|
3389
|
+
return index
|
|
3390
|
+
if value == "(":
|
|
3391
|
+
parentheses += 1
|
|
3392
|
+
elif value == ")":
|
|
3393
|
+
parentheses = max(0, parentheses - 1)
|
|
3394
|
+
elif value == "[":
|
|
3395
|
+
brackets += 1
|
|
3396
|
+
elif value == "]":
|
|
3397
|
+
brackets = max(0, brackets - 1)
|
|
3398
|
+
elif value == "<":
|
|
3399
|
+
angles += 1
|
|
3400
|
+
elif value == ">":
|
|
3401
|
+
angles = max(0, angles - 1)
|
|
3402
|
+
return None
|
|
3403
|
+
|
|
3404
|
+
|
|
3405
|
+
def _is_nested_routine_declaration(
|
|
3406
|
+
tokens: tuple[_Token, ...],
|
|
3407
|
+
index: int,
|
|
3408
|
+
) -> bool:
|
|
3409
|
+
token = tokens[index]
|
|
3410
|
+
if token.value == "operator":
|
|
3411
|
+
return True
|
|
3412
|
+
previous = tokens[index - 1] if index > 0 else None
|
|
3413
|
+
if previous is not None and previous.value in {":", "=", "of", "to", "reference", "."}:
|
|
3414
|
+
return False
|
|
3415
|
+
following = tokens[index + 1] if index + 1 < len(tokens) else None
|
|
3416
|
+
if following is None or not following.word or following.value in {"of", "object"}:
|
|
3417
|
+
return False
|
|
3418
|
+
return True
|
|
3419
|
+
|
|
3420
|
+
|
|
3421
|
+
def _match_end_terminated_block(
|
|
3422
|
+
tokens: tuple[_Token, ...],
|
|
3423
|
+
opener_index: int,
|
|
3424
|
+
) -> int | None:
|
|
3425
|
+
stack = [tokens[opener_index].value]
|
|
3426
|
+
for index in range(opener_index + 1, len(tokens)):
|
|
3427
|
+
token = tokens[index]
|
|
3428
|
+
if token.directive:
|
|
3429
|
+
return None
|
|
3430
|
+
if not token.word or token.escaped:
|
|
3431
|
+
continue
|
|
3432
|
+
value = token.value
|
|
3433
|
+
if stack[-1] == "asm":
|
|
3434
|
+
if value != "end":
|
|
3435
|
+
continue
|
|
3436
|
+
elif value in _BLOCK_WORDS:
|
|
3437
|
+
if value == "case" and stack[-1] in _STRUCTURED_TYPE_WORDS:
|
|
3438
|
+
continue
|
|
3439
|
+
stack.append(value)
|
|
3440
|
+
continue
|
|
3441
|
+
elif value in _STRUCTURED_TYPE_WORDS and _is_structured_type_opener(tokens, index):
|
|
3442
|
+
stack.append(value)
|
|
3443
|
+
continue
|
|
3444
|
+
elif value != "end":
|
|
3445
|
+
continue
|
|
3446
|
+
|
|
3447
|
+
stack.pop()
|
|
3448
|
+
if stack:
|
|
3449
|
+
continue
|
|
3450
|
+
end = token.end
|
|
3451
|
+
if index + 1 < len(tokens) and tokens[index + 1].value in {";", "."}:
|
|
3452
|
+
end = tokens[index + 1].end
|
|
3453
|
+
return end
|
|
3454
|
+
return None
|
|
3455
|
+
|
|
3456
|
+
|
|
3457
|
+
def _is_structured_type_opener(tokens: tuple[_Token, ...], index: int) -> bool:
|
|
3458
|
+
token = tokens[index]
|
|
3459
|
+
previous = tokens[index - 1] if index > 0 else None
|
|
3460
|
+
following = tokens[index + 1] if index + 1 < len(tokens) else None
|
|
3461
|
+
if token.value == "class" and following is not None and following.value == "of":
|
|
3462
|
+
return False
|
|
3463
|
+
if previous is not None and previous.value == "of":
|
|
3464
|
+
return token.value in {"record", "object"} and _is_array_of_context(tokens, index)
|
|
3465
|
+
if previous is not None and previous.value == ":" and _inside_generic_angles(tokens, index):
|
|
3466
|
+
return False
|
|
3467
|
+
if previous is None:
|
|
3468
|
+
return False
|
|
3469
|
+
if previous.value in {"=", ":", "packed"}:
|
|
3470
|
+
return True
|
|
3471
|
+
if previous.value == "^" and token.value in {"record", "object"}:
|
|
3472
|
+
return True
|
|
3473
|
+
return False
|
|
3474
|
+
|
|
3475
|
+
|
|
3476
|
+
def _is_array_of_context(tokens: tuple[_Token, ...], index: int) -> bool:
|
|
3477
|
+
for candidate in range(index - 2, max(-1, index - 200), -1):
|
|
3478
|
+
token = tokens[candidate]
|
|
3479
|
+
if token.value == "array":
|
|
3480
|
+
return True
|
|
3481
|
+
if token.value in {"procedure", "function", "reference", "=", ";"}:
|
|
3482
|
+
return False
|
|
3483
|
+
return False
|
|
3484
|
+
|
|
3485
|
+
|
|
3486
|
+
def _inside_generic_angles(tokens: tuple[_Token, ...], index: int) -> bool:
|
|
3487
|
+
depth = 0
|
|
3488
|
+
for candidate in range(index - 1, max(-1, index - 100), -1):
|
|
3489
|
+
value = tokens[candidate].value
|
|
3490
|
+
if value == ">":
|
|
3491
|
+
depth += 1
|
|
3492
|
+
elif value == "<":
|
|
3493
|
+
if depth == 0:
|
|
3494
|
+
return True
|
|
3495
|
+
depth -= 1
|
|
3496
|
+
elif depth == 0 and value in {";", "begin", "end"}:
|
|
3497
|
+
return False
|
|
3498
|
+
return False
|
|
3499
|
+
|
|
3500
|
+
|
|
3501
|
+
def _source_items(
|
|
3502
|
+
document: _SourceDocument,
|
|
3503
|
+
start: int,
|
|
3504
|
+
end: int,
|
|
3505
|
+
max_chars: int,
|
|
3506
|
+
*,
|
|
3507
|
+
role: str,
|
|
3508
|
+
target_id: str,
|
|
3509
|
+
) -> list[dict[str, object]]:
|
|
3510
|
+
start = min(max(start, 0), len(document.text))
|
|
3511
|
+
end = min(max(end, start), len(document.text))
|
|
3512
|
+
probe_end = min(end, start + 1)
|
|
3513
|
+
compact = max_chars <= 256 or len(
|
|
3514
|
+
_compact_json(
|
|
3515
|
+
_source_item(
|
|
3516
|
+
document,
|
|
3517
|
+
start,
|
|
3518
|
+
probe_end,
|
|
3519
|
+
role=role,
|
|
3520
|
+
target_id=target_id,
|
|
3521
|
+
chunk_index=999999,
|
|
3522
|
+
chunk_count=999999,
|
|
3523
|
+
compact=False,
|
|
3524
|
+
)
|
|
3525
|
+
)
|
|
3526
|
+
) + 2 >= max_chars
|
|
3527
|
+
spans: list[tuple[int, int]] = []
|
|
3528
|
+
offset = start
|
|
3529
|
+
while offset < end:
|
|
3530
|
+
upper = min(end, offset + _SOURCE_CHUNK_CHARS)
|
|
3531
|
+
accepted = _fit_source_end(
|
|
3532
|
+
document,
|
|
3533
|
+
offset,
|
|
3534
|
+
upper,
|
|
3535
|
+
max_chars,
|
|
3536
|
+
role=role,
|
|
3537
|
+
target_id=target_id,
|
|
3538
|
+
compact=compact,
|
|
3539
|
+
)
|
|
3540
|
+
if accepted <= offset:
|
|
3541
|
+
raise AgentProtocolError(
|
|
3542
|
+
"item_too_large",
|
|
3543
|
+
"max_chars is too small for a typed source chunk.",
|
|
3544
|
+
)
|
|
3545
|
+
spans.append((offset, accepted))
|
|
3546
|
+
offset = accepted
|
|
3547
|
+
if not spans:
|
|
3548
|
+
spans.append((start, end))
|
|
3549
|
+
|
|
3550
|
+
total = len(spans)
|
|
3551
|
+
return [
|
|
3552
|
+
_source_item(
|
|
3553
|
+
document,
|
|
3554
|
+
item_start,
|
|
3555
|
+
item_end,
|
|
3556
|
+
role=role,
|
|
3557
|
+
target_id=target_id,
|
|
3558
|
+
chunk_index=index,
|
|
3559
|
+
chunk_count=total,
|
|
3560
|
+
compact=compact,
|
|
3561
|
+
)
|
|
3562
|
+
for index, (item_start, item_end) in enumerate(spans)
|
|
3563
|
+
]
|
|
3564
|
+
|
|
3565
|
+
|
|
3566
|
+
def _fit_source_end(
|
|
3567
|
+
document: _SourceDocument,
|
|
3568
|
+
start: int,
|
|
3569
|
+
upper: int,
|
|
3570
|
+
max_chars: int,
|
|
3571
|
+
*,
|
|
3572
|
+
role: str,
|
|
3573
|
+
target_id: str,
|
|
3574
|
+
compact: bool,
|
|
3575
|
+
) -> int:
|
|
3576
|
+
low = start + 1
|
|
3577
|
+
high = upper
|
|
3578
|
+
accepted = start
|
|
3579
|
+
while low <= high:
|
|
3580
|
+
candidate_end = (low + high) // 2
|
|
3581
|
+
candidate = _source_item(
|
|
3582
|
+
document,
|
|
3583
|
+
start,
|
|
3584
|
+
candidate_end,
|
|
3585
|
+
role=role,
|
|
3586
|
+
target_id=target_id,
|
|
3587
|
+
chunk_index=999999,
|
|
3588
|
+
chunk_count=999999,
|
|
3589
|
+
compact=compact,
|
|
3590
|
+
)
|
|
3591
|
+
if len(_compact_json(candidate)) + 2 <= max_chars:
|
|
3592
|
+
accepted = candidate_end
|
|
3593
|
+
low = candidate_end + 1
|
|
3594
|
+
else:
|
|
3595
|
+
high = candidate_end - 1
|
|
3596
|
+
return accepted
|
|
3597
|
+
|
|
3598
|
+
|
|
3599
|
+
def _source_item(
|
|
3600
|
+
document: _SourceDocument,
|
|
3601
|
+
start: int,
|
|
3602
|
+
end: int,
|
|
3603
|
+
*,
|
|
3604
|
+
role: str,
|
|
3605
|
+
target_id: str,
|
|
3606
|
+
chunk_index: int,
|
|
3607
|
+
chunk_count: int,
|
|
3608
|
+
compact: bool,
|
|
3609
|
+
) -> dict[str, object]:
|
|
3610
|
+
start_line, start_col = document.line_col(start)
|
|
3611
|
+
end_line, end_col = document.line_col(end)
|
|
3612
|
+
item: dict[str, object] = {
|
|
3613
|
+
"item_type": "source_chunk" if compact else "source",
|
|
3614
|
+
"path": document.display_path,
|
|
3615
|
+
"start_line": start_line,
|
|
3616
|
+
"start_col": start_col,
|
|
3617
|
+
"end_line": end_line,
|
|
3618
|
+
"end_col": end_col,
|
|
3619
|
+
"chunk_index": chunk_index,
|
|
3620
|
+
"chunk_count": chunk_count,
|
|
3621
|
+
"text": document.text[start:end],
|
|
3622
|
+
}
|
|
3623
|
+
if not compact:
|
|
3624
|
+
item["role"] = role
|
|
3625
|
+
item["target_id"] = target_id
|
|
3626
|
+
return item
|
|
3627
|
+
|
|
3628
|
+
|
|
3629
|
+
def _line_starts(text: str) -> tuple[int, ...]:
|
|
3630
|
+
starts = [0]
|
|
3631
|
+
starts.extend(index + 1 for index, character in enumerate(text) if character == "\n")
|
|
3632
|
+
return tuple(starts)
|
|
3633
|
+
|
|
3634
|
+
|
|
3635
|
+
def _lex_delphi(text: str) -> list[_Token]:
|
|
3636
|
+
tokens: list[_Token] = []
|
|
3637
|
+
index = 0
|
|
3638
|
+
length = len(text)
|
|
3639
|
+
while index < length:
|
|
3640
|
+
character = text[index]
|
|
3641
|
+
if character.isspace():
|
|
3642
|
+
index += 1
|
|
3643
|
+
continue
|
|
3644
|
+
if text.startswith("//", index):
|
|
3645
|
+
newline = text.find("\n", index + 2)
|
|
3646
|
+
index = length if newline < 0 else newline + 1
|
|
3647
|
+
continue
|
|
3648
|
+
if character == "{":
|
|
3649
|
+
close = text.find("}", index + 1)
|
|
3650
|
+
end = length if close < 0 else close + 1
|
|
3651
|
+
if index + 1 < length and text[index + 1] == "$":
|
|
3652
|
+
tokens.append(
|
|
3653
|
+
_Token(
|
|
3654
|
+
unicodedata.normalize("NFC", text[index:end]),
|
|
3655
|
+
index,
|
|
3656
|
+
end,
|
|
3657
|
+
directive=True,
|
|
3658
|
+
)
|
|
3659
|
+
)
|
|
3660
|
+
index = end
|
|
3661
|
+
continue
|
|
3662
|
+
if text.startswith("(*", index):
|
|
3663
|
+
close = text.find("*)", index + 2)
|
|
3664
|
+
end = length if close < 0 else close + 2
|
|
3665
|
+
if index + 2 < length and text[index + 2] == "$":
|
|
3666
|
+
tokens.append(
|
|
3667
|
+
_Token(
|
|
3668
|
+
unicodedata.normalize("NFC", text[index:end]),
|
|
3669
|
+
index,
|
|
3670
|
+
end,
|
|
3671
|
+
directive=True,
|
|
3672
|
+
)
|
|
3673
|
+
)
|
|
3674
|
+
index = end
|
|
3675
|
+
continue
|
|
3676
|
+
if character == "'":
|
|
3677
|
+
block_end = multiline_string_block_end(text, index)
|
|
3678
|
+
index = block_end if block_end is not None else _quoted_end(text, index)
|
|
3679
|
+
continue
|
|
3680
|
+
if character == "&" and index + 1 < length and _identifier_start(text[index + 1]):
|
|
3681
|
+
end = index + 2
|
|
3682
|
+
while end < length and _identifier_part(text[end]):
|
|
3683
|
+
end += 1
|
|
3684
|
+
tokens.append(_Token(text[index + 1:end].casefold(), index, end, word=True, escaped=True))
|
|
3685
|
+
index = end
|
|
3686
|
+
continue
|
|
3687
|
+
if _identifier_start(character):
|
|
3688
|
+
end = index + 1
|
|
3689
|
+
while end < length and _identifier_part(text[end]):
|
|
3690
|
+
end += 1
|
|
3691
|
+
tokens.append(_Token(text[index:end].casefold(), index, end, word=True))
|
|
3692
|
+
index = end
|
|
3693
|
+
continue
|
|
3694
|
+
if text[index:index + 2] in {":=", "<=", ">=", "<>", ".."}:
|
|
3695
|
+
tokens.append(_Token(text[index:index + 2], index, index + 2))
|
|
3696
|
+
index += 2
|
|
3697
|
+
continue
|
|
3698
|
+
tokens.append(_Token(character, index, index + 1))
|
|
3699
|
+
index += 1
|
|
3700
|
+
return tokens
|
|
3701
|
+
|
|
3702
|
+
|
|
3703
|
+
def _quoted_end(text: str, start: int) -> int:
|
|
3704
|
+
index = start + 1
|
|
3705
|
+
while index < len(text):
|
|
3706
|
+
if text[index] != "'":
|
|
3707
|
+
index += 1
|
|
3708
|
+
continue
|
|
3709
|
+
if index + 1 < len(text) and text[index + 1] == "'":
|
|
3710
|
+
index += 2
|
|
3711
|
+
continue
|
|
3712
|
+
return index + 1
|
|
3713
|
+
return len(text)
|
|
3714
|
+
|
|
3715
|
+
|
|
3716
|
+
def _identifier_start(character: str) -> bool:
|
|
3717
|
+
return character == "_" or character.isalpha()
|
|
3718
|
+
|
|
3719
|
+
|
|
3720
|
+
def _identifier_part(character: str) -> bool:
|
|
3721
|
+
return character == "_" or character.isalnum()
|
|
3722
|
+
|
|
3723
|
+
|
|
3724
|
+
def _ranked_entries(entries: tuple[_SymbolEntry, ...], query: str) -> list[_SymbolEntry]:
|
|
3725
|
+
normalized_query = _normalized(query.strip())
|
|
3726
|
+
ranked: list[tuple[int, tuple[object, ...], _SymbolEntry]] = []
|
|
3727
|
+
for entry in entries:
|
|
3728
|
+
normalized_qualified = entry.normalized_qualified_name
|
|
3729
|
+
relative_offset = entry.relative_name_offset
|
|
3730
|
+
if not normalized_query:
|
|
3731
|
+
rank = 3
|
|
3732
|
+
elif (
|
|
3733
|
+
entry.normalized_name == normalized_query
|
|
3734
|
+
or normalized_qualified == normalized_query
|
|
3735
|
+
or (
|
|
3736
|
+
len(normalized_qualified) - relative_offset == len(normalized_query)
|
|
3737
|
+
and normalized_qualified.endswith(normalized_query)
|
|
3738
|
+
)
|
|
3739
|
+
):
|
|
3740
|
+
rank = 0
|
|
3741
|
+
elif (
|
|
3742
|
+
entry.normalized_name.startswith(normalized_query)
|
|
3743
|
+
or normalized_qualified.startswith(normalized_query)
|
|
3744
|
+
or normalized_qualified.startswith(normalized_query, relative_offset)
|
|
3745
|
+
):
|
|
3746
|
+
rank = 1
|
|
3747
|
+
elif (
|
|
3748
|
+
normalized_query in entry.normalized_name
|
|
3749
|
+
or normalized_query in normalized_qualified
|
|
3750
|
+
):
|
|
3751
|
+
rank = 2
|
|
3752
|
+
else:
|
|
3753
|
+
continue
|
|
3754
|
+
ranked.append((rank, _entry_sort_key(entry), entry))
|
|
3755
|
+
ranked.sort(key=lambda item: (item[0], item[1]))
|
|
3756
|
+
return [item[2] for item in ranked]
|
|
3757
|
+
|
|
3758
|
+
|
|
3759
|
+
def _symbol_sort_key(symbol: Symbol) -> tuple[object, ...]:
|
|
3760
|
+
return (
|
|
3761
|
+
symbol.decl_range.start_line,
|
|
3762
|
+
symbol.decl_range.start_col,
|
|
3763
|
+
symbol.kind.value.casefold(),
|
|
3764
|
+
_normalized(symbol.name),
|
|
3765
|
+
symbol.name,
|
|
3766
|
+
)
|
|
3767
|
+
|
|
3768
|
+
|
|
3769
|
+
def _raw_sort_key(raw: _RawSymbol) -> tuple[object, ...]:
|
|
3770
|
+
return (
|
|
3771
|
+
raw.path.casefold(),
|
|
3772
|
+
raw.path,
|
|
3773
|
+
raw.line,
|
|
3774
|
+
raw.column,
|
|
3775
|
+
raw.kind.value.casefold(),
|
|
3776
|
+
_normalized(raw.qualified_name),
|
|
3777
|
+
raw.qualified_name,
|
|
3778
|
+
)
|
|
3779
|
+
|
|
3780
|
+
|
|
3781
|
+
def _entry_sort_key(entry: _SymbolEntry) -> tuple[object, ...]:
|
|
3782
|
+
return (
|
|
3783
|
+
entry.normalized_qualified_name,
|
|
3784
|
+
entry.kind.value.casefold(),
|
|
3785
|
+
entry.path.casefold(),
|
|
3786
|
+
entry.path,
|
|
3787
|
+
entry.line,
|
|
3788
|
+
entry.column,
|
|
3789
|
+
entry.ordinal,
|
|
3790
|
+
entry.target_id,
|
|
3791
|
+
)
|
|
3792
|
+
|
|
3793
|
+
|
|
3794
|
+
def _normalized(value: str) -> str:
|
|
3795
|
+
return unicodedata.normalize("NFC", value).casefold()
|
|
3796
|
+
|
|
3797
|
+
|
|
3798
|
+
def _normalized_search_fields(
|
|
3799
|
+
name: str,
|
|
3800
|
+
qualified_name: str,
|
|
3801
|
+
unit_name: str,
|
|
3802
|
+
) -> tuple[str, str, int]:
|
|
3803
|
+
normalized_name = _normalized(name)
|
|
3804
|
+
normalized_qualified = _normalized(qualified_name)
|
|
3805
|
+
prefix = f"{_normalized(unit_name)}."
|
|
3806
|
+
relative_name_offset = (
|
|
3807
|
+
len(prefix)
|
|
3808
|
+
if normalized_qualified.startswith(prefix)
|
|
3809
|
+
else 0
|
|
3810
|
+
)
|
|
3811
|
+
return normalized_name, normalized_qualified, relative_name_offset
|
|
3812
|
+
|
|
3813
|
+
|
|
3814
|
+
def _request_fingerprint(
|
|
3815
|
+
request: AgentRequest,
|
|
3816
|
+
*,
|
|
3817
|
+
project_id: str,
|
|
3818
|
+
target_id: str,
|
|
3819
|
+
) -> str:
|
|
3820
|
+
payload = {
|
|
3821
|
+
"action": request.action,
|
|
3822
|
+
"depth": request.depth,
|
|
3823
|
+
"detail": request.detail,
|
|
3824
|
+
"direction": request.direction,
|
|
3825
|
+
"graph": request.graph,
|
|
3826
|
+
"max_chars": request.max_chars,
|
|
3827
|
+
"max_items": request.max_items,
|
|
3828
|
+
"project_id": project_id,
|
|
3829
|
+
"query": request.query,
|
|
3830
|
+
"relation": request.relation,
|
|
3831
|
+
"target_id": target_id,
|
|
3832
|
+
}
|
|
3833
|
+
encoded = _compact_json(payload).encode("utf-8")
|
|
3834
|
+
return f"agent_request_v3_{hashlib.sha256(encoded).hexdigest()}"
|
|
3835
|
+
|
|
3836
|
+
|
|
3837
|
+
def _prepare_items(
|
|
3838
|
+
items: Sequence[dict[str, object]],
|
|
3839
|
+
max_chars: int,
|
|
3840
|
+
) -> list[dict[str, object]]:
|
|
3841
|
+
prepared: list[dict[str, object]] = []
|
|
3842
|
+
for item in items:
|
|
3843
|
+
if len(_compact_json(item)) + 2 <= max_chars:
|
|
3844
|
+
prepared.append(item)
|
|
3845
|
+
continue
|
|
3846
|
+
prepared.extend(_json_chunks(item, max_chars))
|
|
3847
|
+
return prepared
|
|
3848
|
+
|
|
3849
|
+
|
|
3850
|
+
def _json_chunks(item: dict[str, object], max_chars: int) -> list[dict[str, object]]:
|
|
3851
|
+
serialized = _compact_json(item)
|
|
3852
|
+
if item.get("item_type") in {"source", "source_chunk"}:
|
|
3853
|
+
item_type = "source_chunk"
|
|
3854
|
+
else:
|
|
3855
|
+
item_type = "card_chunk" if "target_id" in item else "json_chunk"
|
|
3856
|
+
chunks: list[str] = []
|
|
3857
|
+
offset = 0
|
|
3858
|
+
while offset < len(serialized):
|
|
3859
|
+
low = 1
|
|
3860
|
+
high = len(serialized) - offset
|
|
3861
|
+
accepted = 0
|
|
3862
|
+
while low <= high:
|
|
3863
|
+
size = (low + high) // 2
|
|
3864
|
+
candidate = {
|
|
3865
|
+
"item_type": item_type,
|
|
3866
|
+
"chunk_index": len(chunks),
|
|
3867
|
+
"chunk_count": 999999,
|
|
3868
|
+
"json": serialized[offset:offset + size],
|
|
3869
|
+
}
|
|
3870
|
+
if len(_compact_json(candidate)) + 2 <= max_chars:
|
|
3871
|
+
accepted = size
|
|
3872
|
+
low = size + 1
|
|
3873
|
+
else:
|
|
3874
|
+
high = size - 1
|
|
3875
|
+
if accepted == 0:
|
|
3876
|
+
raise AgentProtocolError(
|
|
3877
|
+
"item_too_large",
|
|
3878
|
+
"max_chars is too small for a structured response chunk.",
|
|
3879
|
+
)
|
|
3880
|
+
chunks.append(serialized[offset:offset + accepted])
|
|
3881
|
+
offset += accepted
|
|
3882
|
+
total = len(chunks)
|
|
3883
|
+
return [
|
|
3884
|
+
{
|
|
3885
|
+
"item_type": item_type,
|
|
3886
|
+
"chunk_index": index,
|
|
3887
|
+
"chunk_count": total,
|
|
3888
|
+
"json": chunk,
|
|
3889
|
+
}
|
|
3890
|
+
for index, chunk in enumerate(chunks)
|
|
3891
|
+
]
|
|
3892
|
+
|
|
3893
|
+
|
|
3894
|
+
def _compact_json(value: object) -> str:
|
|
3895
|
+
try:
|
|
3896
|
+
return json.dumps(
|
|
3897
|
+
value,
|
|
3898
|
+
ensure_ascii=False,
|
|
3899
|
+
sort_keys=True,
|
|
3900
|
+
separators=(",", ":"),
|
|
3901
|
+
allow_nan=False,
|
|
3902
|
+
)
|
|
3903
|
+
except (TypeError, ValueError):
|
|
3904
|
+
raise AgentProtocolError("invalid_item", "Item is not JSON-compatible.") from None
|
|
3905
|
+
|
|
3906
|
+
|
|
3907
|
+
__all__ = ["AgentContext"]
|