python-delphi-lsp 3.6.0__cp310-abi3-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. delphi_lsp/__init__.py +184 -0
  2. delphi_lsp/_version.py +1 -0
  3. delphi_lsp/agent_cache.py +1574 -0
  4. delphi_lsp/agent_cli.py +902 -0
  5. delphi_lsp/agent_context.py +3907 -0
  6. delphi_lsp/agent_cpg.py +349 -0
  7. delphi_lsp/agent_cpg_builder.py +666 -0
  8. delphi_lsp/agent_layers.py +1410 -0
  9. delphi_lsp/agent_metrics.py +526 -0
  10. delphi_lsp/agent_protocol.py +445 -0
  11. delphi_lsp/agent_relations.py +1881 -0
  12. delphi_lsp/agent_templates.py +614 -0
  13. delphi_lsp/agent_wiki.py +2853 -0
  14. delphi_lsp/agent_workspace.py +860 -0
  15. delphi_lsp/ast_serialize.py +126 -0
  16. delphi_lsp/binary.py +250 -0
  17. delphi_lsp/comment_builder.py +28 -0
  18. delphi_lsp/consts.py +345 -0
  19. delphi_lsp/delphi_lsp_native.pyd +0 -0
  20. delphi_lsp/delphiast_lexer.py +399 -0
  21. delphi_lsp/delphiast_parser.py +2189 -0
  22. delphi_lsp/delphiast_tokens.py +41 -0
  23. delphi_lsp/grammar.py +459 -0
  24. delphi_lsp/incremental.py +1139 -0
  25. delphi_lsp/lark_builder.py +2684 -0
  26. delphi_lsp/lark_tokens.py +236 -0
  27. delphi_lsp/lexical_scanner.py +30 -0
  28. delphi_lsp/lsp_server.py +6698 -0
  29. delphi_lsp/metrics.py +1452 -0
  30. delphi_lsp/native_bridge.py +300 -0
  31. delphi_lsp/navigation_cache.py +189 -0
  32. delphi_lsp/nodes.py +462 -0
  33. delphi_lsp/parallel_outline.py +662 -0
  34. delphi_lsp/parse_artifacts.py +969 -0
  35. delphi_lsp/parser.py +342 -0
  36. delphi_lsp/parser_backend.py +139 -0
  37. delphi_lsp/preprocessor.py +1122 -0
  38. delphi_lsp/progress.py +26 -0
  39. delphi_lsp/project_config.py +285 -0
  40. delphi_lsp/project_discovery.py +932 -0
  41. delphi_lsp/project_indexer.py +470 -0
  42. delphi_lsp/py.typed +1 -0
  43. delphi_lsp/response_cache.py +465 -0
  44. delphi_lsp/semantic.py +394 -0
  45. delphi_lsp/semantic_builder.py +2302 -0
  46. delphi_lsp/source_reader.py +17 -0
  47. delphi_lsp/wiki_external_sort.py +392 -0
  48. delphi_lsp/wiki_memory.py +548 -0
  49. delphi_lsp/workspace.py +83 -0
  50. delphi_lsp/writer.py +73 -0
  51. python_delphi_lsp-3.6.0.dist-info/METADATA +766 -0
  52. python_delphi_lsp-3.6.0.dist-info/RECORD +56 -0
  53. python_delphi_lsp-3.6.0.dist-info/WHEEL +4 -0
  54. python_delphi_lsp-3.6.0.dist-info/entry_points.txt +3 -0
  55. python_delphi_lsp-3.6.0.dist-info/licenses/LICENSE +373 -0
  56. python_delphi_lsp-3.6.0.dist-info/sboms/python-delphi-lsp.cyclonedx.json +770 -0
@@ -0,0 +1,3907 @@
1
+ from __future__ import annotations
2
+
3
+ import gc
4
+ import hashlib
5
+ import json
6
+ import re
7
+ import sys
8
+ import time
9
+ import unicodedata
10
+ from bisect import bisect_left, bisect_right
11
+ from collections import OrderedDict
12
+ from collections.abc import Callable, Mapping, Sequence
13
+ from dataclasses import dataclass, field, replace
14
+ from functools import lru_cache
15
+ from heapq import heappop, heappush
16
+ from pathlib import Path, PureWindowsPath
17
+
18
+ from .agent_cpg import CpgSubgraph, CpgTarget
19
+ from .agent_cpg_builder import build_cpg_subgraph
20
+ from .agent_metrics import (
21
+ build_workspace_metrics,
22
+ project_metric_item,
23
+ unit_metric_item,
24
+ )
25
+ from .agent_protocol import (
26
+ AgentProtocolError,
27
+ AgentRequest,
28
+ AgentResponse,
29
+ ContextBudget,
30
+ Focus,
31
+ make_target_id,
32
+ paginate_items,
33
+ )
34
+ from .agent_relations import ProjectRelationIndex, RelationTarget
35
+ from .agent_workspace import (
36
+ AgentUnit,
37
+ AgentWorkspace,
38
+ unit_display_path,
39
+ unit_source_path,
40
+ unit_target_id,
41
+ )
42
+ from .consts import AttributeName, SyntaxNodeType
43
+ from .lsp_server import build_outline_semantic_model, multiline_string_block_end
44
+ from .metrics import ProjectMetrics
45
+ from .navigation_cache import NavigationShardStore, navigation_cache_key
46
+ from .nodes import CompoundSyntaxNode, SyntaxNode
47
+ from .parallel_outline import (
48
+ ParallelBuildStats,
49
+ ParallelOutlineError,
50
+ run_outline_tasks,
51
+ )
52
+ from .parser import DelphiParser
53
+ from .parser_backend import ParserMode
54
+ from .project_config import ProjectPathConfig, workspace_include_loader
55
+ from .semantic import (
56
+ Scope,
57
+ ScopeKind,
58
+ Symbol,
59
+ SymbolKind,
60
+ Visibility,
61
+ )
62
+ from .source_reader import read_source_text
63
+
64
+ _ROUTINE_KINDS = frozenset(
65
+ {
66
+ SymbolKind.METHOD,
67
+ SymbolKind.FUNCTION,
68
+ SymbolKind.PROCEDURE,
69
+ SymbolKind.CONSTRUCTOR,
70
+ SymbolKind.DESTRUCTOR,
71
+ }
72
+ )
73
+ _TYPE_KINDS = frozenset(
74
+ {
75
+ SymbolKind.TYPE,
76
+ SymbolKind.CLASS,
77
+ SymbolKind.RECORD,
78
+ SymbolKind.INTERFACE,
79
+ SymbolKind.ENUM,
80
+ }
81
+ )
82
+ _ROUTINE_WORDS = frozenset(
83
+ {"procedure", "function", "constructor", "destructor", "operator"}
84
+ )
85
+ _BLOCK_WORDS = frozenset({"begin", "case", "try", "asm"})
86
+ _STRUCTURED_TYPE_WORDS = frozenset(
87
+ {"class", "record", "object", "interface", "dispinterface"}
88
+ )
89
+ _NO_BODY_DIRECTIVES = frozenset({"abstract", "external", "forward"})
90
+ _ROUTINE_DIRECTIVES = frozenset(
91
+ {
92
+ "abstract",
93
+ "assembler",
94
+ "cdecl",
95
+ "deprecated",
96
+ "dispid",
97
+ "dynamic",
98
+ "experimental",
99
+ "export",
100
+ "external",
101
+ "final",
102
+ "forward",
103
+ "inline",
104
+ "message",
105
+ "noreturn",
106
+ "overload",
107
+ "override",
108
+ "pascal",
109
+ "platform",
110
+ "register",
111
+ "reintroduce",
112
+ "safecall",
113
+ "static",
114
+ "stdcall",
115
+ "unsafe",
116
+ "varargs",
117
+ "virtual",
118
+ "winapi",
119
+ }
120
+ )
121
+ _CALLING_CONVENTIONS = frozenset(
122
+ {"cdecl", "pascal", "register", "safecall", "stdcall", "winapi"}
123
+ )
124
+ _SOURCE_CHUNK_CHARS = 6000
125
+ _RANKED_QUERY_CACHE_SIZE = 16
126
+ _RANKED_QUERY_CACHE_MAX_ENTRIES = 50_000
127
+ _PREPARED_RESPONSE_CACHE_SIZE = 8
128
+ _DEFAULT_NAVIGATION_CACHE_MAX_BYTES = 512 * 1024**2
129
+
130
+
131
+ @dataclass(frozen=True, slots=True)
132
+ class _Token:
133
+ value: str
134
+ start: int
135
+ end: int
136
+ word: bool = False
137
+ escaped: bool = False
138
+ directive: bool = False
139
+
140
+
141
+ class _SourceDocument:
142
+ def __init__(
143
+ self,
144
+ source_path: Path,
145
+ display_path: str,
146
+ text: str,
147
+ *,
148
+ defines: tuple[str, ...] = (),
149
+ include_paths: tuple[str, ...] = (),
150
+ project_config: ProjectPathConfig | None = None,
151
+ ) -> None:
152
+ self.source_path = source_path
153
+ self.display_path = display_path
154
+ self.text = text
155
+ self.defines = defines
156
+ self.include_paths = include_paths
157
+ self.project_config = project_config
158
+ self.line_starts = _line_starts(text)
159
+ self.tokens = tuple(_lex_delphi(text))
160
+ self.token_starts = tuple(token.start for token in self.tokens)
161
+ self.directive_starts = tuple(token.start for token in self.tokens if token.directive)
162
+ self.declaration_section_indexes = [0]
163
+ self.declaration_section_checkpoints: dict[int, tuple[str, int, int, int]] = {
164
+ 0: ("", 0, 0, 0)
165
+ }
166
+ words = [token for token in self.tokens if token.word and not token.escaped]
167
+ self.unit_kind = next(
168
+ (token.value for token in words if token.value in {"unit", "program", "library", "package"}),
169
+ "",
170
+ )
171
+ implementation = next((token for token in words if token.value == "implementation"), None)
172
+ self.implementation_line = self.line_col(implementation.start)[0] if implementation else 0
173
+ self.routine_spans: dict[int, tuple[int, int] | None] = {}
174
+ self.routine_token_spans: dict[int, tuple[int, int, int] | None] = {}
175
+ self.parser_spans: dict[str, tuple[int, int] | None] = {}
176
+ self._full_parse_attempted = False
177
+ self._full_parse_result: object | None = None
178
+ self.retained_bytes = (
179
+ sys.getsizeof(self)
180
+ + sys.getsizeof(self.text)
181
+ + sys.getsizeof(self.line_starts)
182
+ + len(self.line_starts) * 32
183
+ + sys.getsizeof(self.tokens)
184
+ + len(self.tokens) * 160
185
+ + sys.getsizeof(self.token_starts)
186
+ + len(self.token_starts) * 28
187
+ + sys.getsizeof(self.directive_starts)
188
+ + len(self.directive_starts) * 28
189
+ )
190
+
191
+ def offset(self, line: int, column: int = 1) -> int:
192
+ if not self.line_starts:
193
+ return 0
194
+ line_index = min(max(line - 1, 0), len(self.line_starts) - 1)
195
+ start = self.line_starts[line_index]
196
+ end = self.line_starts[line_index + 1] if line_index + 1 < len(self.line_starts) else len(self.text)
197
+ return min(start + max(column - 1, 0), end)
198
+
199
+ def line_start(self, line: int) -> int:
200
+ return self.offset(line, 1)
201
+
202
+ def line_end(self, line: int, *, include_newline: bool = False) -> int:
203
+ if line < len(self.line_starts):
204
+ end = self.line_starts[max(line, 0)]
205
+ else:
206
+ end = len(self.text)
207
+ if include_newline:
208
+ return end
209
+ while end > 0 and self.text[end - 1] in {"\r", "\n"}:
210
+ end -= 1
211
+ return end
212
+
213
+ def line_col(self, offset: int) -> tuple[int, int]:
214
+ clamped = min(max(offset, 0), len(self.text))
215
+ line_index = max(0, bisect_right(self.line_starts, clamped) - 1)
216
+ return line_index + 1, clamped - self.line_starts[line_index] + 1
217
+
218
+ def first_token_index(self, offset: int) -> int:
219
+ return bisect_left(self.token_starts, offset)
220
+
221
+ def contains_directive(self, start: int, end: int) -> bool:
222
+ index = bisect_left(self.directive_starts, start)
223
+ return index < len(self.directive_starts) and self.directive_starts[index] < end
224
+
225
+ def full_parse(self):
226
+ if not self._full_parse_attempted:
227
+ self._full_parse_attempted = True
228
+ try:
229
+ self._full_parse_result = DelphiParser(
230
+ defines=self.defines,
231
+ include_paths=self.include_paths,
232
+ include_loader=workspace_include_loader(
233
+ self.project_config,
234
+ self.include_paths,
235
+ ),
236
+ ).parse(self.text, str(self.source_path), build_semantic=False)
237
+ except Exception:
238
+ self._full_parse_result = None
239
+ return self._full_parse_result
240
+
241
+
242
+ @dataclass(frozen=True, slots=True)
243
+ class _SourceSpec:
244
+ source_path: Path
245
+ display_path: str
246
+ defines: tuple[str, ...]
247
+ include_paths: tuple[str, ...]
248
+ project_config: ProjectPathConfig | None = None
249
+
250
+
251
+ class _SourceStore:
252
+ """Load expensive tokenized source documents only when source evidence is requested."""
253
+
254
+ def __init__(
255
+ self,
256
+ specs: Mapping[Path, _SourceSpec],
257
+ *,
258
+ max_loaded_bytes: int,
259
+ ) -> None:
260
+ self._specs = dict(specs)
261
+ self._loaded: OrderedDict[Path, _SourceDocument] = OrderedDict()
262
+ self._loaded_bytes = 0
263
+ self._max_loaded_bytes = max(0, max_loaded_bytes)
264
+
265
+ def __getitem__(self, source_path: Path) -> _SourceDocument:
266
+ cached = self._loaded.pop(source_path, None)
267
+ if cached is not None:
268
+ self._loaded[source_path] = cached
269
+ return cached
270
+
271
+ spec = self._specs[source_path]
272
+ document = _SourceDocument(
273
+ spec.source_path,
274
+ spec.display_path,
275
+ read_source_text(spec.source_path),
276
+ defines=spec.defines,
277
+ include_paths=spec.include_paths,
278
+ project_config=spec.project_config,
279
+ )
280
+ if document.retained_bytes > self._max_loaded_bytes:
281
+ return document
282
+ while (
283
+ self._loaded
284
+ and self._loaded_bytes + document.retained_bytes > self._max_loaded_bytes
285
+ ):
286
+ _, evicted = self._loaded.popitem(last=False)
287
+ self._loaded_bytes -= evicted.retained_bytes
288
+ self._loaded[source_path] = document
289
+ self._loaded_bytes += document.retained_bytes
290
+ return document
291
+
292
+ def add_spec(self, spec: _SourceSpec) -> None:
293
+ self._specs.setdefault(spec.source_path, spec)
294
+
295
+ @property
296
+ def loaded_count(self) -> int:
297
+ return len(self._loaded)
298
+
299
+ @property
300
+ def retained_bytes(self) -> int:
301
+ return self._loaded_bytes
302
+
303
+ @property
304
+ def metadata_bytes(self) -> int:
305
+ return 512 + sum(
306
+ 256
307
+ + sys.getsizeof(path)
308
+ + sys.getsizeof(spec)
309
+ + sys.getsizeof(spec.display_path)
310
+ for path, spec in self._specs.items()
311
+ )
312
+
313
+ def clear_loaded(self) -> None:
314
+ self._loaded.clear()
315
+ self._loaded_bytes = 0
316
+
317
+
318
+ @dataclass(frozen=True, slots=True)
319
+ class _RawSymbol:
320
+ name: str
321
+ kind: SymbolKind
322
+ line: int
323
+ column: int
324
+ visibility: Visibility
325
+ type_name: str
326
+ source_path: Path
327
+ path: str
328
+ unit_id: str
329
+ unit_name: str
330
+ qualified_name: str
331
+ owner: str
332
+ parent_qualified_name: str
333
+ signature: str
334
+ context_ambiguous: bool = False
335
+
336
+
337
+ @dataclass(frozen=True, slots=True)
338
+ class _NavigationTask:
339
+ ordinal: int
340
+ source_path: str
341
+ display_path: str
342
+ unit_name: str
343
+ unit_path: str
344
+ unit_id: str
345
+ unit_has_error: bool
346
+ defines: tuple[str, ...]
347
+ include_paths: tuple[str, ...]
348
+ project_config: ProjectPathConfig | None = None
349
+ exact_name: str = ""
350
+ cache_key: str = ""
351
+
352
+
353
+ @dataclass(frozen=True, slots=True)
354
+ class _NavigationResult:
355
+ ordinal: int
356
+ source_path: str
357
+ text: str
358
+ model: None
359
+ lines_processed: int
360
+ symbols_discovered: int
361
+ read_error: str
362
+ raw_symbols: tuple[_RawSymbol, ...]
363
+ cache_key: str = ""
364
+
365
+
366
+ _THIN_UNIT_WRAPPER_RE = re.compile(
367
+ r"\A\s*unit\s+[\w.&]+\s*;\s*(?P<directives>(?:\{\$[^}]+\}\s*)+)\Z",
368
+ re.IGNORECASE,
369
+ )
370
+ _WRAPPER_DIRECTIVE_RE = re.compile(r"\{\$\s*(\w+)\s*([^}]*)\}", re.IGNORECASE)
371
+
372
+
373
+ def _thin_unit_include(
374
+ task: _NavigationTask,
375
+ text: str,
376
+ ) -> tuple[Path, str, str, tuple[str, ...]] | None:
377
+ match = _THIN_UNIT_WRAPPER_RE.fullmatch(text)
378
+ if match is None:
379
+ return None
380
+ directives = _WRAPPER_DIRECTIVE_RE.findall(match.group("directives"))
381
+ includes = [
382
+ param.strip().strip("'\"")
383
+ for name, param in directives
384
+ if name.casefold() in {"i", "include"}
385
+ ]
386
+ if len(includes) != 1 or not includes[0]:
387
+ return None
388
+ include_name = includes[0]
389
+ search_roots = (
390
+ Path(task.source_path).expanduser().resolve().parent,
391
+ *(Path(path).expanduser().resolve() for path in task.include_paths),
392
+ )
393
+ include_path = Path(include_name.replace("\\", "/"))
394
+ for root in search_roots:
395
+ candidate = (root / include_path).resolve()
396
+ if (
397
+ task.project_config is not None
398
+ and task.project_config.excludes_workspace_path(candidate)
399
+ ):
400
+ continue
401
+ try:
402
+ include_text = read_source_text(candidate)
403
+ except (OSError, UnicodeError):
404
+ continue
405
+ define_directives = (
406
+ param.strip().split(None, 1)[0]
407
+ for name, param in directives
408
+ if name.casefold() == "define" and param.strip()
409
+ )
410
+ include_defines = tuple(dict.fromkeys((*task.defines, *define_directives)))
411
+ return (
412
+ candidate,
413
+ _related_display_path(task, candidate),
414
+ include_text,
415
+ include_defines,
416
+ )
417
+ return None
418
+
419
+
420
+ def _related_display_path(task: _NavigationTask, source_path: Path) -> str:
421
+ display_parts = Path(task.display_path).parts
422
+ root = Path(task.source_path).expanduser().resolve()
423
+ for _part in display_parts:
424
+ root = root.parent
425
+ try:
426
+ return source_path.relative_to(root).as_posix()
427
+ except ValueError:
428
+ return source_path.name
429
+
430
+
431
+ def _navigation_cache_text(
432
+ text: str,
433
+ wrapper: tuple[Path, str, str, tuple[str, ...]] | None,
434
+ ) -> str:
435
+ if wrapper is None:
436
+ return text
437
+ _include_path, include_display_path, include_text, _include_defines = wrapper
438
+ return (
439
+ text
440
+ + "\0thin-unit-include\0"
441
+ + include_display_path
442
+ + "\0"
443
+ + include_text
444
+ )
445
+
446
+
447
+ def _parse_navigation_task(task: _NavigationTask) -> _NavigationResult:
448
+ source_path = Path(task.source_path)
449
+ try:
450
+ text = read_source_text(source_path)
451
+ except (OSError, UnicodeError) as error:
452
+ return _NavigationResult(
453
+ task.ordinal,
454
+ task.source_path,
455
+ "",
456
+ None,
457
+ 0,
458
+ 0,
459
+ str(error),
460
+ (),
461
+ "",
462
+ )
463
+ try:
464
+ wrapper = _thin_unit_include(task, text)
465
+ model = build_outline_semantic_model(
466
+ text,
467
+ task.source_path,
468
+ defines=task.defines,
469
+ _expand_thin_wrapper=wrapper is None,
470
+ )
471
+ document = _SourceDocument(
472
+ source_path,
473
+ task.display_path,
474
+ text,
475
+ defines=task.defines,
476
+ include_paths=task.include_paths,
477
+ project_config=task.project_config,
478
+ )
479
+ unit = AgentUnit(
480
+ unit_id=task.unit_id,
481
+ name=task.unit_name,
482
+ path=task.unit_path,
483
+ has_error=task.unit_has_error,
484
+ )
485
+ symbols = _collect_raw_symbols(model.unit_scope, unit, source_path, document)
486
+ raw_symbols = list(_exclude_routine_locals(symbols, document))
487
+ cache_text = text
488
+ if wrapper is not None:
489
+ include_path, include_display_path, include_text, include_defines = wrapper
490
+ include_model = build_outline_semantic_model(
491
+ include_text,
492
+ str(include_path),
493
+ defines=include_defines,
494
+ )
495
+ include_document = _SourceDocument(
496
+ include_path,
497
+ include_display_path,
498
+ include_text,
499
+ defines=include_defines,
500
+ include_paths=task.include_paths,
501
+ project_config=task.project_config,
502
+ )
503
+ included = _collect_raw_symbols(
504
+ include_model.unit_scope,
505
+ unit,
506
+ include_path,
507
+ include_document,
508
+ )
509
+ raw_symbols.extend(
510
+ replace(
511
+ _raw_symbol_with_unit_name(raw, task.unit_name),
512
+ unit_id=task.unit_id,
513
+ )
514
+ for raw in _exclude_routine_locals(included, include_document)
515
+ if raw.kind != SymbolKind.UNIT
516
+ )
517
+ cache_text = _navigation_cache_text(text, wrapper)
518
+ if task.exact_name and not any(
519
+ _normalized(symbol.name) == task.exact_name
520
+ for symbol in raw_symbols
521
+ ):
522
+ return _NavigationResult(
523
+ task.ordinal,
524
+ task.source_path,
525
+ "",
526
+ None,
527
+ text.count("\n")
528
+ + (0 if not text or text.endswith(("\n", "\r")) else 1),
529
+ 0,
530
+ "",
531
+ (),
532
+ "",
533
+ )
534
+ except Exception as error:
535
+ raise ParallelOutlineError(
536
+ f"failed to build navigation shard for {task.source_path}: {error}"
537
+ ) from error
538
+ return _NavigationResult(
539
+ task.ordinal,
540
+ task.source_path,
541
+ "",
542
+ None,
543
+ text.count("\n") + (0 if not text or text.endswith(("\n", "\r")) else 1),
544
+ len(raw_symbols),
545
+ "",
546
+ tuple(raw_symbols),
547
+ navigation_cache_key(cache_text, task.defines) if task.cache_key else "",
548
+ )
549
+
550
+
551
+ def _navigation_shard_payload(result: _NavigationResult) -> dict[str, object]:
552
+ unit_name = result.raw_symbols[0].unit_name if result.raw_symbols else ""
553
+ symbols: list[dict[str, object]] = []
554
+ for raw in result.raw_symbols:
555
+ record: dict[str, object] = {
556
+ "name": raw.name,
557
+ "kind": raw.kind.value,
558
+ "source_path": str(raw.source_path),
559
+ "path": raw.path,
560
+ "line": raw.line,
561
+ "column": raw.column,
562
+ "visibility": raw.visibility.value,
563
+ "type": raw.type_name,
564
+ "qualified_name": raw.qualified_name,
565
+ "owner": raw.owner,
566
+ "parent_qualified_name": raw.parent_qualified_name,
567
+ "signature": raw.signature,
568
+ }
569
+ if raw.context_ambiguous:
570
+ record["context_ambiguous"] = True
571
+ symbols.append(record)
572
+ return {
573
+ "lines_processed": result.lines_processed,
574
+ "unit_name": unit_name,
575
+ "symbols": symbols,
576
+ }
577
+
578
+
579
+ def _navigation_result_from_shard(
580
+ task: _NavigationTask,
581
+ payload: Mapping[str, object],
582
+ ) -> _NavigationResult | None:
583
+ try:
584
+ lines_processed = _required_int(payload.get("lines_processed"))
585
+ unit_name = _required_string(payload.get("unit_name"))
586
+ records = payload.get("symbols")
587
+ if not isinstance(records, list):
588
+ return None
589
+ unit_id = make_target_id("unit", task.display_path, task.unit_name)
590
+ raw_symbols: list[_RawSymbol] = []
591
+ for value in records:
592
+ if not isinstance(value, Mapping):
593
+ return None
594
+ raw_symbols.append(
595
+ _RawSymbol(
596
+ name=_required_string(value.get("name")),
597
+ kind=SymbolKind(_required_string(value.get("kind"))),
598
+ line=_required_int(value.get("line")),
599
+ column=_required_int(value.get("column")),
600
+ visibility=Visibility(_required_string(value.get("visibility"))),
601
+ type_name=_required_string(value.get("type")),
602
+ source_path=Path(
603
+ _required_string(value.get("source_path"))
604
+ if value.get("source_path") is not None
605
+ else task.source_path
606
+ ),
607
+ path=(
608
+ _required_string(value.get("path"))
609
+ if value.get("path") is not None
610
+ else task.display_path
611
+ ),
612
+ unit_id=unit_id,
613
+ unit_name=unit_name or task.unit_name,
614
+ qualified_name=_required_string(value.get("qualified_name")),
615
+ owner=_required_string(value.get("owner")),
616
+ parent_qualified_name=_required_string(
617
+ value.get("parent_qualified_name")
618
+ ),
619
+ signature=_required_string(value.get("signature")),
620
+ context_ambiguous=value.get("context_ambiguous") is True,
621
+ )
622
+ )
623
+ except (TypeError, ValueError):
624
+ return None
625
+ return _NavigationResult(
626
+ ordinal=task.ordinal,
627
+ source_path=task.source_path,
628
+ text="",
629
+ model=None,
630
+ lines_processed=lines_processed,
631
+ symbols_discovered=len(raw_symbols),
632
+ read_error="",
633
+ raw_symbols=tuple(raw_symbols),
634
+ )
635
+
636
+
637
+ def _required_string(value: object) -> str:
638
+ if not isinstance(value, str):
639
+ raise TypeError("navigation string is malformed")
640
+ return value
641
+
642
+
643
+ def _required_int(value: object) -> int:
644
+ if isinstance(value, bool) or not isinstance(value, int):
645
+ raise TypeError("navigation integer is malformed")
646
+ return value
647
+
648
+
649
+ @dataclass(frozen=True, slots=True)
650
+ class _SymbolEntry:
651
+ name: str
652
+ kind: SymbolKind
653
+ line: int
654
+ column: int
655
+ visibility: Visibility
656
+ type_name: str
657
+ source_path: Path
658
+ path: str
659
+ unit_id: str
660
+ unit_name: str
661
+ qualified_name: str
662
+ normalized_name: str
663
+ normalized_qualified_name: str
664
+ relative_name_offset: int
665
+ owner: str
666
+ signature: str
667
+ ordinal: int
668
+ target_id: str
669
+ card_json_chars: int
670
+ card_json_upper_bound: int
671
+ parent_target_id: str = ""
672
+ context_ambiguous: bool = False
673
+
674
+ def card(self) -> dict[str, object]:
675
+ card: dict[str, object] = {
676
+ "target_id": self.target_id,
677
+ "unit_id": self.unit_id,
678
+ "name": self.name,
679
+ "qualified_name": self.qualified_name,
680
+ "kind": self.kind.value,
681
+ "path": self.path,
682
+ "line": self.line,
683
+ "column": self.column,
684
+ "visibility": self.visibility.value,
685
+ "owner": self.owner,
686
+ "type": self.type_name,
687
+ }
688
+ if self.context_ambiguous:
689
+ card["context_ambiguous"] = True
690
+ return card
691
+
692
+
693
+ class _SymbolCardSequence(Sequence[dict[str, object]]):
694
+ def __init__(
695
+ self,
696
+ entries: Sequence[_SymbolEntry],
697
+ max_chars: int,
698
+ ) -> None:
699
+ self._entries = entries
700
+ self._max_chars = max_chars
701
+ self._chunk_payload_chars = _card_chunk_payload_chars(max_chars)
702
+ self._ends: list[int] = []
703
+ total = 0
704
+ for entry in entries:
705
+ if (
706
+ entry.card_json_upper_bound + 2 <= max_chars
707
+ or entry.card_json_chars + 2 <= max_chars
708
+ ):
709
+ count = 1
710
+ else:
711
+ count = (
712
+ entry.card_json_chars + self._chunk_payload_chars - 1
713
+ ) // self._chunk_payload_chars
714
+ total += count
715
+ self._ends.append(total)
716
+
717
+ def __len__(self) -> int:
718
+ return self._ends[-1] if self._ends else 0
719
+
720
+ def __getitem__(
721
+ self,
722
+ index: int | slice,
723
+ ) -> dict[str, object] | list[dict[str, object]]:
724
+ if isinstance(index, slice):
725
+ return [self[item_index] for item_index in range(*index.indices(len(self)))]
726
+ if index < 0:
727
+ index += len(self)
728
+ if index < 0 or index >= len(self):
729
+ raise IndexError(index)
730
+ entry_index = bisect_right(self._ends, index)
731
+ entry = self._entries[entry_index]
732
+ previous_end = self._ends[entry_index - 1] if entry_index else 0
733
+ chunk_index = index - previous_end
734
+ if (
735
+ entry.card_json_upper_bound + 2 <= self._max_chars
736
+ or entry.card_json_chars + 2 <= self._max_chars
737
+ ):
738
+ return entry.card()
739
+ serialized = _compact_json(entry.card())
740
+ chunk_count = self._ends[entry_index] - previous_end
741
+ start = chunk_index * self._chunk_payload_chars
742
+ return {
743
+ "item_type": "card_chunk",
744
+ "chunk_index": chunk_index,
745
+ "chunk_count": chunk_count,
746
+ "json": serialized[start:start + self._chunk_payload_chars],
747
+ }
748
+
749
+
750
+ def _card_chunk_payload_chars(max_chars: int) -> int:
751
+ wrapper_chars = len(
752
+ _compact_json(
753
+ {
754
+ "item_type": "card_chunk",
755
+ "chunk_index": 999999,
756
+ "chunk_count": 999999,
757
+ "json": "",
758
+ }
759
+ )
760
+ ) + 2
761
+ payload_chars = (max_chars - wrapper_chars) // 2
762
+ if payload_chars < 1:
763
+ raise AgentProtocolError(
764
+ "item_too_large",
765
+ "max_chars is too small for a structured response chunk.",
766
+ )
767
+ return payload_chars
768
+
769
+
770
+ @dataclass(slots=True)
771
+ class _Registry:
772
+ project_id: str
773
+ revision: str
774
+ entries: tuple[_SymbolEntry, ...]
775
+ by_target: dict[str, _SymbolEntry]
776
+ sources: _SourceStore
777
+ ranked_queries: OrderedDict[str, tuple[_SymbolEntry, ...]]
778
+ static_retained_bytes: int
779
+ indexed_include_paths: set[Path] = field(default_factory=set)
780
+ include_queries: set[str] = field(default_factory=set)
781
+
782
+
783
+ class _CpgCandidates(Mapping[str, tuple[CpgTarget, ...]]):
784
+ def __init__(self, entries: Sequence[_SymbolEntry]) -> None:
785
+ self._entries = entries
786
+ self._resolved: dict[str, tuple[CpgTarget, ...]] = {}
787
+
788
+ def __getitem__(self, name: str) -> tuple[CpgTarget, ...]:
789
+ key = _normalized(name)
790
+ cached = self._resolved.get(key)
791
+ if cached is not None:
792
+ return cached
793
+ resolved = tuple(
794
+ _cpg_target(entry)
795
+ for entry in self._entries
796
+ if entry.kind in _ROUTINE_KINDS
797
+ and (
798
+ entry.normalized_name == key
799
+ or entry.normalized_qualified_name == key
800
+ )
801
+ )
802
+ self._resolved[key] = resolved
803
+ return resolved
804
+
805
+ def __iter__(self):
806
+ return iter(self._resolved)
807
+
808
+ def __len__(self) -> int:
809
+ return len(self._resolved)
810
+
811
+ def resolve_many(
812
+ self,
813
+ names: set[str],
814
+ ) -> dict[str, tuple[CpgTarget, ...]]:
815
+ keys = {_normalized(name) for name in names}
816
+ missing = keys.difference(self._resolved)
817
+ if missing:
818
+ collected: dict[str, list[CpgTarget]] = {
819
+ name: [] for name in missing
820
+ }
821
+ for entry in self._entries:
822
+ if entry.kind not in _ROUTINE_KINDS:
823
+ continue
824
+ name_match = entry.normalized_name in missing
825
+ qualified_match = (
826
+ entry.normalized_qualified_name in missing
827
+ and entry.normalized_qualified_name
828
+ != entry.normalized_name
829
+ )
830
+ if not name_match and not qualified_match:
831
+ continue
832
+ target = _cpg_target(entry)
833
+ if name_match:
834
+ collected[entry.normalized_name].append(target)
835
+ if qualified_match:
836
+ collected[entry.normalized_qualified_name].append(target)
837
+ self._resolved.update(
838
+ (name, tuple(matches))
839
+ for name, matches in collected.items()
840
+ )
841
+ return {
842
+ name: self._resolved.get(name, ())
843
+ for name in keys
844
+ }
845
+
846
+
847
+ def _cpg_target(entry: _SymbolEntry) -> CpgTarget:
848
+ return CpgTarget(
849
+ target_id=entry.target_id,
850
+ source_path=str(entry.source_path),
851
+ path=entry.path,
852
+ unit_id=entry.unit_id,
853
+ name=entry.name,
854
+ qualified_name=entry.qualified_name,
855
+ kind=entry.kind.value,
856
+ line=entry.line,
857
+ column=entry.column,
858
+ visibility=entry.visibility.value,
859
+ type_name=entry.type_name,
860
+ owner=entry.owner,
861
+ parent_target_id=entry.parent_target_id,
862
+ )
863
+
864
+
865
+ def _relation_target(entry: _SymbolEntry) -> RelationTarget:
866
+ return RelationTarget(
867
+ target_id=entry.target_id,
868
+ source_path=str(entry.source_path),
869
+ path=entry.path,
870
+ unit_id=entry.unit_id,
871
+ unit_name=entry.unit_name,
872
+ name=entry.name,
873
+ qualified_name=entry.qualified_name,
874
+ kind=entry.kind.value,
875
+ signature=entry.signature,
876
+ line=entry.line,
877
+ column=entry.column,
878
+ card={},
879
+ visibility=entry.visibility.value,
880
+ type_name=entry.type_name,
881
+ owner=entry.owner,
882
+ )
883
+
884
+
885
+ class AgentContext:
886
+ def __init__(
887
+ self,
888
+ workspace: AgentWorkspace,
889
+ *,
890
+ workers: int = 0,
891
+ worker_memory_budget_bytes: int | None = None,
892
+ revision_check_interval_seconds: float = 0.0,
893
+ navigation_cache_dir: str | Path | None = None,
894
+ navigation_cache_max_bytes: int = _DEFAULT_NAVIGATION_CACHE_MAX_BYTES,
895
+ ) -> None:
896
+ self._workspace = workspace
897
+ self._workers = workers
898
+ self._worker_memory_budget_bytes = worker_memory_budget_bytes
899
+ self._revision_check_interval_seconds = max(
900
+ 0.0,
901
+ revision_check_interval_seconds,
902
+ )
903
+ self._last_revision_check_at = (
904
+ time.monotonic() if self._revision_check_interval_seconds > 0.0 else 0.0
905
+ )
906
+ self._revision_epoch = 0
907
+ self._parallel_stats = ParallelBuildStats(0, 0, 0, 0.0, 0)
908
+ self._navigation_store = (
909
+ NavigationShardStore(navigation_cache_dir)
910
+ if navigation_cache_dir is not None
911
+ else None
912
+ )
913
+ self._navigation_cache_max_bytes = max(0, navigation_cache_max_bytes)
914
+ self._navigation_disk_hits = 0
915
+ self._navigation_disk_misses = 0
916
+ project_id = workspace.active_project_id
917
+ self._focus = Focus(project_id=project_id) if project_id else Focus()
918
+ self._last_revision = workspace.current_revision
919
+ self._registry: _Registry | None = None
920
+ self._relation_index: ProjectRelationIndex | None = None
921
+ self._metrics: ProjectMetrics | None = None
922
+ self._metrics_revision = ""
923
+ cpg_budget_base = (
924
+ worker_memory_budget_bytes
925
+ if worker_memory_budget_bytes is not None
926
+ else 512 * 1024**2
927
+ )
928
+ self._cpg_cache_limit = min(
929
+ 128 * 1024**2,
930
+ max(1, cpg_budget_base // 5),
931
+ )
932
+ self._cpg_cache: OrderedDict[
933
+ tuple[str, str, str, str, int],
934
+ CpgSubgraph,
935
+ ] = OrderedDict()
936
+ self._cpg_cache_bytes = 0
937
+ self._cpg_sources_parsed = 0
938
+ self._prepared_response_cache: OrderedDict[
939
+ tuple[str, str],
940
+ Sequence[dict[str, object]],
941
+ ] = OrderedDict()
942
+
943
+ @classmethod
944
+ def open(
945
+ cls,
946
+ root: str | Path,
947
+ project_file: str | Path | None = None,
948
+ *,
949
+ workers: int = 0,
950
+ worker_memory_budget_bytes: int | None = None,
951
+ revision_check_interval_seconds: float = 0.0,
952
+ reference_query: str = "",
953
+ reference_path: str = "",
954
+ navigation_cache_dir: str | Path | None = None,
955
+ navigation_cache_max_bytes: int = _DEFAULT_NAVIGATION_CACHE_MAX_BYTES,
956
+ ) -> AgentContext:
957
+ context = cls(
958
+ AgentWorkspace.open(root, project_file=project_file),
959
+ workers=workers,
960
+ worker_memory_budget_bytes=worker_memory_budget_bytes,
961
+ revision_check_interval_seconds=revision_check_interval_seconds,
962
+ navigation_cache_dir=navigation_cache_dir,
963
+ navigation_cache_max_bytes=navigation_cache_max_bytes,
964
+ )
965
+ if reference_query and len(context.workspace.units) >= 256:
966
+ context._prepare_reference_registry(
967
+ reference_query,
968
+ source_path=reference_path,
969
+ )
970
+ return context
971
+
972
+ @property
973
+ def workspace(self) -> AgentWorkspace:
974
+ return self._workspace
975
+
976
+ @property
977
+ def navigation_cache_is_warm(self) -> bool:
978
+ return self._registry is not None
979
+
980
+ @property
981
+ def parallel_stats(self) -> ParallelBuildStats:
982
+ return self._parallel_stats
983
+
984
+ @property
985
+ def navigation_disk_hits(self) -> int:
986
+ return self._navigation_disk_hits
987
+
988
+ @property
989
+ def navigation_disk_misses(self) -> int:
990
+ return self._navigation_disk_misses
991
+
992
+ @property
993
+ def cpg_cache_entries(self) -> int:
994
+ return len(self._cpg_cache)
995
+
996
+ @property
997
+ def cpg_cache_bytes(self) -> int:
998
+ return self._cpg_cache_bytes
999
+
1000
+ @property
1001
+ def cpg_sources_parsed(self) -> int:
1002
+ return self._cpg_sources_parsed
1003
+
1004
+ def cache_roots(self) -> tuple[object, ...]:
1005
+ return (
1006
+ *self._workspace.cache_roots(),
1007
+ self._registry,
1008
+ self._relation_index,
1009
+ self._metrics,
1010
+ tuple(self._cpg_cache.values()),
1011
+ tuple(self._prepared_response_cache.values()),
1012
+ )
1013
+
1014
+ @property
1015
+ def estimated_cache_bytes(self) -> int:
1016
+ retained = self._workspace.estimated_cache_bytes
1017
+ if self._registry is not None:
1018
+ retained += (
1019
+ self._registry.static_retained_bytes
1020
+ + self._registry.sources.retained_bytes
1021
+ + sum(
1022
+ sys.getsizeof(query)
1023
+ + sys.getsizeof(ranked)
1024
+ + len(ranked) * 8
1025
+ for query, ranked in self._registry.ranked_queries.items()
1026
+ )
1027
+ )
1028
+ if self._relation_index is not None:
1029
+ retained += self._relation_index.estimated_cache_bytes
1030
+ if self._metrics is not None:
1031
+ retained += 4096 + len(self._metrics.file_metrics) * 1024
1032
+ retained += self._cpg_cache_bytes
1033
+ retained += sum(
1034
+ sys.getsizeof(key)
1035
+ + sys.getsizeof(items)
1036
+ + len(items) * 8
1037
+ for key, items in self._prepared_response_cache.items()
1038
+ )
1039
+ return retained
1040
+
1041
+ def evict_auxiliary_caches(self) -> None:
1042
+ self._relation_index = None
1043
+ self._metrics = None
1044
+ self._metrics_revision = ""
1045
+ self._clear_cpg_cache()
1046
+ self._prepared_response_cache.clear()
1047
+ if self._registry is not None:
1048
+ self._registry.sources.clear_loaded()
1049
+
1050
+ def evict_navigation_caches(self) -> None:
1051
+ self.evict_auxiliary_caches()
1052
+ self._registry = None
1053
+ self._workspace.evict_recomputable_caches()
1054
+
1055
+ def prewarm_navigation(self) -> str:
1056
+ revision_epoch = self._revision_epoch
1057
+ revision = self._refresh_workspace("")
1058
+ self._require_registry(revision)
1059
+ if self._revision_epoch == revision_epoch:
1060
+ self._last_revision_check_at = time.monotonic()
1061
+ return revision
1062
+
1063
+ def _prepare_reference_registry(
1064
+ self,
1065
+ query: str,
1066
+ *,
1067
+ source_path: str = "",
1068
+ ) -> None:
1069
+ project_id = self._require_selected_project()
1070
+ (
1071
+ self._registry,
1072
+ self._parallel_stats,
1073
+ self._navigation_disk_hits,
1074
+ self._navigation_disk_misses,
1075
+ ) = _build_registry(
1076
+ self._workspace,
1077
+ project_id,
1078
+ self._last_revision,
1079
+ workers=self._workers,
1080
+ worker_memory_budget_bytes=self._worker_memory_budget_bytes,
1081
+ navigation_store=self._navigation_store,
1082
+ navigation_cache_max_bytes=self._navigation_cache_max_bytes,
1083
+ exact_query=query,
1084
+ exact_path=source_path,
1085
+ )
1086
+
1087
+ def invalidate_revision_cache(self) -> None:
1088
+ self._revision_epoch += 1
1089
+ self._last_revision_check_at = float("-inf")
1090
+
1091
+ def handle(
1092
+ self,
1093
+ request: AgentRequest | Mapping[str, object],
1094
+ *,
1095
+ cancel_check: Callable[[], None] | None = None,
1096
+ ) -> AgentResponse:
1097
+ parsed = _validated_request(request)
1098
+ if cancel_check is not None:
1099
+ cancel_check()
1100
+ revision = self._refresh_workspace(parsed.project_id)
1101
+
1102
+ if parsed.action == "cpg":
1103
+ registry = self._require_registry(revision)
1104
+ entry = self._resolve_target(registry, parsed.target_id)
1105
+ graph = self._require_cpg_subgraph(registry, entry, parsed)
1106
+ return self._response(
1107
+ parsed,
1108
+ revision,
1109
+ graph.to_items(),
1110
+ target_id=entry.target_id,
1111
+ )
1112
+ if parsed.action == "trace":
1113
+ if cancel_check is not None:
1114
+ cancel_check()
1115
+ if parsed.relation is None:
1116
+ raise AgentProtocolError("relation_required", "Trace requires a relation.")
1117
+ targeted_references = (
1118
+ parsed.relation == "references"
1119
+ and len(self._workspace.units) >= 256
1120
+ )
1121
+ relation_index = (
1122
+ self._relation_index
1123
+ if (
1124
+ targeted_references
1125
+ and self._relation_index is not None
1126
+ and self._relation_index.project_id
1127
+ == self._workspace.active_project_id
1128
+ and self._relation_index.revision == revision
1129
+ and self._relation_index.has_target(parsed.target_id)
1130
+ )
1131
+ else None
1132
+ )
1133
+ resolved_target_id = parsed.target_id
1134
+ if relation_index is None:
1135
+ registry = self._require_registry(revision)
1136
+ entry = self._resolve_target(registry, parsed.target_id)
1137
+ resolved_target_id = entry.target_id
1138
+ if targeted_references:
1139
+ relation_index = self._require_reference_relation_index(
1140
+ registry,
1141
+ entry,
1142
+ )
1143
+ registry.sources.clear_loaded()
1144
+ self._registry = None
1145
+ del registry
1146
+ gc.collect()
1147
+ else:
1148
+ relation_index = self._require_relation_index(
1149
+ registry,
1150
+ parsed.relation,
1151
+ )
1152
+ items = relation_index.trace(
1153
+ resolved_target_id,
1154
+ parsed.relation,
1155
+ cancel_check=cancel_check,
1156
+ max_relations=parsed.max_items,
1157
+ )
1158
+ return self._response(
1159
+ parsed,
1160
+ revision,
1161
+ items,
1162
+ target_id=resolved_target_id,
1163
+ )
1164
+ if parsed.action == "open":
1165
+ items = self._open_items()
1166
+ return self._response(parsed, revision, items)
1167
+ if parsed.action == "problems":
1168
+ self._require_selected_project()
1169
+ items = self._problem_items()
1170
+ return self._response(parsed, revision, items)
1171
+ if parsed.action == "metrics":
1172
+ return self._handle_metrics(
1173
+ parsed,
1174
+ revision,
1175
+ cancel_check=cancel_check,
1176
+ )
1177
+ if parsed.action == "focus":
1178
+ return self._handle_focus(parsed, revision)
1179
+ if parsed.action == "find":
1180
+ registry = self._require_registry(revision)
1181
+ if parsed.query.strip() and not _has_exact_query_match(
1182
+ registry.entries,
1183
+ parsed.query,
1184
+ ):
1185
+ registry_augmented = _augment_registry_with_include_matches(
1186
+ self._workspace,
1187
+ registry,
1188
+ parsed.query,
1189
+ )
1190
+ if registry_augmented:
1191
+ self._prepared_response_cache.clear()
1192
+ ranked = registry.ranked_queries.pop(parsed.query, None)
1193
+ if ranked is None:
1194
+ ranked = (
1195
+ registry.entries
1196
+ if not parsed.query.strip()
1197
+ else tuple(_ranked_entries(registry.entries, parsed.query))
1198
+ )
1199
+ if len(ranked) <= _RANKED_QUERY_CACHE_MAX_ENTRIES:
1200
+ registry.ranked_queries[parsed.query] = ranked
1201
+ while len(registry.ranked_queries) > _RANKED_QUERY_CACHE_SIZE:
1202
+ registry.ranked_queries.popitem(last=False)
1203
+ return self._response(
1204
+ parsed,
1205
+ revision,
1206
+ _SymbolCardSequence(ranked, parsed.max_chars),
1207
+ items_prepared=True,
1208
+ )
1209
+ if parsed.action == "inspect":
1210
+ registry = self._require_registry(revision)
1211
+ entry = self._resolve_target(registry, parsed.target_id)
1212
+ items = self._inspect_items(registry, entry, parsed)
1213
+ return self._response(parsed, revision, items, target_id=entry.target_id)
1214
+ raise AgentProtocolError("invalid_action", f"Unsupported action value: {parsed.action!r}.")
1215
+
1216
+ def _refresh_workspace(self, requested_project_id: str) -> str:
1217
+ previous_project_id = self._workspace.active_project_id
1218
+ selected_project_id = requested_project_id or previous_project_id
1219
+ now = time.monotonic()
1220
+ revision_epoch = self._revision_epoch
1221
+ selection_changed = bool(
1222
+ requested_project_id
1223
+ and requested_project_id != previous_project_id
1224
+ )
1225
+ revision_is_fresh = (
1226
+ not selection_changed
1227
+ and self._revision_check_interval_seconds > 0.0
1228
+ and now - self._last_revision_check_at
1229
+ < self._revision_check_interval_seconds
1230
+ )
1231
+ if revision_is_fresh:
1232
+ revision = self._last_revision
1233
+ elif selected_project_id:
1234
+ revision = self._workspace._select_project_with_revision(selected_project_id)
1235
+ else:
1236
+ revision = self._workspace.workspace_revision
1237
+ if not revision_is_fresh and self._revision_epoch == revision_epoch:
1238
+ self._last_revision_check_at = now
1239
+ current_project_id = self._workspace.active_project_id
1240
+
1241
+ if current_project_id != previous_project_id:
1242
+ self._registry = None
1243
+ self._relation_index = None
1244
+ self._metrics = None
1245
+ self._metrics_revision = ""
1246
+ self._clear_cpg_cache()
1247
+ self._prepared_response_cache.clear()
1248
+ self._focus = Focus(project_id=current_project_id) if current_project_id else Focus()
1249
+ elif revision != self._last_revision:
1250
+ self._registry = None
1251
+ self._relation_index = None
1252
+ self._metrics = None
1253
+ self._metrics_revision = ""
1254
+ self._clear_cpg_cache()
1255
+ self._prepared_response_cache.clear()
1256
+ elif self._focus.project_id != current_project_id:
1257
+ self._focus = Focus(project_id=current_project_id) if current_project_id else Focus()
1258
+ self._last_revision = revision
1259
+ return revision
1260
+
1261
+ def _clear_cpg_cache(self) -> None:
1262
+ self._cpg_cache.clear()
1263
+ self._cpg_cache_bytes = 0
1264
+
1265
+ def _require_cpg_subgraph(
1266
+ self,
1267
+ registry: _Registry,
1268
+ entry: _SymbolEntry,
1269
+ request: AgentRequest,
1270
+ ) -> CpgSubgraph:
1271
+ if entry.kind not in _TYPE_KINDS | _ROUTINE_KINDS | {SymbolKind.UNIT}:
1272
+ raise AgentProtocolError(
1273
+ "cpg_not_applicable",
1274
+ f"CPG does not apply to {entry.kind.value} target {entry.target_id}.",
1275
+ )
1276
+ key = (
1277
+ registry.revision,
1278
+ entry.target_id,
1279
+ request.graph,
1280
+ request.direction,
1281
+ request.depth,
1282
+ )
1283
+ cached = self._cpg_cache.pop(key, None)
1284
+ if cached is not None:
1285
+ self._cpg_cache[key] = cached
1286
+ return cached
1287
+ try:
1288
+ document = registry.sources[entry.source_path]
1289
+ self._cpg_sources_parsed += 1
1290
+ parsed = DelphiParser(
1291
+ defines=document.defines,
1292
+ include_paths=document.include_paths,
1293
+ include_loader=workspace_include_loader(
1294
+ document.project_config,
1295
+ document.include_paths,
1296
+ ),
1297
+ mode=(
1298
+ ParserMode.TOLERANT
1299
+ if len(self._workspace.units) >= 256
1300
+ else ParserMode.STRICT
1301
+ ),
1302
+ ).parse(
1303
+ document.text,
1304
+ str(entry.source_path),
1305
+ build_semantic=False,
1306
+ )
1307
+ graph = build_cpg_subgraph(
1308
+ target=_cpg_target(entry),
1309
+ syntax_root=parsed.root,
1310
+ candidates=_CpgCandidates(registry.entries),
1311
+ graph=request.graph,
1312
+ direction=request.direction,
1313
+ depth=request.depth,
1314
+ )
1315
+ except AgentProtocolError:
1316
+ raise
1317
+ except (OSError, UnicodeError, KeyError):
1318
+ raise AgentProtocolError(
1319
+ "source_unavailable",
1320
+ f"Could not read selected source {entry.path}.",
1321
+ ) from None
1322
+ except Exception:
1323
+ raise AgentProtocolError(
1324
+ "cpg_build_failed",
1325
+ "Could not build the selected CPG subgraph.",
1326
+ ) from None
1327
+ if graph.retained_bytes <= self._cpg_cache_limit:
1328
+ while (
1329
+ self._cpg_cache
1330
+ and self._cpg_cache_bytes + graph.retained_bytes
1331
+ > self._cpg_cache_limit
1332
+ ):
1333
+ _, evicted = self._cpg_cache.popitem(last=False)
1334
+ self._cpg_cache_bytes -= evicted.retained_bytes
1335
+ self._cpg_cache[key] = graph
1336
+ self._cpg_cache_bytes += graph.retained_bytes
1337
+ return graph
1338
+
1339
+ def _open_items(self) -> list[dict[str, object]]:
1340
+ active_project_id = self._workspace.active_project_id
1341
+ items: list[dict[str, object]] = []
1342
+ for project in self._workspace.projects:
1343
+ project_item = _sanitize_workspace_mapping(
1344
+ project.to_mapping(),
1345
+ self._workspace.root,
1346
+ path_namespace="project",
1347
+ )
1348
+ items.append(
1349
+ {
1350
+ "item_type": "project",
1351
+ **project_item,
1352
+ "active": project.project_id == active_project_id,
1353
+ }
1354
+ )
1355
+ if not active_project_id:
1356
+ return items
1357
+
1358
+ for unit in self._workspace.units:
1359
+ display_path = unit_display_path(self._workspace.root, unit)
1360
+ items.append(
1361
+ {
1362
+ "item_type": "unit",
1363
+ "unit_id": unit_target_id(self._workspace.root, unit),
1364
+ "name": unit.name,
1365
+ "path": display_path,
1366
+ "has_error": unit.has_error,
1367
+ }
1368
+ )
1369
+ for include_file in self._workspace.include_files:
1370
+ items.append(
1371
+ {
1372
+ "item_type": "include_file",
1373
+ **_sanitize_workspace_mapping(
1374
+ include_file,
1375
+ self._workspace.root,
1376
+ path_namespace="include",
1377
+ ),
1378
+ }
1379
+ )
1380
+ for entry in self._workspace.search_path_entries:
1381
+ items.append(
1382
+ {
1383
+ "item_type": "search_path",
1384
+ **_sanitize_workspace_mapping(
1385
+ entry,
1386
+ self._workspace.root,
1387
+ path_namespace="search-path",
1388
+ ),
1389
+ }
1390
+ )
1391
+ for entry in self._workspace.include_path_entries:
1392
+ items.append(
1393
+ {
1394
+ "item_type": "include_path",
1395
+ **_sanitize_workspace_mapping(
1396
+ entry,
1397
+ self._workspace.root,
1398
+ path_namespace="include-path",
1399
+ ),
1400
+ }
1401
+ )
1402
+ for entry in self._workspace.define_entries:
1403
+ items.append(
1404
+ {
1405
+ "item_type": "define",
1406
+ **_sanitize_workspace_mapping(
1407
+ entry,
1408
+ self._workspace.root,
1409
+ path_namespace="define",
1410
+ ),
1411
+ }
1412
+ )
1413
+ items.extend(self._problem_items())
1414
+ return items
1415
+
1416
+ def _problem_items(self) -> list[dict[str, object]]:
1417
+ return [
1418
+ {
1419
+ "item_type": "problem",
1420
+ **_sanitize_workspace_mapping(
1421
+ problem,
1422
+ self._workspace.root,
1423
+ path_namespace="problem",
1424
+ ),
1425
+ }
1426
+ for problem in self._workspace.problems
1427
+ ]
1428
+
1429
+ def _handle_focus(self, request: AgentRequest, revision: str) -> AgentResponse:
1430
+ if request.target_id:
1431
+ registry = self._require_registry(revision)
1432
+ entry = self._resolve_target(registry, request.target_id, allow_focused=False)
1433
+ self._focus = Focus(
1434
+ project_id=registry.project_id,
1435
+ unit_id=entry.unit_id,
1436
+ target_id=entry.target_id,
1437
+ )
1438
+ elif request.project_id:
1439
+ self._focus = Focus(project_id=self._workspace.active_project_id)
1440
+ self._require_registry(revision)
1441
+ return self._response(request, revision, [self._focus.to_mapping()])
1442
+
1443
+ def _handle_metrics(
1444
+ self,
1445
+ request: AgentRequest,
1446
+ revision: str,
1447
+ *,
1448
+ cancel_check: Callable[[], None] | None = None,
1449
+ ) -> AgentResponse:
1450
+ if request.detail not in {"summary", "members"}:
1451
+ raise AgentProtocolError(
1452
+ "invalid_detail",
1453
+ "Metrics supports only summary or members detail.",
1454
+ )
1455
+ metrics = self._require_metrics(revision, cancel_check=cancel_check)
1456
+ if cancel_check is not None:
1457
+ cancel_check()
1458
+ detail = request.detail == "members"
1459
+ if request.target_id:
1460
+ unit = next(
1461
+ (
1462
+ candidate
1463
+ for candidate in metrics.file_metrics
1464
+ if candidate.unit_id == request.target_id
1465
+ ),
1466
+ None,
1467
+ )
1468
+ if unit is None:
1469
+ raise AgentProtocolError(
1470
+ "target_not_found",
1471
+ f"Target not found: {request.target_id}.",
1472
+ )
1473
+ return self._response(
1474
+ request,
1475
+ revision,
1476
+ [unit_metric_item(unit, detail=detail)],
1477
+ target_id=unit.unit_id,
1478
+ )
1479
+
1480
+ units = metrics.file_metrics
1481
+ if request.query:
1482
+ query = request.query.casefold()
1483
+ units = tuple(
1484
+ unit
1485
+ for unit in units
1486
+ if query in unit.name.casefold() or query in unit.path.casefold()
1487
+ )
1488
+ items = [unit_metric_item(unit, detail=detail) for unit in units]
1489
+ else:
1490
+ items = [
1491
+ project_metric_item(metrics),
1492
+ *(unit_metric_item(unit, detail=detail) for unit in units),
1493
+ ]
1494
+ return self._response(request, revision, items)
1495
+
1496
+ def _require_metrics(
1497
+ self,
1498
+ revision: str,
1499
+ *,
1500
+ cancel_check: Callable[[], None] | None = None,
1501
+ ) -> ProjectMetrics:
1502
+ self._require_selected_project()
1503
+ if self._metrics is not None and self._metrics_revision == revision:
1504
+ return self._metrics
1505
+ self._metrics = build_workspace_metrics(
1506
+ self._workspace,
1507
+ workers=max(1, self._workers),
1508
+ worker_memory_budget_bytes=self._worker_memory_budget_bytes,
1509
+ cancel_check=cancel_check,
1510
+ )
1511
+ self._metrics_revision = revision
1512
+ return self._metrics
1513
+
1514
+ def _require_registry(self, revision: str) -> _Registry:
1515
+ project_id = self._require_selected_project()
1516
+ if (
1517
+ self._registry is not None
1518
+ and self._registry.project_id == project_id
1519
+ and self._registry.revision == revision
1520
+ ):
1521
+ return self._registry
1522
+ (
1523
+ self._registry,
1524
+ self._parallel_stats,
1525
+ self._navigation_disk_hits,
1526
+ self._navigation_disk_misses,
1527
+ ) = _build_registry(
1528
+ self._workspace,
1529
+ project_id,
1530
+ revision,
1531
+ workers=self._workers,
1532
+ worker_memory_budget_bytes=self._worker_memory_budget_bytes,
1533
+ navigation_store=self._navigation_store,
1534
+ navigation_cache_max_bytes=self._navigation_cache_max_bytes,
1535
+ )
1536
+ if self._focus.target_id:
1537
+ focused_entry = self._registry.by_target.get(self._focus.target_id)
1538
+ if focused_entry is None:
1539
+ self._focus = Focus(project_id=project_id)
1540
+ else:
1541
+ self._focus = Focus(
1542
+ project_id=project_id,
1543
+ unit_id=focused_entry.unit_id,
1544
+ target_id=focused_entry.target_id,
1545
+ )
1546
+ return self._registry
1547
+
1548
+ def _require_relation_index(
1549
+ self,
1550
+ registry: _Registry,
1551
+ relation: str,
1552
+ ) -> ProjectRelationIndex:
1553
+ requires_complete_targets = relation not in {"uses", "used_by"}
1554
+ if (
1555
+ self._relation_index is not None
1556
+ and self._relation_index.project_id == registry.project_id
1557
+ and self._relation_index.revision == registry.revision
1558
+ and (
1559
+ not requires_complete_targets
1560
+ or self._relation_index.targets_complete
1561
+ )
1562
+ and (
1563
+ relation not in {"uses", "used_by"}
1564
+ or self._relation_index.unit_targets_complete
1565
+ )
1566
+ ):
1567
+ return self._relation_index
1568
+ targets = tuple(
1569
+ _relation_target(entry)
1570
+ for entry in registry.entries
1571
+ if requires_complete_targets or entry.kind == SymbolKind.UNIT
1572
+ )
1573
+ self._relation_index = ProjectRelationIndex(
1574
+ self._workspace,
1575
+ registry.project_id,
1576
+ registry.revision,
1577
+ targets,
1578
+ targets_complete=requires_complete_targets,
1579
+ unit_targets_complete=True,
1580
+ )
1581
+ return self._relation_index
1582
+
1583
+ def _require_reference_relation_index(
1584
+ self,
1585
+ registry: _Registry,
1586
+ entry: _SymbolEntry,
1587
+ ) -> ProjectRelationIndex:
1588
+ if (
1589
+ self._relation_index is not None
1590
+ and self._relation_index.project_id == registry.project_id
1591
+ and self._relation_index.revision == registry.revision
1592
+ and self._relation_index.has_target(entry.target_id)
1593
+ ):
1594
+ return self._relation_index
1595
+ targets = tuple(
1596
+ _relation_target(candidate)
1597
+ for candidate in registry.entries
1598
+ if candidate.source_path == entry.source_path
1599
+ )
1600
+ self._relation_index = ProjectRelationIndex(
1601
+ self._workspace,
1602
+ registry.project_id,
1603
+ registry.revision,
1604
+ targets,
1605
+ targets_complete=False,
1606
+ unit_targets_complete=False,
1607
+ )
1608
+ return self._relation_index
1609
+
1610
+ def _require_selected_project(self) -> str:
1611
+ project_id = self._workspace.active_project_id
1612
+ if not project_id:
1613
+ raise AgentProtocolError(
1614
+ "project_required",
1615
+ "Multiple projects were found. Run 'query open', then select one "
1616
+ "with 'query focus --project-id PROJECT_ID'.",
1617
+ )
1618
+ return project_id
1619
+
1620
+ def _resolve_target(
1621
+ self,
1622
+ registry: _Registry,
1623
+ target_id: str,
1624
+ *,
1625
+ allow_focused: bool = True,
1626
+ ) -> _SymbolEntry:
1627
+ resolved_id = target_id
1628
+ if not resolved_id and allow_focused:
1629
+ resolved_id = self._focus.target_id
1630
+ if not resolved_id:
1631
+ raise AgentProtocolError("target_required", "A target_id or focused target is required.")
1632
+ entry = registry.by_target.get(resolved_id)
1633
+ if entry is None:
1634
+ raise AgentProtocolError("target_not_found", f"Target not found: {resolved_id}.")
1635
+ return entry
1636
+
1637
+ def _inspect_items(
1638
+ self,
1639
+ registry: _Registry,
1640
+ entry: _SymbolEntry,
1641
+ request: AgentRequest,
1642
+ ) -> list[dict[str, object]]:
1643
+ if request.detail == "summary":
1644
+ return [entry.card()]
1645
+ if request.detail == "members":
1646
+ return [
1647
+ candidate.card()
1648
+ for candidate in registry.entries
1649
+ if candidate.parent_target_id == entry.target_id
1650
+ ]
1651
+
1652
+ document = registry.sources[entry.source_path]
1653
+ if request.detail == "declaration":
1654
+ start, end = _declaration_span(document, entry)
1655
+ return _source_items(
1656
+ document,
1657
+ start,
1658
+ end,
1659
+ request.max_chars,
1660
+ role="declaration",
1661
+ target_id=entry.target_id,
1662
+ )
1663
+ if request.detail == "context":
1664
+ declaration_start, declaration_end = _declaration_span(document, entry)
1665
+ start_line = max(1, document.line_col(declaration_start)[0] - 3)
1666
+ end_line = document.line_col(declaration_end)[0] + 5
1667
+ start = document.line_start(start_line)
1668
+ end = document.line_end(end_line, include_newline=True)
1669
+ return [entry.card(), *_source_items(
1670
+ document,
1671
+ start,
1672
+ end,
1673
+ request.max_chars,
1674
+ role="context",
1675
+ target_id=entry.target_id,
1676
+ )]
1677
+ if request.detail == "body":
1678
+ body_entry, span = _body_entry_and_span(registry, entry)
1679
+ if span is None:
1680
+ raise AgentProtocolError(
1681
+ "body_unavailable",
1682
+ f"No routine or type body is available for target: {entry.target_id}.",
1683
+ )
1684
+ body_document = registry.sources[body_entry.source_path]
1685
+ return _source_items(
1686
+ body_document,
1687
+ span[0],
1688
+ span[1],
1689
+ request.max_chars,
1690
+ role="body",
1691
+ target_id=body_entry.target_id,
1692
+ )
1693
+ if request.detail == "implementations":
1694
+ items: list[dict[str, object]] = []
1695
+ for counterpart in _matching_counterparts(registry, entry):
1696
+ card = counterpart.card()
1697
+ card["item_type"] = "counterpart"
1698
+ items.append(card)
1699
+ counterpart_document = registry.sources[counterpart.source_path]
1700
+ start, end = _declaration_span(counterpart_document, counterpart)
1701
+ items.extend(
1702
+ _source_items(
1703
+ counterpart_document,
1704
+ start,
1705
+ end,
1706
+ request.max_chars,
1707
+ role="counterpart_declaration",
1708
+ target_id=counterpart.target_id,
1709
+ )
1710
+ )
1711
+ return items
1712
+ raise AgentProtocolError(
1713
+ "invalid_detail",
1714
+ f"Unsupported detail value: {request.detail!r}.",
1715
+ )
1716
+
1717
+ def _response(
1718
+ self,
1719
+ request: AgentRequest,
1720
+ revision: str,
1721
+ items: Sequence[dict[str, object]],
1722
+ *,
1723
+ target_id: str = "",
1724
+ items_prepared: bool = False,
1725
+ ) -> AgentResponse:
1726
+ fingerprint = _request_fingerprint(
1727
+ request,
1728
+ project_id=self._workspace.active_project_id,
1729
+ target_id=target_id or request.target_id,
1730
+ )
1731
+ cache_key = (revision, fingerprint)
1732
+ prepared = self._prepared_response_cache.pop(cache_key, None)
1733
+ if prepared is None:
1734
+ prepared = (
1735
+ items
1736
+ if items_prepared
1737
+ else _prepare_items(items, request.max_chars)
1738
+ )
1739
+ self._prepared_response_cache[cache_key] = prepared
1740
+ while len(self._prepared_response_cache) > _PREPARED_RESPONSE_CACHE_SIZE:
1741
+ self._prepared_response_cache.popitem(last=False)
1742
+ page, selected = paginate_items(
1743
+ prepared,
1744
+ revision,
1745
+ fingerprint,
1746
+ request.max_items,
1747
+ request.max_chars,
1748
+ request.cursor,
1749
+ )
1750
+ context_chars = len(_compact_json(selected))
1751
+ return AgentResponse(
1752
+ workspace_revision=revision,
1753
+ focus=self._focus,
1754
+ result=selected,
1755
+ page=page,
1756
+ context=ContextBudget(chars=context_chars),
1757
+ )
1758
+
1759
+
1760
+ def _validated_request(request: AgentRequest | Mapping[str, object]) -> AgentRequest:
1761
+ if isinstance(request, AgentRequest):
1762
+ return AgentRequest.from_mapping(request.to_mapping())
1763
+ return AgentRequest.from_mapping(request)
1764
+
1765
+
1766
+ def _build_registry(
1767
+ workspace: AgentWorkspace,
1768
+ project_id: str,
1769
+ revision: str,
1770
+ *,
1771
+ workers: int = 0,
1772
+ worker_memory_budget_bytes: int | None = None,
1773
+ navigation_store: NavigationShardStore | None = None,
1774
+ navigation_cache_max_bytes: int = _DEFAULT_NAVIGATION_CACHE_MAX_BYTES,
1775
+ exact_query: str = "",
1776
+ exact_path: str = "",
1777
+ ) -> tuple[_Registry, ParallelBuildStats, int, int]:
1778
+ build_started = time.perf_counter()
1779
+ raw_symbols: list[_RawSymbol] = []
1780
+ units = tuple(workspace.units)
1781
+ exact_needle = _normalized(exact_query.strip())
1782
+ exact_path_needle = exact_path.strip().replace("\\", "/").casefold()
1783
+ include_files = workspace.include_files
1784
+ selected_include_paths = tuple(
1785
+ include["path"]
1786
+ for include in include_files
1787
+ if exact_path_needle
1788
+ and include["path"].replace("\\", "/").casefold()
1789
+ == exact_path_needle
1790
+ )
1791
+ name_needle = exact_needle.rsplit('.', 1)[-1] if exact_needle else ""
1792
+ outline_name_needle = name_needle if "." not in exact_needle else ""
1793
+ tasks = (
1794
+ ()
1795
+ if selected_include_paths
1796
+ else tuple(
1797
+ _NavigationTask(
1798
+ ordinal,
1799
+ str(unit_source_path(workspace.root, unit)),
1800
+ unit_display_path(workspace.root, unit),
1801
+ unit.name,
1802
+ unit.path,
1803
+ unit.unit_id,
1804
+ unit.has_error,
1805
+ workspace.defines,
1806
+ workspace.include_paths,
1807
+ workspace.project_config,
1808
+ outline_name_needle,
1809
+ )
1810
+ for ordinal, unit in enumerate(units)
1811
+ )
1812
+ )
1813
+
1814
+ if exact_path_needle:
1815
+ selected_tasks: tuple[_NavigationTask, ...]
1816
+ if selected_include_paths:
1817
+ selected_tasks = ()
1818
+ else:
1819
+ selected_tasks = tuple(
1820
+ task
1821
+ for task in tasks
1822
+ if task.display_path.replace("\\", "/").casefold()
1823
+ == exact_path_needle
1824
+ )
1825
+ if not selected_tasks:
1826
+ selected_tasks = tuple(
1827
+ task
1828
+ for task in tasks
1829
+ if _source_may_include_selected_path(
1830
+ task.source_path,
1831
+ exact_path_needle,
1832
+ )
1833
+ )
1834
+ tasks = selected_tasks
1835
+ elif exact_needle:
1836
+ tasks = tuple(
1837
+ task
1838
+ for task in tasks
1839
+ if _source_may_contain_exact_name(task.source_path, name_needle)
1840
+ )
1841
+ source_specs = {
1842
+ Path(task.source_path): _SourceSpec(
1843
+ Path(task.source_path),
1844
+ task.display_path,
1845
+ workspace.defines,
1846
+ workspace.include_paths,
1847
+ workspace.project_config,
1848
+ )
1849
+ for task in tasks
1850
+ }
1851
+
1852
+ def consume_result(result: _NavigationResult) -> None:
1853
+ if result.read_error:
1854
+ unit = units[result.ordinal]
1855
+ display_path = unit_display_path(workspace.root, unit)
1856
+ raise AgentProtocolError(
1857
+ "source_unavailable",
1858
+ f"Could not read selected source {display_path}.",
1859
+ )
1860
+ if navigation_store is not None and result.cache_key:
1861
+ try:
1862
+ navigation_store.store(
1863
+ result.cache_key,
1864
+ _navigation_shard_payload(result),
1865
+ )
1866
+ except (OSError, TypeError, ValueError):
1867
+ pass
1868
+ if exact_needle and not any(
1869
+ _normalized(raw.name) == exact_needle
1870
+ or _normalized(raw.qualified_name) == exact_needle
1871
+ for raw in result.raw_symbols
1872
+ ):
1873
+ return
1874
+ raw_symbols.extend(result.raw_symbols)
1875
+
1876
+ disk_hits = 0
1877
+ disk_misses = 0
1878
+ live_navigation_keys: set[str] = set()
1879
+ pending_tasks: list[_NavigationTask] = []
1880
+ if navigation_store is None:
1881
+ pending_tasks.extend(tasks)
1882
+ else:
1883
+ for task in tasks:
1884
+ try:
1885
+ text = read_source_text(Path(task.source_path))
1886
+ cache_key = navigation_cache_key(
1887
+ _navigation_cache_text(
1888
+ text,
1889
+ _thin_unit_include(task, text),
1890
+ ),
1891
+ task.defines,
1892
+ )
1893
+ live_navigation_keys.add(cache_key)
1894
+ payload = navigation_store.load(cache_key)
1895
+ except (OSError, UnicodeError, ValueError):
1896
+ cache_key = ""
1897
+ payload = None
1898
+ cached = (
1899
+ _navigation_result_from_shard(task, payload)
1900
+ if payload is not None
1901
+ else None
1902
+ )
1903
+ if cached is None:
1904
+ disk_misses += 1
1905
+ pending_tasks.append(replace(task, cache_key=cache_key or "miss"))
1906
+ continue
1907
+ disk_hits += 1
1908
+ consume_result(cached)
1909
+
1910
+ outline_batch = run_outline_tasks(
1911
+ pending_tasks,
1912
+ configured_workers=workers,
1913
+ memory_budget_bytes=worker_memory_budget_bytes,
1914
+ on_complete=consume_result,
1915
+ retain_results=False,
1916
+ task_runner=_parse_navigation_task,
1917
+ )
1918
+ if navigation_store is not None:
1919
+ try:
1920
+ navigation_store.prune(
1921
+ live_navigation_keys,
1922
+ navigation_cache_max_bytes,
1923
+ )
1924
+ except (OSError, ValueError):
1925
+ pass
1926
+ parallel_stats = replace(
1927
+ outline_batch.stats,
1928
+ files_completed=len(tasks) + len(selected_include_paths),
1929
+ elapsed_seconds=time.perf_counter() - build_started,
1930
+ )
1931
+
1932
+ entries_tuple = _entries_from_raw_symbols(raw_symbols)
1933
+ for raw in raw_symbols:
1934
+ source_specs.setdefault(
1935
+ raw.source_path,
1936
+ _SourceSpec(
1937
+ raw.source_path,
1938
+ raw.path,
1939
+ workspace.defines,
1940
+ workspace.include_paths,
1941
+ workspace.project_config,
1942
+ ),
1943
+ )
1944
+ source_cache_bytes = _source_cache_budget(worker_memory_budget_bytes)
1945
+ sources = _SourceStore(source_specs, max_loaded_bytes=source_cache_bytes)
1946
+ static_retained_bytes = _estimate_registry_bytes(entries_tuple, sources)
1947
+ registry = _Registry(
1948
+ project_id=project_id,
1949
+ revision=revision,
1950
+ entries=entries_tuple,
1951
+ by_target={entry.target_id: entry for entry in entries_tuple},
1952
+ sources=sources,
1953
+ ranked_queries=OrderedDict(),
1954
+ static_retained_bytes=static_retained_bytes,
1955
+ )
1956
+ if selected_include_paths:
1957
+ _augment_registry_with_include_matches(
1958
+ workspace,
1959
+ registry,
1960
+ exact_query,
1961
+ exact_path=exact_path,
1962
+ )
1963
+ return (
1964
+ registry,
1965
+ parallel_stats,
1966
+ disk_hits,
1967
+ disk_misses,
1968
+ )
1969
+
1970
+
1971
+ def _entries_from_raw_symbols(
1972
+ raw_symbols: Sequence[_RawSymbol],
1973
+ ) -> tuple[_SymbolEntry, ...]:
1974
+ ordered = sorted(raw_symbols, key=_raw_sort_key)
1975
+ overload_groups: dict[tuple[str, str, str, str], list[_RawSymbol]] = {}
1976
+ for raw in raw_symbols:
1977
+ identity = (
1978
+ raw.kind.value.casefold(),
1979
+ raw.path.casefold(),
1980
+ _normalized(raw.qualified_name),
1981
+ _normalized(raw.signature),
1982
+ )
1983
+ overload_groups.setdefault(identity, []).append(raw)
1984
+ ordinals: dict[int, int] = {}
1985
+ for group in overload_groups.values():
1986
+ overload_order = sorted(
1987
+ group,
1988
+ key=lambda raw: (
1989
+ raw.line,
1990
+ raw.column,
1991
+ _raw_sort_key(raw),
1992
+ ),
1993
+ )
1994
+ for ordinal, raw in enumerate(overload_order):
1995
+ ordinals[id(raw)] = ordinal
1996
+
1997
+ shared_strings: dict[str, str] = {}
1998
+
1999
+ def shared(value: str) -> str:
2000
+ return shared_strings.setdefault(value, value)
2001
+
2002
+ target_ids = {
2003
+ id(raw): make_target_id(
2004
+ raw.kind.value,
2005
+ raw.path,
2006
+ _target_identity_name(raw),
2007
+ ordinals[id(raw)],
2008
+ )
2009
+ for raw in raw_symbols
2010
+ }
2011
+ parent_ids: dict[tuple[Path, str], str] = {}
2012
+ for raw in raw_symbols:
2013
+ if raw.kind in {
2014
+ SymbolKind.CLASS,
2015
+ SymbolKind.RECORD,
2016
+ SymbolKind.INTERFACE,
2017
+ SymbolKind.TYPE,
2018
+ }:
2019
+ parent_ids.setdefault(
2020
+ (raw.source_path, _normalized(raw.qualified_name)),
2021
+ target_ids[id(raw)],
2022
+ )
2023
+
2024
+ entries: list[_SymbolEntry] = []
2025
+ for raw in ordered:
2026
+ ordinal = ordinals[id(raw)]
2027
+ (
2028
+ normalized_name,
2029
+ normalized_qualified_name,
2030
+ relative_name_offset,
2031
+ ) = _normalized_search_fields(
2032
+ raw.name,
2033
+ raw.qualified_name,
2034
+ raw.unit_name,
2035
+ )
2036
+ entry = _SymbolEntry(
2037
+ name=shared(raw.name),
2038
+ kind=raw.kind,
2039
+ line=raw.line,
2040
+ column=raw.column,
2041
+ visibility=raw.visibility,
2042
+ type_name=shared(raw.type_name),
2043
+ source_path=raw.source_path,
2044
+ path=shared(raw.path),
2045
+ unit_id=shared(raw.unit_id),
2046
+ unit_name=shared(raw.unit_name),
2047
+ qualified_name=shared(raw.qualified_name),
2048
+ normalized_name=shared(normalized_name),
2049
+ normalized_qualified_name=shared(normalized_qualified_name),
2050
+ relative_name_offset=relative_name_offset,
2051
+ owner=shared(raw.owner),
2052
+ signature=shared(raw.signature),
2053
+ ordinal=ordinal,
2054
+ target_id=target_ids[id(raw)],
2055
+ card_json_chars=0,
2056
+ card_json_upper_bound=0,
2057
+ parent_target_id=parent_ids.get(
2058
+ (
2059
+ raw.source_path,
2060
+ _normalized(raw.parent_qualified_name),
2061
+ ),
2062
+ "",
2063
+ ),
2064
+ context_ambiguous=raw.context_ambiguous,
2065
+ )
2066
+ entries.append(
2067
+ replace(
2068
+ entry,
2069
+ card_json_chars=_symbol_card_json_chars(entry),
2070
+ card_json_upper_bound=_symbol_card_json_upper_bound(entry),
2071
+ )
2072
+ )
2073
+ return tuple(sorted(entries, key=_entry_sort_key))
2074
+
2075
+
2076
+ def _augment_registry_with_include_matches(
2077
+ workspace: AgentWorkspace,
2078
+ registry: _Registry,
2079
+ query: str,
2080
+ *,
2081
+ exact_path: str = "",
2082
+ ) -> bool:
2083
+ normalized_query = unicodedata.normalize('NFC', query.strip()).casefold()
2084
+ if normalized_query in registry.include_queries:
2085
+ return False
2086
+ registry.include_queries.add(normalized_query)
2087
+ query_tail = normalized_query.rsplit('.', 1)[-1]
2088
+ if not query_tail:
2089
+ return False
2090
+ exact_path_needle = exact_path.strip().replace("\\", "/").casefold()
2091
+
2092
+ raw_symbols: list[_RawSymbol] = []
2093
+ for include in workspace.include_files:
2094
+ recorded_path = include['path']
2095
+ display_path = recorded_path
2096
+ if (
2097
+ exact_path_needle
2098
+ and display_path.replace("\\", "/").casefold()
2099
+ != exact_path_needle
2100
+ ):
2101
+ continue
2102
+ source_path = workspace.include_source_path(recorded_path)
2103
+ if source_path in registry.indexed_include_paths:
2104
+ continue
2105
+ try:
2106
+ source = read_source_text(source_path)
2107
+ except (OSError, UnicodeError):
2108
+ continue
2109
+ if query_tail not in source.casefold():
2110
+ continue
2111
+
2112
+ registry.indexed_include_paths.add(source_path)
2113
+ include_name = (
2114
+ query.strip().split('.', 1)[0]
2115
+ if exact_path_needle and '.' in query
2116
+ else Path(include['name']).stem
2117
+ )
2118
+ unit_id = make_target_id('unit', display_path, include_name)
2119
+ try:
2120
+ result = _parse_navigation_task(
2121
+ _NavigationTask(
2122
+ 0,
2123
+ str(source_path),
2124
+ display_path,
2125
+ include_name,
2126
+ display_path,
2127
+ unit_id,
2128
+ False,
2129
+ workspace.defines,
2130
+ workspace.include_paths,
2131
+ workspace.project_config,
2132
+ )
2133
+ )
2134
+ except ParallelOutlineError:
2135
+ continue
2136
+ if result.read_error:
2137
+ continue
2138
+ raw_symbols.extend(
2139
+ _raw_symbol_with_unit_name(raw, include_name)
2140
+ for raw in result.raw_symbols
2141
+ )
2142
+ registry.sources.add_spec(
2143
+ _SourceSpec(
2144
+ source_path,
2145
+ display_path,
2146
+ workspace.defines,
2147
+ workspace.include_paths,
2148
+ workspace.project_config,
2149
+ )
2150
+ )
2151
+
2152
+ if not raw_symbols:
2153
+ return False
2154
+ new_entries = _entries_from_raw_symbols(raw_symbols)
2155
+ registry.entries = tuple(
2156
+ sorted((*registry.entries, *new_entries), key=_entry_sort_key)
2157
+ )
2158
+ registry.by_target.update(
2159
+ (entry.target_id, entry) for entry in new_entries
2160
+ )
2161
+ registry.ranked_queries.clear()
2162
+ registry.static_retained_bytes = _estimate_registry_bytes(
2163
+ registry.entries,
2164
+ registry.sources,
2165
+ )
2166
+ return True
2167
+
2168
+
2169
+ def _raw_symbol_with_unit_name(
2170
+ raw: _RawSymbol,
2171
+ unit_name: str,
2172
+ ) -> _RawSymbol:
2173
+ old_unit_name = raw.unit_name
2174
+
2175
+ def rebased(value: str) -> str:
2176
+ if _normalized(value) == _normalized(old_unit_name):
2177
+ return unit_name
2178
+ prefix = f"{old_unit_name}."
2179
+ if value.casefold().startswith(prefix.casefold()):
2180
+ return f"{unit_name}{value[len(old_unit_name):]}"
2181
+ return value
2182
+
2183
+ return replace(
2184
+ raw,
2185
+ name=(
2186
+ unit_name
2187
+ if raw.kind == SymbolKind.UNIT
2188
+ and _normalized(raw.name) == _normalized(old_unit_name)
2189
+ else raw.name
2190
+ ),
2191
+ unit_name=unit_name,
2192
+ qualified_name=rebased(raw.qualified_name),
2193
+ owner=rebased(raw.owner),
2194
+ parent_qualified_name=rebased(raw.parent_qualified_name),
2195
+ )
2196
+
2197
+
2198
+ def _has_exact_query_match(
2199
+ entries: Sequence[_SymbolEntry],
2200
+ query: str,
2201
+ ) -> bool:
2202
+ needle = unicodedata.normalize('NFC', query.strip()).casefold()
2203
+ return any(
2204
+ entry.normalized_name == needle
2205
+ or entry.normalized_qualified_name == needle
2206
+ for entry in entries
2207
+ )
2208
+
2209
+
2210
+ def _source_may_contain_exact_name(source_path: str, needle: str) -> bool:
2211
+ try:
2212
+ source = read_source_text(Path(source_path))
2213
+ except (OSError, UnicodeError):
2214
+ # Preserve the generic registry's source-unavailable error contract.
2215
+ return True
2216
+ if not needle.isascii():
2217
+ source = unicodedata.normalize('NFC', source)
2218
+ return needle in source.casefold()
2219
+
2220
+
2221
+ def _source_may_include_selected_path(
2222
+ source_path: str,
2223
+ selected_path: str,
2224
+ ) -> bool:
2225
+ include_name = Path(selected_path.replace("\\", "/")).name.casefold()
2226
+ if not include_name:
2227
+ return False
2228
+ try:
2229
+ source = read_source_text(Path(source_path))
2230
+ except (OSError, UnicodeError):
2231
+ return True
2232
+ return include_name in source.casefold()
2233
+
2234
+
2235
+ def _source_cache_budget(total_budget_bytes: int | None) -> int:
2236
+ if total_budget_bytes is None:
2237
+ return 64 * 1024**2
2238
+ return max(0, min(128 * 1024**2, total_budget_bytes // 4))
2239
+
2240
+
2241
+ def _estimate_registry_bytes(
2242
+ entries: tuple[_SymbolEntry, ...],
2243
+ sources: _SourceStore,
2244
+ ) -> int:
2245
+ retained = 4096 + sources.metadata_bytes + sys.getsizeof(entries)
2246
+ seen_values: set[int] = set()
2247
+ for entry in entries:
2248
+ retained += 384 + sys.getsizeof(entry)
2249
+ for value in (
2250
+ entry.source_path,
2251
+ entry.name,
2252
+ entry.type_name,
2253
+ entry.path,
2254
+ entry.unit_id,
2255
+ entry.unit_name,
2256
+ entry.qualified_name,
2257
+ entry.normalized_name,
2258
+ entry.normalized_qualified_name,
2259
+ entry.owner,
2260
+ entry.signature,
2261
+ entry.target_id,
2262
+ entry.parent_target_id,
2263
+ ):
2264
+ identifier = id(value)
2265
+ if identifier in seen_values:
2266
+ continue
2267
+ seen_values.add(identifier)
2268
+ retained += sys.getsizeof(value)
2269
+ retained += len(entries) * 96
2270
+ return retained
2271
+
2272
+
2273
+ def _symbol_card_json_upper_bound(entry: _SymbolEntry) -> int:
2274
+ strings = (
2275
+ entry.target_id,
2276
+ entry.unit_id,
2277
+ entry.name,
2278
+ entry.qualified_name,
2279
+ entry.kind.value,
2280
+ entry.path,
2281
+ entry.visibility.value,
2282
+ entry.owner,
2283
+ entry.type_name,
2284
+ )
2285
+ numeric_chars = len(str(entry.line)) + len(str(entry.column))
2286
+ # JSON string escaping expands one input character to at most six characters.
2287
+ # The fixed allowance covers keys, quotes, separators, brackets, and numbers.
2288
+ ambiguity_chars = (
2289
+ _CONTEXT_AMBIGUOUS_JSON_CHARS if entry.context_ambiguous else 0
2290
+ )
2291
+ return (
2292
+ 512
2293
+ + 6 * sum(len(value) for value in strings)
2294
+ + numeric_chars
2295
+ + ambiguity_chars
2296
+ )
2297
+
2298
+
2299
+ _SYMBOL_CARD_JSON_KEYS = (
2300
+ "target_id",
2301
+ "unit_id",
2302
+ "name",
2303
+ "qualified_name",
2304
+ "kind",
2305
+ "path",
2306
+ "line",
2307
+ "column",
2308
+ "visibility",
2309
+ "owner",
2310
+ "type",
2311
+ )
2312
+ _SYMBOL_CARD_JSON_FIXED_CHARS = (
2313
+ 2
2314
+ + len(_SYMBOL_CARD_JSON_KEYS) - 1
2315
+ + sum(len(key) + 3 for key in _SYMBOL_CARD_JSON_KEYS)
2316
+ )
2317
+ _CONTEXT_AMBIGUOUS_JSON_CHARS = len(',"context_ambiguous":true')
2318
+
2319
+
2320
+ def _symbol_card_json_chars(entry: _SymbolEntry) -> int:
2321
+ strings = (
2322
+ entry.target_id,
2323
+ entry.unit_id,
2324
+ entry.name,
2325
+ entry.qualified_name,
2326
+ entry.kind.value,
2327
+ entry.path,
2328
+ entry.visibility.value,
2329
+ entry.owner,
2330
+ entry.type_name,
2331
+ )
2332
+ return (
2333
+ _SYMBOL_CARD_JSON_FIXED_CHARS
2334
+ + sum(_json_string_chars(value) for value in strings)
2335
+ + len(str(entry.line))
2336
+ + len(str(entry.column))
2337
+ + (
2338
+ _CONTEXT_AMBIGUOUS_JSON_CHARS
2339
+ if entry.context_ambiguous
2340
+ else 0
2341
+ )
2342
+ )
2343
+
2344
+
2345
+ @lru_cache(maxsize=16_384)
2346
+ def _json_string_chars(value: str) -> int:
2347
+ if (
2348
+ value.isascii()
2349
+ and value.isprintable()
2350
+ and '"' not in value
2351
+ and "\\" not in value
2352
+ ):
2353
+ return len(value) + 2
2354
+ return len(json.dumps(value, ensure_ascii=False))
2355
+
2356
+
2357
+ def _stable_path_component(value: str) -> str:
2358
+ normalized = unicodedata.normalize("NFC", value).replace("\\", "_").replace("/", "_")
2359
+ return normalized or "unknown"
2360
+
2361
+
2362
+ def _sanitize_workspace_mapping(
2363
+ value: Mapping[str, object],
2364
+ root: Path,
2365
+ *,
2366
+ path_namespace: str,
2367
+ ) -> dict[str, object]:
2368
+ sanitized = dict(value)
2369
+ message = sanitized.get("message")
2370
+ path = sanitized.get("path")
2371
+ if isinstance(path, str):
2372
+ safe_path = _sanitize_workspace_path(path, root, path_namespace)
2373
+ sanitized["path"] = safe_path
2374
+ if isinstance(message, str):
2375
+ message = _replace_path_in_message(message, path, safe_path)
2376
+ origin = sanitized.get("origin")
2377
+ if isinstance(origin, str):
2378
+ safe_origin = _sanitize_workspace_path(origin, root, "origin")
2379
+ sanitized["origin"] = safe_origin
2380
+ if isinstance(message, str):
2381
+ message = _replace_path_in_message(message, origin, safe_origin)
2382
+ origins = sanitized.get("origins")
2383
+ if isinstance(origins, (list, tuple)):
2384
+ sanitized["origins"] = [
2385
+ _sanitize_workspace_path(item, root, "origin") if isinstance(item, str) else item
2386
+ for item in origins
2387
+ ]
2388
+ if isinstance(message, str):
2389
+ sanitized["message"] = message
2390
+ return sanitized
2391
+
2392
+
2393
+ def _replace_path_in_message(message: str, path: str, replacement: str) -> str:
2394
+ variants = {
2395
+ path,
2396
+ path.replace("\\", "/"),
2397
+ path.replace("/", "\\"),
2398
+ }
2399
+ for variant in sorted(variants, key=len, reverse=True):
2400
+ message = message.replace(variant, replacement)
2401
+ return message
2402
+
2403
+
2404
+ def _sanitize_workspace_path(value: str, root: Path, namespace: str) -> str:
2405
+ normalized = unicodedata.normalize("NFC", value).replace("\\", "/")
2406
+ native_path = Path(value).expanduser()
2407
+ if native_path.is_absolute():
2408
+ resolved = native_path.resolve()
2409
+ try:
2410
+ return resolved.relative_to(root).as_posix()
2411
+ except ValueError:
2412
+ component = resolved.name
2413
+ else:
2414
+ windows_path = PureWindowsPath(value)
2415
+ if not windows_path.is_absolute():
2416
+ return normalized
2417
+ component = windows_path.name or windows_path.drive.rstrip(":\\/")
2418
+ return f"@external/{namespace}/{_stable_path_component(component)}"
2419
+
2420
+
2421
+ def _target_identity_name(raw: _RawSymbol) -> str:
2422
+ if raw.kind in _ROUTINE_KINDS:
2423
+ return f"{raw.qualified_name}\x1f{_normalized(raw.signature)}"
2424
+ return raw.qualified_name
2425
+
2426
+
2427
+ def _collect_raw_symbols(
2428
+ scope: Scope,
2429
+ unit: AgentUnit,
2430
+ source_path: Path,
2431
+ document: _SourceDocument,
2432
+ ) -> list[_RawSymbol]:
2433
+ unit_name = scope.name or unit.name
2434
+ unit_id = make_target_id("unit", document.display_path, unit.name)
2435
+ collected: list[_RawSymbol] = []
2436
+ symbols = [symbol for group in scope.symbols.values() for symbol in group]
2437
+ symbols.sort(key=_symbol_sort_key)
2438
+ for symbol in symbols:
2439
+ if symbol.scope.kind != ScopeKind.UNIT:
2440
+ continue
2441
+ _correct_outline_symbol_kind(document, symbol)
2442
+ if symbol.kind == SymbolKind.UNIT:
2443
+ qualified_name = unit_name
2444
+ owner = ""
2445
+ else:
2446
+ declared_name = _declared_symbol_name(document, symbol).strip(".")
2447
+ qualified_name = f"{unit_name}.{declared_name}"
2448
+ owner = qualified_name.rsplit(".", 1)[0]
2449
+ collected.append(
2450
+ _RawSymbol(
2451
+ name=symbol.name,
2452
+ kind=symbol.kind,
2453
+ line=symbol.decl_range.start_line,
2454
+ column=symbol.decl_range.start_col,
2455
+ visibility=symbol.visibility,
2456
+ type_name=symbol.type_ref.display_name(),
2457
+ source_path=source_path,
2458
+ path=document.display_path,
2459
+ unit_id=unit_id,
2460
+ unit_name=unit_name,
2461
+ qualified_name=qualified_name,
2462
+ owner=owner,
2463
+ parent_qualified_name="",
2464
+ signature=_symbol_signature(document, symbol),
2465
+ context_ambiguous=(
2466
+ symbol.attributes.get("context_ambiguous") == "true"
2467
+ ),
2468
+ )
2469
+ )
2470
+ member_scope = symbol.member_scope
2471
+ if member_scope is None or member_scope.kind != ScopeKind.TYPE:
2472
+ continue
2473
+ members = [member for group in member_scope.symbols.values() for member in group]
2474
+ members.sort(key=_symbol_sort_key)
2475
+ for member in members:
2476
+ member_name = _declared_symbol_name(document, member).strip(".")
2477
+ member_qualified_name = f"{qualified_name}.{member_name}"
2478
+ collected.append(
2479
+ _RawSymbol(
2480
+ name=member.name,
2481
+ kind=member.kind,
2482
+ line=member.decl_range.start_line,
2483
+ column=member.decl_range.start_col,
2484
+ visibility=member.visibility,
2485
+ type_name=member.type_ref.display_name(),
2486
+ source_path=source_path,
2487
+ path=document.display_path,
2488
+ unit_id=unit_id,
2489
+ unit_name=unit_name,
2490
+ qualified_name=member_qualified_name,
2491
+ owner=qualified_name,
2492
+ parent_qualified_name=qualified_name,
2493
+ signature=_symbol_signature(document, member),
2494
+ context_ambiguous=(
2495
+ member.attributes.get("context_ambiguous") == "true"
2496
+ ),
2497
+ )
2498
+ )
2499
+ return collected
2500
+
2501
+
2502
+ def _correct_outline_symbol_kind(document: _SourceDocument, symbol: Symbol) -> None:
2503
+ if symbol.kind != SymbolKind.CONSTANT:
2504
+ return
2505
+ start = _declaration_start(document, symbol.decl_range.start_line)
2506
+ if _declaration_section(document, start) != "type":
2507
+ return
2508
+ token_index = document.first_token_index(start)
2509
+ equals_index = _next_token_value(document.tokens, token_index, "=")
2510
+ if equals_index is None or equals_index + 1 >= len(document.tokens):
2511
+ return
2512
+ rhs = document.tokens[equals_index + 1:]
2513
+ index = 0
2514
+ if rhs and rhs[0].value == "packed":
2515
+ index = 1
2516
+ if index >= len(rhs):
2517
+ return
2518
+ value = rhs[index].value
2519
+ if value == "class" and not (
2520
+ index + 1 < len(rhs) and rhs[index + 1].value == "of"
2521
+ ):
2522
+ symbol.kind = SymbolKind.CLASS
2523
+ elif value == "record":
2524
+ symbol.kind = SymbolKind.RECORD
2525
+ elif value == "interface":
2526
+ symbol.kind = SymbolKind.INTERFACE
2527
+ elif value == "(":
2528
+ symbol.kind = SymbolKind.ENUM
2529
+ else:
2530
+ symbol.kind = SymbolKind.TYPE
2531
+
2532
+
2533
+ def _advance_declaration_section(
2534
+ state: tuple[str, int, int, int],
2535
+ token: _Token,
2536
+ ) -> tuple[str, int, int, int]:
2537
+ section, parentheses, brackets, angles = state
2538
+ if token.directive:
2539
+ return state
2540
+ if token.value == "(":
2541
+ parentheses += 1
2542
+ elif token.value == ")":
2543
+ parentheses = max(0, parentheses - 1)
2544
+ elif token.value == "[":
2545
+ brackets += 1
2546
+ elif token.value == "]":
2547
+ brackets = max(0, brackets - 1)
2548
+ elif token.value == "<":
2549
+ angles += 1
2550
+ elif token.value == ">":
2551
+ angles = max(0, angles - 1)
2552
+ elif not parentheses and not brackets and not angles and token.word:
2553
+ if token.value in {"const", "resourcestring", "threadvar", "type", "var"}:
2554
+ section = token.value
2555
+ elif token.value in {"implementation", "initialization", "finalization"}:
2556
+ section = ""
2557
+ return section, parentheses, brackets, angles
2558
+
2559
+
2560
+ def _declaration_section(document: _SourceDocument, offset: int) -> str:
2561
+ target_index = document.first_token_index(offset)
2562
+ cached = document.declaration_section_checkpoints.get(target_index)
2563
+ if cached is not None:
2564
+ return cached[0]
2565
+ checkpoint_position = bisect_right(document.declaration_section_indexes, target_index) - 1
2566
+ checkpoint_index = document.declaration_section_indexes[checkpoint_position]
2567
+ state = document.declaration_section_checkpoints[checkpoint_index]
2568
+ for token in document.tokens[checkpoint_index:target_index]:
2569
+ state = _advance_declaration_section(state, token)
2570
+ insert_at = bisect_left(document.declaration_section_indexes, target_index)
2571
+ document.declaration_section_indexes.insert(insert_at, target_index)
2572
+ document.declaration_section_checkpoints[target_index] = state
2573
+ return state[0]
2574
+
2575
+
2576
+ def _declared_symbol_name(document: _SourceDocument, symbol: Symbol) -> str:
2577
+ if symbol.kind in _ROUTINE_KINDS:
2578
+ return _routine_declared_name(document, symbol) or symbol.name
2579
+ if symbol.kind in _TYPE_KINDS:
2580
+ return _type_declared_name(document, symbol) or symbol.name
2581
+ return symbol.name
2582
+
2583
+
2584
+ def _routine_declared_name(document: _SourceDocument, symbol: Symbol) -> str:
2585
+ start = _declaration_start(document, symbol.decl_range.start_line)
2586
+ token_index = document.first_token_index(start)
2587
+ routine_index = _routine_keyword_index(document.tokens, token_index)
2588
+ if routine_index is None:
2589
+ return ""
2590
+ heading_end = _heading_semicolon_index(document.tokens, routine_index)
2591
+ if heading_end is None:
2592
+ return ""
2593
+ name_span = _routine_name_token_span(document.tokens, routine_index, heading_end)
2594
+ if name_span is None:
2595
+ return ""
2596
+ return _join_source_tokens(document, document.tokens[name_span[0]:name_span[1]])
2597
+
2598
+
2599
+ def _type_declared_name(document: _SourceDocument, symbol: Symbol) -> str:
2600
+ start = _declaration_start(document, symbol.decl_range.start_line)
2601
+ token_index = document.first_token_index(start)
2602
+ equals_index = next(
2603
+ (
2604
+ index
2605
+ for index in range(token_index, len(document.tokens))
2606
+ if document.tokens[index].value in {"=", ";"}
2607
+ ),
2608
+ None,
2609
+ )
2610
+ if equals_index is None or document.tokens[equals_index].value != "=":
2611
+ return ""
2612
+ name_index = next(
2613
+ (
2614
+ index
2615
+ for index in range(token_index, equals_index)
2616
+ if document.tokens[index].word
2617
+ and _normalized(document.tokens[index].value) == _normalized(symbol.name)
2618
+ ),
2619
+ token_index,
2620
+ )
2621
+ return _join_source_tokens(document, document.tokens[name_index:equals_index])
2622
+
2623
+
2624
+ def _join_source_tokens(document: _SourceDocument, tokens: tuple[_Token, ...]) -> str:
2625
+ return unicodedata.normalize(
2626
+ "NFC",
2627
+ "".join(document.text[token.start:token.end] for token in tokens),
2628
+ )
2629
+
2630
+
2631
+ def _symbol_signature(document: _SourceDocument, symbol: Symbol) -> str:
2632
+ if symbol.kind not in _ROUTINE_KINDS:
2633
+ return ""
2634
+ start = _declaration_start(document, symbol.decl_range.start_line)
2635
+ start_index = document.first_token_index(start)
2636
+ routine_index = _routine_keyword_index(document.tokens, start_index)
2637
+ if routine_index is None:
2638
+ return ""
2639
+ heading_end = _heading_semicolon_index(document.tokens, routine_index)
2640
+ if heading_end is None:
2641
+ return ""
2642
+ name_span = _routine_name_token_span(document.tokens, routine_index, heading_end)
2643
+ if name_span is None:
2644
+ return ""
2645
+ signature_tokens = document.tokens[name_span[1]:heading_end]
2646
+ declaration_end = _routine_declaration_end_index(document.tokens, routine_index)
2647
+ calling_conventions = ""
2648
+ if declaration_end is not None:
2649
+ conventions = sorted(
2650
+ {
2651
+ token.value
2652
+ for token in document.tokens[heading_end + 1:declaration_end]
2653
+ if token.word and not token.escaped and token.value in _CALLING_CONVENTIONS
2654
+ }
2655
+ )
2656
+ if conventions:
2657
+ calling_conventions = f"|cc:{','.join(conventions)}"
2658
+ return _normalized(f"{_normalize_routine_signature(signature_tokens)}{calling_conventions}")
2659
+
2660
+
2661
+ def _routine_name_token_span(
2662
+ tokens: tuple[_Token, ...],
2663
+ routine_index: int,
2664
+ heading_end: int,
2665
+ ) -> tuple[int, int] | None:
2666
+ start = routine_index + 1
2667
+ if start >= heading_end:
2668
+ return None
2669
+ end = start + 1
2670
+ while end < heading_end:
2671
+ if (
2672
+ tokens[end].value == "."
2673
+ and end + 1 < heading_end
2674
+ and (tokens[end + 1].word or tokens[routine_index].value == "operator")
2675
+ ):
2676
+ end += 2
2677
+ continue
2678
+ if tokens[end].value == "<":
2679
+ generic_end = _matching_token_index(tokens, end, "<", ">")
2680
+ if (
2681
+ generic_end is not None
2682
+ and generic_end + 1 < heading_end
2683
+ and tokens[generic_end + 1].value == "."
2684
+ ):
2685
+ end = generic_end + 1
2686
+ continue
2687
+ break
2688
+ return start, end
2689
+
2690
+
2691
+ def _normalize_routine_signature(tokens: tuple[_Token, ...]) -> str:
2692
+ open_index = next((index for index, token in enumerate(tokens) if token.value == "("), None)
2693
+ if open_index is None:
2694
+ colon = _top_level_token_index(tokens, ":")
2695
+ if colon is None:
2696
+ return f"{_join_token_values(tokens)}()"
2697
+ generic = _join_token_values(tokens[:colon])
2698
+ return f"{generic}():{_normalize_type_tokens(tokens[colon + 1:])}"
2699
+
2700
+ close_index = _matching_token_index(tokens, open_index, "(", ")")
2701
+ if close_index is None:
2702
+ return _join_token_values(tokens)
2703
+ generic = _join_token_values(tokens[:open_index])
2704
+ parameters = _normalize_parameters(tokens[open_index + 1:close_index])
2705
+ result_type = _normalize_result_type(tokens[close_index + 1:])
2706
+ return f"{generic}({parameters}){result_type}"
2707
+
2708
+
2709
+ def _normalize_parameters(tokens: tuple[_Token, ...]) -> str:
2710
+ groups = _split_tokens_at_top_level(tokens, ";")
2711
+ normalized: list[str] = []
2712
+ modes = {"const", "constref", "out", "var"}
2713
+ for group in groups:
2714
+ if not group:
2715
+ continue
2716
+ colon = _top_level_token_index(group, ":")
2717
+ if colon is None:
2718
+ normalized.append(_join_token_values(group))
2719
+ continue
2720
+ names = list(group[:colon])
2721
+ mode = ""
2722
+ if names and names[0].word and names[0].value in modes:
2723
+ mode = names.pop(0).value
2724
+ parameter_count = 1 + sum(token.value == "," for token in names)
2725
+ type_tokens = group[colon + 1:]
2726
+ default = _top_level_token_index(type_tokens, "=")
2727
+ if default is not None:
2728
+ type_tokens = type_tokens[:default]
2729
+ normalized.append(f"{mode}#{parameter_count}:{_normalize_type_tokens(type_tokens)}")
2730
+ return ";".join(normalized)
2731
+
2732
+
2733
+ def _normalize_result_type(tokens: tuple[_Token, ...]) -> str:
2734
+ if not tokens or tokens[0].value != ":":
2735
+ return ""
2736
+ return f":{_normalize_type_tokens(tokens[1:])}"
2737
+
2738
+
2739
+ def _normalize_type_tokens(tokens: tuple[_Token, ...]) -> str:
2740
+ parts: list[str] = []
2741
+ index = 0
2742
+ while index < len(tokens):
2743
+ token = tokens[index]
2744
+ parts.append(_normalized(token.value))
2745
+ if (
2746
+ token.word
2747
+ and token.value in {"procedure", "function"}
2748
+ and index + 1 < len(tokens)
2749
+ and tokens[index + 1].value == "("
2750
+ ):
2751
+ close = _matching_token_index(tokens, index + 1, "(", ")")
2752
+ if close is None:
2753
+ index += 1
2754
+ continue
2755
+ parts.append("(")
2756
+ parts.append(_normalize_parameters(tokens[index + 2:close]))
2757
+ parts.append(")")
2758
+ index = close + 1
2759
+ continue
2760
+ index += 1
2761
+ return unicodedata.normalize("NFC", "".join(parts)).casefold()
2762
+
2763
+
2764
+ def _split_tokens_at_top_level(
2765
+ tokens: tuple[_Token, ...],
2766
+ separator: str,
2767
+ ) -> list[tuple[_Token, ...]]:
2768
+ groups: list[tuple[_Token, ...]] = []
2769
+ start = 0
2770
+ depths = {"(": 0, "[": 0, "<": 0}
2771
+ closing = {")": "(", "]": "[", ">": "<"}
2772
+ for index, token in enumerate(tokens):
2773
+ if token.value in depths:
2774
+ depths[token.value] += 1
2775
+ elif token.value in closing:
2776
+ opener = closing[token.value]
2777
+ depths[opener] = max(0, depths[opener] - 1)
2778
+ elif token.value == separator and not any(depths.values()):
2779
+ groups.append(tokens[start:index])
2780
+ start = index + 1
2781
+ groups.append(tokens[start:])
2782
+ return groups
2783
+
2784
+
2785
+ def _top_level_token_index(tokens: tuple[_Token, ...], value: str) -> int | None:
2786
+ depths = {"(": 0, "[": 0, "<": 0}
2787
+ closing = {")": "(", "]": "[", ">": "<"}
2788
+ for index, token in enumerate(tokens):
2789
+ if token.value in depths:
2790
+ depths[token.value] += 1
2791
+ elif token.value in closing:
2792
+ opener = closing[token.value]
2793
+ depths[opener] = max(0, depths[opener] - 1)
2794
+ elif token.value == value and not any(depths.values()):
2795
+ return index
2796
+ return None
2797
+
2798
+
2799
+ def _matching_token_index(
2800
+ tokens: tuple[_Token, ...],
2801
+ start: int,
2802
+ opener: str,
2803
+ closer: str,
2804
+ ) -> int | None:
2805
+ depth = 0
2806
+ for index in range(start, len(tokens)):
2807
+ if tokens[index].value == opener:
2808
+ depth += 1
2809
+ elif tokens[index].value == closer:
2810
+ depth -= 1
2811
+ if depth == 0:
2812
+ return index
2813
+ return None
2814
+
2815
+
2816
+ def _join_token_values(tokens: tuple[_Token, ...]) -> str:
2817
+ return unicodedata.normalize("NFC", "".join(token.value for token in tokens)).casefold()
2818
+
2819
+
2820
+ def _exclude_routine_locals(
2821
+ symbols: list[_RawSymbol],
2822
+ document: _SourceDocument,
2823
+ ) -> list[_RawSymbol]:
2824
+ containers: list[tuple[int, int, _RawSymbol]] = []
2825
+ for raw in symbols:
2826
+ if raw.parent_qualified_name or raw.kind not in _ROUTINE_KINDS:
2827
+ continue
2828
+ span = _raw_routine_span(raw, document)
2829
+ if span is not None:
2830
+ containers.append((span[0], span[1], raw))
2831
+ if not containers:
2832
+ return symbols
2833
+
2834
+ containers.sort(key=lambda item: (item[0], item[1]))
2835
+ positioned = sorted(
2836
+ (
2837
+ document.offset(
2838
+ raw.line,
2839
+ raw.column,
2840
+ ),
2841
+ order,
2842
+ raw,
2843
+ )
2844
+ for order, raw in enumerate(symbols)
2845
+ )
2846
+ active_ends: list[tuple[int, int, int]] = []
2847
+ active_ids: set[int] = set()
2848
+ excluded_orders: set[int] = set()
2849
+ container_index = 0
2850
+ for offset, order, raw in positioned:
2851
+ while (
2852
+ container_index < len(containers)
2853
+ and containers[container_index][0] < offset
2854
+ ):
2855
+ _, end, container = containers[container_index]
2856
+ container_id = id(container)
2857
+ heappush(active_ends, (end, container_index, container_id))
2858
+ active_ids.add(container_id)
2859
+ container_index += 1
2860
+ while active_ends and active_ends[0][0] <= offset:
2861
+ _, _, container_id = heappop(active_ends)
2862
+ active_ids.discard(container_id)
2863
+ raw_id = id(raw)
2864
+ if active_ids and (raw_id not in active_ids or len(active_ids) > 1):
2865
+ excluded_orders.add(order)
2866
+
2867
+ return [
2868
+ raw
2869
+ for order, raw in enumerate(symbols)
2870
+ if order not in excluded_orders
2871
+ ]
2872
+
2873
+
2874
+ def _raw_routine_span(
2875
+ raw: _RawSymbol,
2876
+ document: _SourceDocument,
2877
+ ) -> tuple[int, int] | None:
2878
+ if raw.kind not in _ROUTINE_KINDS or raw.parent_qualified_name:
2879
+ return None
2880
+ line = raw.line
2881
+ if document.unit_kind == "unit" and (
2882
+ not document.implementation_line or line < document.implementation_line
2883
+ ):
2884
+ return None
2885
+ start = _declaration_start(document, line)
2886
+ return _routine_span(document, start)
2887
+
2888
+
2889
+ def _body_entry_and_span(
2890
+ registry: _Registry,
2891
+ entry: _SymbolEntry,
2892
+ ) -> tuple[_SymbolEntry, tuple[int, int] | None]:
2893
+ candidates = [entry]
2894
+ if entry.kind in _ROUTINE_KINDS:
2895
+ candidates.extend(_matching_counterparts(registry, entry))
2896
+ for candidate in candidates:
2897
+ document = registry.sources[candidate.source_path]
2898
+ span = _entry_body_span(candidate, document)
2899
+ if span is not None:
2900
+ return candidate, span
2901
+ return entry, None
2902
+
2903
+
2904
+ def _matching_counterparts(
2905
+ registry: _Registry,
2906
+ entry: _SymbolEntry,
2907
+ ) -> list[_SymbolEntry]:
2908
+ return [
2909
+ candidate
2910
+ for candidate in registry.entries
2911
+ if candidate.target_id != entry.target_id
2912
+ and candidate.kind == entry.kind
2913
+ and _normalized(candidate.qualified_name) == _normalized(entry.qualified_name)
2914
+ and candidate.signature == entry.signature
2915
+ ]
2916
+
2917
+
2918
+ def _entry_body_span(
2919
+ entry: _SymbolEntry,
2920
+ document: _SourceDocument,
2921
+ ) -> tuple[int, int] | None:
2922
+ if entry.kind in _TYPE_KINDS:
2923
+ if _is_forward_type(document, entry):
2924
+ return None
2925
+ span = _type_span(document, entry)
2926
+ if span is not None and not document.contains_directive(*span):
2927
+ return span
2928
+ return _full_parser_span(document, entry)
2929
+ if entry.kind not in _ROUTINE_KINDS or entry.parent_target_id:
2930
+ return None
2931
+ line = entry.line
2932
+ if document.unit_kind == "unit" and (
2933
+ not document.implementation_line or line < document.implementation_line
2934
+ ):
2935
+ return None
2936
+ start = _declaration_start(document, line)
2937
+ span = _routine_span(document, start)
2938
+ if span is not None and not document.contains_directive(*span):
2939
+ return span
2940
+ return _full_parser_span(document, entry)
2941
+
2942
+
2943
+ def _full_parser_span(
2944
+ document: _SourceDocument,
2945
+ entry: _SymbolEntry,
2946
+ ) -> tuple[int, int] | None:
2947
+ if entry.target_id in document.parser_spans:
2948
+ return document.parser_spans[entry.target_id]
2949
+ result = document.full_parse()
2950
+ if result is None:
2951
+ document.parser_spans[entry.target_id] = None
2952
+ return None
2953
+
2954
+ expected_type = (
2955
+ SyntaxNodeType.ntMethod
2956
+ if entry.kind in _ROUTINE_KINDS
2957
+ else SyntaxNodeType.ntTypeDecl
2958
+ )
2959
+ expected_line = entry.line
2960
+ candidates: list[tuple[int, int]] = []
2961
+ for node in _walk_syntax_nodes(result.root):
2962
+ if node.typ != expected_type or not isinstance(node, CompoundSyntaxNode):
2963
+ continue
2964
+ mapped = _mapped_tree_span(document, result, node)
2965
+ if mapped is None or document.line_col(mapped[0])[0] != expected_line:
2966
+ continue
2967
+ if expected_type == SyntaxNodeType.ntMethod and not _syntax_method_has_body(node):
2968
+ continue
2969
+ candidates.append(mapped)
2970
+
2971
+ span = candidates[0] if len(candidates) == 1 else None
2972
+ document.parser_spans[entry.target_id] = span
2973
+ return span
2974
+
2975
+
2976
+ def _walk_syntax_nodes(root: SyntaxNode):
2977
+ stack = [root]
2978
+ while stack:
2979
+ node = stack.pop()
2980
+ yield node
2981
+ stack.extend(reversed(node.child_nodes))
2982
+
2983
+
2984
+ def _syntax_method_has_body(node: SyntaxNode) -> bool:
2985
+ if any(
2986
+ node.has_attribute(attribute)
2987
+ for attribute in (
2988
+ AttributeName.anAbstract,
2989
+ AttributeName.anExternal,
2990
+ AttributeName.anForwarded,
2991
+ )
2992
+ ):
2993
+ return False
2994
+ return any(
2995
+ candidate.typ == SyntaxNodeType.ntStatements
2996
+ for candidate in _walk_syntax_nodes(node)
2997
+ if candidate is not node
2998
+ )
2999
+
3000
+
3001
+ def _mapped_tree_span(document: _SourceDocument, result, node: SyntaxNode) -> tuple[int, int] | None:
3002
+ end_line, end_col = _syntax_tree_end(node)
3003
+ start_file, start_line, _ = result.preprocessed.map_position(node.line, node.col)
3004
+ end_file, mapped_end_line, mapped_end_col = result.preprocessed.map_position(end_line, end_col)
3005
+ if not start_file or not end_file:
3006
+ return None
3007
+ try:
3008
+ start_path = Path(start_file).expanduser().resolve()
3009
+ end_path = Path(end_file).expanduser().resolve()
3010
+ except OSError:
3011
+ return None
3012
+ if start_path != document.source_path or end_path != document.source_path:
3013
+ return None
3014
+
3015
+ previous_line = start_line - 1
3016
+ for preprocessed_line in range(node.line, end_line + 1):
3017
+ mapped_file, mapped_line, _ = result.preprocessed.map_position(preprocessed_line, 1)
3018
+ if not mapped_file or Path(mapped_file).expanduser().resolve() != document.source_path:
3019
+ return None
3020
+ if mapped_line != previous_line + 1:
3021
+ return None
3022
+ previous_line = mapped_line
3023
+
3024
+ start = _declaration_start(document, start_line)
3025
+ end = document.offset(mapped_end_line, mapped_end_col)
3026
+ if end < len(document.text) and not document.text[end].isspace():
3027
+ end += 1
3028
+ if end <= start:
3029
+ return None
3030
+ return start, end
3031
+
3032
+
3033
+ def _syntax_tree_end(node: SyntaxNode) -> tuple[int, int]:
3034
+ end = (
3035
+ (node.end_line, node.end_col)
3036
+ if isinstance(node, CompoundSyntaxNode)
3037
+ else (node.line, node.col)
3038
+ )
3039
+ for child in node.child_nodes:
3040
+ end = max(end, _syntax_tree_end(child))
3041
+ return end
3042
+
3043
+
3044
+ def _declaration_span(
3045
+ document: _SourceDocument,
3046
+ entry: _SymbolEntry,
3047
+ ) -> tuple[int, int]:
3048
+ line = entry.line
3049
+ start = _declaration_start(document, line)
3050
+ if entry.kind in _TYPE_KINDS:
3051
+ full_span, direct_structured = _type_declaration_layout(document, entry)
3052
+ if direct_structured:
3053
+ return start, document.line_end(line)
3054
+ if full_span is not None:
3055
+ return full_span
3056
+ return start, document.line_end(line)
3057
+
3058
+ token_index = document.first_token_index(start)
3059
+ if entry.kind in _ROUTINE_KINDS:
3060
+ routine_index = _routine_keyword_index(document.tokens, token_index)
3061
+ if routine_index is not None:
3062
+ declaration_end = _routine_declaration_end_index(document.tokens, routine_index)
3063
+ if declaration_end is not None:
3064
+ adjacent_routine = _top_level_routine_boundary_index(
3065
+ document.tokens,
3066
+ routine_index + 1,
3067
+ declaration_end,
3068
+ )
3069
+ if adjacent_routine is not None:
3070
+ boundary_line = document.line_col(
3071
+ document.tokens[adjacent_routine].start
3072
+ )[0]
3073
+ return start, document.line_end(max(line, boundary_line - 1))
3074
+ return start, document.tokens[declaration_end].end
3075
+ for token in document.tokens[token_index:]:
3076
+ if token.value == ";":
3077
+ return start, token.end
3078
+ return start, document.line_end(line)
3079
+
3080
+
3081
+ def _declaration_start(document: _SourceDocument, line: int) -> int:
3082
+ start = document.line_start(line)
3083
+ end = document.line_end(line)
3084
+ while start < end and document.text[start] in {" ", "\t"}:
3085
+ start += 1
3086
+ return start
3087
+
3088
+
3089
+ def _type_span(
3090
+ document: _SourceDocument,
3091
+ entry: _SymbolEntry,
3092
+ ) -> tuple[int, int] | None:
3093
+ declaration, _ = _type_declaration_layout(document, entry)
3094
+ return declaration
3095
+
3096
+
3097
+ def _is_forward_type(document: _SourceDocument, entry: _SymbolEntry) -> bool:
3098
+ start = _declaration_start(document, entry.line)
3099
+ token_index = document.first_token_index(start)
3100
+ equals_index = _next_token_value(document.tokens, token_index, "=")
3101
+ if equals_index is None:
3102
+ return False
3103
+ structure_index = _structured_type_index(document.tokens, equals_index + 1)
3104
+ return (
3105
+ structure_index is not None
3106
+ and structure_index + 1 < len(document.tokens)
3107
+ and document.tokens[structure_index + 1].value == ";"
3108
+ )
3109
+
3110
+
3111
+ def _type_declaration_layout(
3112
+ document: _SourceDocument,
3113
+ entry: _SymbolEntry,
3114
+ ) -> tuple[tuple[int, int] | None, bool]:
3115
+ start = _declaration_start(document, entry.line)
3116
+ token_index = document.first_token_index(start)
3117
+ equals_index = _next_token_value(document.tokens, token_index, "=")
3118
+ if equals_index is None:
3119
+ return None, False
3120
+ structure_index = _structured_type_index(document.tokens, equals_index + 1)
3121
+ if structure_index is not None:
3122
+ direct_structured = all(
3123
+ token.value in {"packed"}
3124
+ for token in document.tokens[equals_index + 1:structure_index]
3125
+ )
3126
+ if (
3127
+ structure_index + 1 < len(document.tokens)
3128
+ and document.tokens[structure_index + 1].value == ";"
3129
+ ):
3130
+ return (start, document.tokens[structure_index + 1].end), True
3131
+ end = _match_end_terminated_block(document.tokens, structure_index)
3132
+ return ((start, end) if end is not None else None), direct_structured
3133
+
3134
+ declaration_end = _top_level_semicolon_index(document.tokens, equals_index + 1)
3135
+ if declaration_end is None:
3136
+ return None, False
3137
+ return (start, document.tokens[declaration_end].end), False
3138
+
3139
+
3140
+ def _next_token_value(
3141
+ tokens: tuple[_Token, ...],
3142
+ start_index: int,
3143
+ value: str,
3144
+ ) -> int | None:
3145
+ for index in range(start_index, len(tokens)):
3146
+ if tokens[index].value == value:
3147
+ return index
3148
+ if tokens[index].value == ";":
3149
+ return None
3150
+ return None
3151
+
3152
+
3153
+ def _structured_type_index(
3154
+ tokens: tuple[_Token, ...],
3155
+ start_index: int,
3156
+ ) -> int | None:
3157
+ for index in range(start_index, len(tokens)):
3158
+ token = tokens[index]
3159
+ if token.value == ";":
3160
+ return None
3161
+ if (
3162
+ token.word
3163
+ and not token.escaped
3164
+ and token.value in _STRUCTURED_TYPE_WORDS
3165
+ and _is_structured_type_opener(tokens, index)
3166
+ ):
3167
+ return index
3168
+ return None
3169
+
3170
+
3171
+ def _routine_span(
3172
+ document: _SourceDocument,
3173
+ start: int,
3174
+ ) -> tuple[int, int] | None:
3175
+ cached = document.routine_spans.get(start, ...)
3176
+ if cached is not ...:
3177
+ return cached
3178
+ token_index = document.first_token_index(start)
3179
+ found = _find_routine_token_span(
3180
+ document.tokens,
3181
+ document.token_starts,
3182
+ token_index,
3183
+ cache=document.routine_token_spans,
3184
+ )
3185
+ span = (start, found[1]) if found is not None else None
3186
+ document.routine_spans[start] = span
3187
+ return span
3188
+
3189
+
3190
+ def _find_routine_token_span(
3191
+ tokens: tuple[_Token, ...],
3192
+ token_starts: tuple[int, ...],
3193
+ start_index: int,
3194
+ *,
3195
+ cache: dict[int, tuple[int, int, int] | None] | None = None,
3196
+ depth: int = 0,
3197
+ ) -> tuple[int, int, int] | None:
3198
+ if depth > 64:
3199
+ return None
3200
+ routine_index = _routine_keyword_index(tokens, start_index)
3201
+ if routine_index is None:
3202
+ return None
3203
+ spans = cache if cache is not None else {}
3204
+ missing = object()
3205
+ cached = spans.get(routine_index, missing)
3206
+ if cached is not missing:
3207
+ if cached is None:
3208
+ return None
3209
+ return tokens[start_index].start, cached[1], cached[2]
3210
+
3211
+ heading_end = _heading_semicolon_index(tokens, routine_index)
3212
+ if heading_end is None:
3213
+ spans[routine_index] = None
3214
+ return None
3215
+
3216
+ frames: list[list[int]] = [[routine_index, heading_end + 1]]
3217
+
3218
+ def reject_active_frames() -> None:
3219
+ for active_routine_index, _ in frames:
3220
+ spans[active_routine_index] = None
3221
+ frames.clear()
3222
+
3223
+ while frames:
3224
+ frame = frames[-1]
3225
+ frame_routine_index, index = frame
3226
+ if index >= len(tokens):
3227
+ reject_active_frames()
3228
+ continue
3229
+
3230
+ token = tokens[index]
3231
+ if token.directive:
3232
+ reject_active_frames()
3233
+ continue
3234
+ if token.word and not token.escaped:
3235
+ if token.value in _NO_BODY_DIRECTIVES:
3236
+ spans[frame_routine_index] = None
3237
+ frames.pop()
3238
+ continue
3239
+ if token.value in {"implementation", "initialization", "finalization"}:
3240
+ reject_active_frames()
3241
+ continue
3242
+ if (
3243
+ token.value in _STRUCTURED_TYPE_WORDS
3244
+ and _is_structured_type_opener(tokens, index)
3245
+ ):
3246
+ if index + 1 < len(tokens) and tokens[index + 1].value == ";":
3247
+ frame[1] = index + 2
3248
+ continue
3249
+ structured_end = _match_end_terminated_block(tokens, index)
3250
+ if structured_end is None:
3251
+ reject_active_frames()
3252
+ continue
3253
+ frame[1] = bisect_left(token_starts, structured_end)
3254
+ continue
3255
+ if token.value == "end":
3256
+ reject_active_frames()
3257
+ continue
3258
+ if token.value in {"begin", "asm"}:
3259
+ end = _match_end_terminated_block(tokens, index)
3260
+ if end is None:
3261
+ reject_active_frames()
3262
+ continue
3263
+ end_index = bisect_left(token_starts, end)
3264
+ spans[frame_routine_index] = (
3265
+ tokens[frame_routine_index].start,
3266
+ end,
3267
+ end_index,
3268
+ )
3269
+ frames.pop()
3270
+ continue
3271
+ if token.value in _ROUTINE_WORDS and _is_nested_routine_declaration(tokens, index):
3272
+ nested_routine_index = _routine_keyword_index(tokens, index)
3273
+ if nested_routine_index is None:
3274
+ frame[1] = index + 1
3275
+ continue
3276
+ nested = spans.get(nested_routine_index, missing)
3277
+ if nested is missing:
3278
+ nested_heading_end = _heading_semicolon_index(tokens, nested_routine_index)
3279
+ if nested_heading_end is None:
3280
+ spans[nested_routine_index] = None
3281
+ continue
3282
+ frames.append([nested_routine_index, nested_heading_end + 1])
3283
+ continue
3284
+ if nested is not None:
3285
+ frame[1] = max(index + 1, nested[2])
3286
+ continue
3287
+ skipped = _routine_declaration_end_index(tokens, index)
3288
+ if skipped is not None:
3289
+ frame[1] = skipped + 1
3290
+ continue
3291
+ frame[1] = index + 1
3292
+
3293
+ result = spans.get(routine_index)
3294
+ if result is None:
3295
+ return None
3296
+ return tokens[start_index].start, result[1], result[2]
3297
+
3298
+
3299
+ def _routine_keyword_index(
3300
+ tokens: tuple[_Token, ...],
3301
+ start_index: int,
3302
+ ) -> int | None:
3303
+ for index in range(start_index, min(len(tokens), start_index + 6)):
3304
+ token = tokens[index]
3305
+ if token.word and not token.escaped and token.value in _ROUTINE_WORDS:
3306
+ return index
3307
+ if token.value == ";":
3308
+ return None
3309
+ return None
3310
+
3311
+
3312
+ def _heading_semicolon_index(
3313
+ tokens: tuple[_Token, ...],
3314
+ routine_index: int,
3315
+ ) -> int | None:
3316
+ return _top_level_semicolon_index(tokens, routine_index + 1)
3317
+
3318
+
3319
+ def _routine_declaration_end_index(
3320
+ tokens: tuple[_Token, ...],
3321
+ routine_index: int,
3322
+ ) -> int | None:
3323
+ declaration_end = _heading_semicolon_index(tokens, routine_index)
3324
+ if declaration_end is None:
3325
+ return None
3326
+ cursor = declaration_end + 1
3327
+ while cursor < len(tokens):
3328
+ directive = tokens[cursor]
3329
+ if (
3330
+ not directive.word
3331
+ or directive.escaped
3332
+ or directive.value not in _ROUTINE_DIRECTIVES
3333
+ ):
3334
+ break
3335
+ directive_end = _top_level_semicolon_index(tokens, cursor + 1)
3336
+ if directive_end is None:
3337
+ break
3338
+ declaration_end = directive_end
3339
+ cursor = directive_end + 1
3340
+ return declaration_end
3341
+
3342
+
3343
+ def _top_level_semicolon_index(
3344
+ tokens: tuple[_Token, ...],
3345
+ start_index: int,
3346
+ ) -> int | None:
3347
+ parentheses = 0
3348
+ brackets = 0
3349
+ angles = 0
3350
+ for index in range(start_index, len(tokens)):
3351
+ value = tokens[index].value
3352
+ if value == "(":
3353
+ parentheses += 1
3354
+ elif value == ")":
3355
+ parentheses = max(0, parentheses - 1)
3356
+ elif value == "[":
3357
+ brackets += 1
3358
+ elif value == "]":
3359
+ brackets = max(0, brackets - 1)
3360
+ elif value == "<":
3361
+ angles += 1
3362
+ elif value == ">":
3363
+ angles = max(0, angles - 1)
3364
+ elif value == ";" and parentheses == 0 and brackets == 0 and angles == 0:
3365
+ return index
3366
+ return None
3367
+
3368
+
3369
+ def _top_level_routine_boundary_index(
3370
+ tokens: tuple[_Token, ...],
3371
+ start_index: int,
3372
+ stop_index: int,
3373
+ ) -> int | None:
3374
+ parentheses = 0
3375
+ brackets = 0
3376
+ angles = 0
3377
+ for index in range(start_index, min(stop_index, len(tokens))):
3378
+ token = tokens[index]
3379
+ value = token.value
3380
+ if (
3381
+ parentheses == 0
3382
+ and brackets == 0
3383
+ and angles == 0
3384
+ and token.word
3385
+ and not token.escaped
3386
+ and value in _ROUTINE_WORDS
3387
+ and _is_nested_routine_declaration(tokens, index)
3388
+ ):
3389
+ return index
3390
+ if value == "(":
3391
+ parentheses += 1
3392
+ elif value == ")":
3393
+ parentheses = max(0, parentheses - 1)
3394
+ elif value == "[":
3395
+ brackets += 1
3396
+ elif value == "]":
3397
+ brackets = max(0, brackets - 1)
3398
+ elif value == "<":
3399
+ angles += 1
3400
+ elif value == ">":
3401
+ angles = max(0, angles - 1)
3402
+ return None
3403
+
3404
+
3405
+ def _is_nested_routine_declaration(
3406
+ tokens: tuple[_Token, ...],
3407
+ index: int,
3408
+ ) -> bool:
3409
+ token = tokens[index]
3410
+ if token.value == "operator":
3411
+ return True
3412
+ previous = tokens[index - 1] if index > 0 else None
3413
+ if previous is not None and previous.value in {":", "=", "of", "to", "reference", "."}:
3414
+ return False
3415
+ following = tokens[index + 1] if index + 1 < len(tokens) else None
3416
+ if following is None or not following.word or following.value in {"of", "object"}:
3417
+ return False
3418
+ return True
3419
+
3420
+
3421
+ def _match_end_terminated_block(
3422
+ tokens: tuple[_Token, ...],
3423
+ opener_index: int,
3424
+ ) -> int | None:
3425
+ stack = [tokens[opener_index].value]
3426
+ for index in range(opener_index + 1, len(tokens)):
3427
+ token = tokens[index]
3428
+ if token.directive:
3429
+ return None
3430
+ if not token.word or token.escaped:
3431
+ continue
3432
+ value = token.value
3433
+ if stack[-1] == "asm":
3434
+ if value != "end":
3435
+ continue
3436
+ elif value in _BLOCK_WORDS:
3437
+ if value == "case" and stack[-1] in _STRUCTURED_TYPE_WORDS:
3438
+ continue
3439
+ stack.append(value)
3440
+ continue
3441
+ elif value in _STRUCTURED_TYPE_WORDS and _is_structured_type_opener(tokens, index):
3442
+ stack.append(value)
3443
+ continue
3444
+ elif value != "end":
3445
+ continue
3446
+
3447
+ stack.pop()
3448
+ if stack:
3449
+ continue
3450
+ end = token.end
3451
+ if index + 1 < len(tokens) and tokens[index + 1].value in {";", "."}:
3452
+ end = tokens[index + 1].end
3453
+ return end
3454
+ return None
3455
+
3456
+
3457
+ def _is_structured_type_opener(tokens: tuple[_Token, ...], index: int) -> bool:
3458
+ token = tokens[index]
3459
+ previous = tokens[index - 1] if index > 0 else None
3460
+ following = tokens[index + 1] if index + 1 < len(tokens) else None
3461
+ if token.value == "class" and following is not None and following.value == "of":
3462
+ return False
3463
+ if previous is not None and previous.value == "of":
3464
+ return token.value in {"record", "object"} and _is_array_of_context(tokens, index)
3465
+ if previous is not None and previous.value == ":" and _inside_generic_angles(tokens, index):
3466
+ return False
3467
+ if previous is None:
3468
+ return False
3469
+ if previous.value in {"=", ":", "packed"}:
3470
+ return True
3471
+ if previous.value == "^" and token.value in {"record", "object"}:
3472
+ return True
3473
+ return False
3474
+
3475
+
3476
+ def _is_array_of_context(tokens: tuple[_Token, ...], index: int) -> bool:
3477
+ for candidate in range(index - 2, max(-1, index - 200), -1):
3478
+ token = tokens[candidate]
3479
+ if token.value == "array":
3480
+ return True
3481
+ if token.value in {"procedure", "function", "reference", "=", ";"}:
3482
+ return False
3483
+ return False
3484
+
3485
+
3486
+ def _inside_generic_angles(tokens: tuple[_Token, ...], index: int) -> bool:
3487
+ depth = 0
3488
+ for candidate in range(index - 1, max(-1, index - 100), -1):
3489
+ value = tokens[candidate].value
3490
+ if value == ">":
3491
+ depth += 1
3492
+ elif value == "<":
3493
+ if depth == 0:
3494
+ return True
3495
+ depth -= 1
3496
+ elif depth == 0 and value in {";", "begin", "end"}:
3497
+ return False
3498
+ return False
3499
+
3500
+
3501
+ def _source_items(
3502
+ document: _SourceDocument,
3503
+ start: int,
3504
+ end: int,
3505
+ max_chars: int,
3506
+ *,
3507
+ role: str,
3508
+ target_id: str,
3509
+ ) -> list[dict[str, object]]:
3510
+ start = min(max(start, 0), len(document.text))
3511
+ end = min(max(end, start), len(document.text))
3512
+ probe_end = min(end, start + 1)
3513
+ compact = max_chars <= 256 or len(
3514
+ _compact_json(
3515
+ _source_item(
3516
+ document,
3517
+ start,
3518
+ probe_end,
3519
+ role=role,
3520
+ target_id=target_id,
3521
+ chunk_index=999999,
3522
+ chunk_count=999999,
3523
+ compact=False,
3524
+ )
3525
+ )
3526
+ ) + 2 >= max_chars
3527
+ spans: list[tuple[int, int]] = []
3528
+ offset = start
3529
+ while offset < end:
3530
+ upper = min(end, offset + _SOURCE_CHUNK_CHARS)
3531
+ accepted = _fit_source_end(
3532
+ document,
3533
+ offset,
3534
+ upper,
3535
+ max_chars,
3536
+ role=role,
3537
+ target_id=target_id,
3538
+ compact=compact,
3539
+ )
3540
+ if accepted <= offset:
3541
+ raise AgentProtocolError(
3542
+ "item_too_large",
3543
+ "max_chars is too small for a typed source chunk.",
3544
+ )
3545
+ spans.append((offset, accepted))
3546
+ offset = accepted
3547
+ if not spans:
3548
+ spans.append((start, end))
3549
+
3550
+ total = len(spans)
3551
+ return [
3552
+ _source_item(
3553
+ document,
3554
+ item_start,
3555
+ item_end,
3556
+ role=role,
3557
+ target_id=target_id,
3558
+ chunk_index=index,
3559
+ chunk_count=total,
3560
+ compact=compact,
3561
+ )
3562
+ for index, (item_start, item_end) in enumerate(spans)
3563
+ ]
3564
+
3565
+
3566
+ def _fit_source_end(
3567
+ document: _SourceDocument,
3568
+ start: int,
3569
+ upper: int,
3570
+ max_chars: int,
3571
+ *,
3572
+ role: str,
3573
+ target_id: str,
3574
+ compact: bool,
3575
+ ) -> int:
3576
+ low = start + 1
3577
+ high = upper
3578
+ accepted = start
3579
+ while low <= high:
3580
+ candidate_end = (low + high) // 2
3581
+ candidate = _source_item(
3582
+ document,
3583
+ start,
3584
+ candidate_end,
3585
+ role=role,
3586
+ target_id=target_id,
3587
+ chunk_index=999999,
3588
+ chunk_count=999999,
3589
+ compact=compact,
3590
+ )
3591
+ if len(_compact_json(candidate)) + 2 <= max_chars:
3592
+ accepted = candidate_end
3593
+ low = candidate_end + 1
3594
+ else:
3595
+ high = candidate_end - 1
3596
+ return accepted
3597
+
3598
+
3599
+ def _source_item(
3600
+ document: _SourceDocument,
3601
+ start: int,
3602
+ end: int,
3603
+ *,
3604
+ role: str,
3605
+ target_id: str,
3606
+ chunk_index: int,
3607
+ chunk_count: int,
3608
+ compact: bool,
3609
+ ) -> dict[str, object]:
3610
+ start_line, start_col = document.line_col(start)
3611
+ end_line, end_col = document.line_col(end)
3612
+ item: dict[str, object] = {
3613
+ "item_type": "source_chunk" if compact else "source",
3614
+ "path": document.display_path,
3615
+ "start_line": start_line,
3616
+ "start_col": start_col,
3617
+ "end_line": end_line,
3618
+ "end_col": end_col,
3619
+ "chunk_index": chunk_index,
3620
+ "chunk_count": chunk_count,
3621
+ "text": document.text[start:end],
3622
+ }
3623
+ if not compact:
3624
+ item["role"] = role
3625
+ item["target_id"] = target_id
3626
+ return item
3627
+
3628
+
3629
+ def _line_starts(text: str) -> tuple[int, ...]:
3630
+ starts = [0]
3631
+ starts.extend(index + 1 for index, character in enumerate(text) if character == "\n")
3632
+ return tuple(starts)
3633
+
3634
+
3635
+ def _lex_delphi(text: str) -> list[_Token]:
3636
+ tokens: list[_Token] = []
3637
+ index = 0
3638
+ length = len(text)
3639
+ while index < length:
3640
+ character = text[index]
3641
+ if character.isspace():
3642
+ index += 1
3643
+ continue
3644
+ if text.startswith("//", index):
3645
+ newline = text.find("\n", index + 2)
3646
+ index = length if newline < 0 else newline + 1
3647
+ continue
3648
+ if character == "{":
3649
+ close = text.find("}", index + 1)
3650
+ end = length if close < 0 else close + 1
3651
+ if index + 1 < length and text[index + 1] == "$":
3652
+ tokens.append(
3653
+ _Token(
3654
+ unicodedata.normalize("NFC", text[index:end]),
3655
+ index,
3656
+ end,
3657
+ directive=True,
3658
+ )
3659
+ )
3660
+ index = end
3661
+ continue
3662
+ if text.startswith("(*", index):
3663
+ close = text.find("*)", index + 2)
3664
+ end = length if close < 0 else close + 2
3665
+ if index + 2 < length and text[index + 2] == "$":
3666
+ tokens.append(
3667
+ _Token(
3668
+ unicodedata.normalize("NFC", text[index:end]),
3669
+ index,
3670
+ end,
3671
+ directive=True,
3672
+ )
3673
+ )
3674
+ index = end
3675
+ continue
3676
+ if character == "'":
3677
+ block_end = multiline_string_block_end(text, index)
3678
+ index = block_end if block_end is not None else _quoted_end(text, index)
3679
+ continue
3680
+ if character == "&" and index + 1 < length and _identifier_start(text[index + 1]):
3681
+ end = index + 2
3682
+ while end < length and _identifier_part(text[end]):
3683
+ end += 1
3684
+ tokens.append(_Token(text[index + 1:end].casefold(), index, end, word=True, escaped=True))
3685
+ index = end
3686
+ continue
3687
+ if _identifier_start(character):
3688
+ end = index + 1
3689
+ while end < length and _identifier_part(text[end]):
3690
+ end += 1
3691
+ tokens.append(_Token(text[index:end].casefold(), index, end, word=True))
3692
+ index = end
3693
+ continue
3694
+ if text[index:index + 2] in {":=", "<=", ">=", "<>", ".."}:
3695
+ tokens.append(_Token(text[index:index + 2], index, index + 2))
3696
+ index += 2
3697
+ continue
3698
+ tokens.append(_Token(character, index, index + 1))
3699
+ index += 1
3700
+ return tokens
3701
+
3702
+
3703
+ def _quoted_end(text: str, start: int) -> int:
3704
+ index = start + 1
3705
+ while index < len(text):
3706
+ if text[index] != "'":
3707
+ index += 1
3708
+ continue
3709
+ if index + 1 < len(text) and text[index + 1] == "'":
3710
+ index += 2
3711
+ continue
3712
+ return index + 1
3713
+ return len(text)
3714
+
3715
+
3716
+ def _identifier_start(character: str) -> bool:
3717
+ return character == "_" or character.isalpha()
3718
+
3719
+
3720
+ def _identifier_part(character: str) -> bool:
3721
+ return character == "_" or character.isalnum()
3722
+
3723
+
3724
+ def _ranked_entries(entries: tuple[_SymbolEntry, ...], query: str) -> list[_SymbolEntry]:
3725
+ normalized_query = _normalized(query.strip())
3726
+ ranked: list[tuple[int, tuple[object, ...], _SymbolEntry]] = []
3727
+ for entry in entries:
3728
+ normalized_qualified = entry.normalized_qualified_name
3729
+ relative_offset = entry.relative_name_offset
3730
+ if not normalized_query:
3731
+ rank = 3
3732
+ elif (
3733
+ entry.normalized_name == normalized_query
3734
+ or normalized_qualified == normalized_query
3735
+ or (
3736
+ len(normalized_qualified) - relative_offset == len(normalized_query)
3737
+ and normalized_qualified.endswith(normalized_query)
3738
+ )
3739
+ ):
3740
+ rank = 0
3741
+ elif (
3742
+ entry.normalized_name.startswith(normalized_query)
3743
+ or normalized_qualified.startswith(normalized_query)
3744
+ or normalized_qualified.startswith(normalized_query, relative_offset)
3745
+ ):
3746
+ rank = 1
3747
+ elif (
3748
+ normalized_query in entry.normalized_name
3749
+ or normalized_query in normalized_qualified
3750
+ ):
3751
+ rank = 2
3752
+ else:
3753
+ continue
3754
+ ranked.append((rank, _entry_sort_key(entry), entry))
3755
+ ranked.sort(key=lambda item: (item[0], item[1]))
3756
+ return [item[2] for item in ranked]
3757
+
3758
+
3759
+ def _symbol_sort_key(symbol: Symbol) -> tuple[object, ...]:
3760
+ return (
3761
+ symbol.decl_range.start_line,
3762
+ symbol.decl_range.start_col,
3763
+ symbol.kind.value.casefold(),
3764
+ _normalized(symbol.name),
3765
+ symbol.name,
3766
+ )
3767
+
3768
+
3769
+ def _raw_sort_key(raw: _RawSymbol) -> tuple[object, ...]:
3770
+ return (
3771
+ raw.path.casefold(),
3772
+ raw.path,
3773
+ raw.line,
3774
+ raw.column,
3775
+ raw.kind.value.casefold(),
3776
+ _normalized(raw.qualified_name),
3777
+ raw.qualified_name,
3778
+ )
3779
+
3780
+
3781
+ def _entry_sort_key(entry: _SymbolEntry) -> tuple[object, ...]:
3782
+ return (
3783
+ entry.normalized_qualified_name,
3784
+ entry.kind.value.casefold(),
3785
+ entry.path.casefold(),
3786
+ entry.path,
3787
+ entry.line,
3788
+ entry.column,
3789
+ entry.ordinal,
3790
+ entry.target_id,
3791
+ )
3792
+
3793
+
3794
+ def _normalized(value: str) -> str:
3795
+ return unicodedata.normalize("NFC", value).casefold()
3796
+
3797
+
3798
+ def _normalized_search_fields(
3799
+ name: str,
3800
+ qualified_name: str,
3801
+ unit_name: str,
3802
+ ) -> tuple[str, str, int]:
3803
+ normalized_name = _normalized(name)
3804
+ normalized_qualified = _normalized(qualified_name)
3805
+ prefix = f"{_normalized(unit_name)}."
3806
+ relative_name_offset = (
3807
+ len(prefix)
3808
+ if normalized_qualified.startswith(prefix)
3809
+ else 0
3810
+ )
3811
+ return normalized_name, normalized_qualified, relative_name_offset
3812
+
3813
+
3814
+ def _request_fingerprint(
3815
+ request: AgentRequest,
3816
+ *,
3817
+ project_id: str,
3818
+ target_id: str,
3819
+ ) -> str:
3820
+ payload = {
3821
+ "action": request.action,
3822
+ "depth": request.depth,
3823
+ "detail": request.detail,
3824
+ "direction": request.direction,
3825
+ "graph": request.graph,
3826
+ "max_chars": request.max_chars,
3827
+ "max_items": request.max_items,
3828
+ "project_id": project_id,
3829
+ "query": request.query,
3830
+ "relation": request.relation,
3831
+ "target_id": target_id,
3832
+ }
3833
+ encoded = _compact_json(payload).encode("utf-8")
3834
+ return f"agent_request_v3_{hashlib.sha256(encoded).hexdigest()}"
3835
+
3836
+
3837
+ def _prepare_items(
3838
+ items: Sequence[dict[str, object]],
3839
+ max_chars: int,
3840
+ ) -> list[dict[str, object]]:
3841
+ prepared: list[dict[str, object]] = []
3842
+ for item in items:
3843
+ if len(_compact_json(item)) + 2 <= max_chars:
3844
+ prepared.append(item)
3845
+ continue
3846
+ prepared.extend(_json_chunks(item, max_chars))
3847
+ return prepared
3848
+
3849
+
3850
+ def _json_chunks(item: dict[str, object], max_chars: int) -> list[dict[str, object]]:
3851
+ serialized = _compact_json(item)
3852
+ if item.get("item_type") in {"source", "source_chunk"}:
3853
+ item_type = "source_chunk"
3854
+ else:
3855
+ item_type = "card_chunk" if "target_id" in item else "json_chunk"
3856
+ chunks: list[str] = []
3857
+ offset = 0
3858
+ while offset < len(serialized):
3859
+ low = 1
3860
+ high = len(serialized) - offset
3861
+ accepted = 0
3862
+ while low <= high:
3863
+ size = (low + high) // 2
3864
+ candidate = {
3865
+ "item_type": item_type,
3866
+ "chunk_index": len(chunks),
3867
+ "chunk_count": 999999,
3868
+ "json": serialized[offset:offset + size],
3869
+ }
3870
+ if len(_compact_json(candidate)) + 2 <= max_chars:
3871
+ accepted = size
3872
+ low = size + 1
3873
+ else:
3874
+ high = size - 1
3875
+ if accepted == 0:
3876
+ raise AgentProtocolError(
3877
+ "item_too_large",
3878
+ "max_chars is too small for a structured response chunk.",
3879
+ )
3880
+ chunks.append(serialized[offset:offset + accepted])
3881
+ offset += accepted
3882
+ total = len(chunks)
3883
+ return [
3884
+ {
3885
+ "item_type": item_type,
3886
+ "chunk_index": index,
3887
+ "chunk_count": total,
3888
+ "json": chunk,
3889
+ }
3890
+ for index, chunk in enumerate(chunks)
3891
+ ]
3892
+
3893
+
3894
+ def _compact_json(value: object) -> str:
3895
+ try:
3896
+ return json.dumps(
3897
+ value,
3898
+ ensure_ascii=False,
3899
+ sort_keys=True,
3900
+ separators=(",", ":"),
3901
+ allow_nan=False,
3902
+ )
3903
+ except (TypeError, ValueError):
3904
+ raise AgentProtocolError("invalid_item", "Item is not JSON-compatible.") from None
3905
+
3906
+
3907
+ __all__ = ["AgentContext"]