open-context-engine 0.1.2 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,312 @@
1
+ """Evidence-oriented retrieval with per-question scoring and complete snippets.
2
+
3
+ The historical engines remain available for frozen evaluations. This engine
4
+ keeps independent recall channels, scores actual subquestions, and selects
5
+ implementation evidence before supporting tests when that is what was asked.
6
+ """
7
+ from collections import defaultdict
8
+ import math
9
+ import re
10
+ import time
11
+
12
+ import numpy as np
13
+
14
+ from engine import Engine, document, post
15
+ from reranker import rerank_pairs
16
+
17
+ VERSION = 'evidence-v2'
18
+
19
+
20
+ def source_role(unit):
21
+ path = unit['path'].lower()
22
+ if re.search(r'(^|/)(tests?|__tests__|fixtures)(/|$)|(^|/)test_[^/]+|[._](test|spec)\.', path):
23
+ return 'test'
24
+ if unit.get('language') == 'text' or re.search(r'(^|/)(docs?|examples?)/|\.(md|rst|txt)$', path):
25
+ return 'documentation'
26
+ return 'implementation'
27
+
28
+
29
+ def requested_role(query):
30
+ if re.search(r'^(?:find|show|locate|list)\s+(?:the\s+)?(?:(?:unit|regression)\s+)?tests?\b|^(?:查找|找到|列出|展示).{0,5}(?:测试|用例)', query, re.I):
31
+ return 'test'
32
+ if re.search(r'^(?:find|show|locate|list)\s+(?:the\s+)?(?:docs?|documentation|readme)\b|^(?:查找|找到|列出|展示).{0,5}文档', query, re.I):
33
+ return 'documentation'
34
+ return 'implementation'
35
+
36
+
37
+ def requested_languages(query):
38
+ """Honor explicit language scope without guessing from repository names."""
39
+ languages = set()
40
+ for pattern, values in [
41
+ (r'\bpython\b', {'python'}),
42
+ (r'\bgolang\b', {'go'}),
43
+ (r'\btypescript\b', {'typescript'}),
44
+ (r'\bjavascript\b', {'javascript', 'typescript'}),
45
+ (r'\b(?:frontend|front-end|react)\b|前端', {'javascript', 'typescript'}),
46
+ ]:
47
+ if re.search(pattern, query, re.I):
48
+ languages.update(values)
49
+ if re.search(r'\bGo\b|\bgo\s+(?:code|service|backend|worker|implementation)\b', query):
50
+ languages.add('go')
51
+ return languages
52
+
53
+
54
+ class EvidenceEngine(Engine):
55
+ version = VERSION
56
+
57
+ def __init__(self, *args, **kwargs):
58
+ super().__init__(*args, **kwargs)
59
+ self.roles = [source_role(u) for u in self.units]
60
+ self.symbols = defaultdict(list)
61
+ self.neighbors = [set() for _ in self.units]
62
+ self.callees = [set() for _ in self.units]
63
+ for u in self.units:
64
+ if u.get('name'):
65
+ self.symbols[u['symbol']].append(u['id'])
66
+ for relation in u.get('relations', []):
67
+ if relation['kind'] not in {'calls', 'same_symbol', 'references_value'}:
68
+ continue
69
+ target = relation['target']
70
+ self.neighbors[u['id']].add(target)
71
+ self.neighbors[target].add(u['id'])
72
+ if relation['kind'] in {'calls', 'references_value'}:
73
+ self.callees[u['id']].add(target)
74
+
75
+ def substance(self, uid):
76
+ unit = self.units[uid]
77
+ if self.roles[uid] != 'implementation':
78
+ return 1.0
79
+ if unit['kind'] in {'property', 'parameter', 'type', 'interface', 'type-alias'}:
80
+ return .25
81
+ lines = [line.strip() for line in unit['text'].splitlines() if line.strip()]
82
+ executable = [line for line in lines if not line.startswith(
83
+ ('//', '/*', '*', '#', 'import ', 'from ', 'export type ', '}'))]
84
+ if not executable:
85
+ return .1
86
+ if unit['kind'] in {'function', 'method'}:
87
+ return 1.0 if len(executable) >= 2 else .5
88
+ return 1.0 if len(executable) >= 5 else .25
89
+
90
+ def bundle(self, uid, budget):
91
+ unit = self.units[uid]
92
+ family = self.symbols.get(unit['symbol'], [uid]) if unit.get('name') else [uid]
93
+ if unit['kind'] not in {'function', 'method'}:
94
+ return [uid]
95
+ if sum(self.costs[i] for i in family) <= 3200:
96
+ return sorted(family, key=lambda i: self.units[i]['start'])
97
+ return [uid]
98
+
99
+ def render_selection(self, selected):
100
+ # Coalesce only returned source; never invent lines between snippets.
101
+ paths = {}
102
+ for uid in selected:
103
+ unit = self.units[uid]
104
+ lines = paths.setdefault(unit['path'], {})
105
+ for number, line in enumerate(unit['text'].split('\n'), unit['start']):
106
+ lines[number] = line
107
+ blocks = []
108
+ for path, lines in paths.items():
109
+ previous = None
110
+ for number in sorted(lines):
111
+ if previous is None or number != previous + 1:
112
+ blocks.append('Path: ' + path)
113
+ blocks.append(f'{number}\t{lines[number]}')
114
+ previous = number
115
+ blocks.append('')
116
+ return '\n'.join(blocks)
117
+
118
+ def search(self, plan, budget=4000):
119
+ started = time.monotonic()
120
+ declarations = bool(re.search(r'\b(interface|type alias|schema)\b|类型定义|接口类型', plan['intent'], re.I))
121
+ def substance(uid):
122
+ if declarations and self.units[uid]['kind'] in {'type', 'interface', 'type-alias'}:
123
+ return 1.0
124
+ return self.substance(uid)
125
+ intent = plan['intent']
126
+ languages = requested_languages(intent)
127
+ scoped = np.asarray([not languages or u.get('language') in languages for u in self.units])
128
+ facets = list(dict.fromkeys(f['question'] for f in plan['facets'] if f['question'] != intent))[:4]
129
+ queries = [intent] + [facet + '\nContext for this part of the request: ' + intent for facet in facets]
130
+ result = post(self.embed_url + '/embeddings', {'model': self.embedding_model,
131
+ 'input': ['Instruct: Retrieve source code implementing the requested behavior.\nQuery: ' + q for q in queries]}, self.embedding_key)
132
+ rows = sorted(result['data'], key=lambda row: row['index'])
133
+ if [row['index'] for row in rows] != list(range(len(queries))):
134
+ raise ValueError('Invalid query embedding indices')
135
+ matrix = np.asarray([row['embedding'] for row in rows], dtype=np.float32)
136
+ if matrix.shape != (len(queries), self.vectors.shape[1]) or not np.isfinite(matrix).all():
137
+ raise ValueError('Invalid query embedding vectors')
138
+ dense = self.vectors @ matrix.T
139
+ candidates, recall = set(), []
140
+ for col, query in enumerate(queries):
141
+ channels = []
142
+ for scores in (dense[:, col], self.lexical(query)):
143
+ order = np.argsort(-scores, kind='stable')
144
+ # Candidate depth is independent of the response token budget.
145
+ # Keep global recall and add a separate explicit-language lane.
146
+ ids = [int(uid) for uid in order[:40] if scores[uid] > 0]
147
+ if languages:
148
+ local = [int(uid) for uid in order if scoped[uid] and scores[uid] > 0][:80]
149
+ ids = list(dict.fromkeys(ids + local))
150
+ candidates.update(ids)
151
+ channels.append(ids)
152
+ recall.append(channels)
153
+ scores, waves = {}, []
154
+
155
+ def score(ids):
156
+ ids = sorted(ids)
157
+ pairs = [(q, j) for q in range(len(queries)) for j, uid in enumerate(ids) if (q, uid) not in scores]
158
+ if not pairs:
159
+ return
160
+ before = time.monotonic()
161
+ result = rerank_pairs(self.reranker, queries, [document(self.units[i], 5000) for i in ids], pairs, post)
162
+ for row in result['results']:
163
+ q, j = pairs[row['index']]
164
+ scores[q, ids[j]] = row['relevance_score']
165
+ waves.append({'pairs': len(pairs), 'requests': result['meta']['request_count'],
166
+ 'elapsedMs': round((time.monotonic() - before) * 1000)})
167
+
168
+ score(candidates)
169
+ primary = requested_role(plan['intent'])
170
+ seeds = set()
171
+ for q in range(len(queries)):
172
+ order = sorted(candidates, key=lambda uid: (-scores[q, uid], uid))
173
+ seeds.update(order[:8])
174
+ seeds.update([i for i in order if self.roles[i] == primary and substance(i) >= .5][:6])
175
+ support = defaultdict(float)
176
+ for uid in sorted(seeds):
177
+ outgoing = self.callees[uid]
178
+ neighbors = outgoing | self.neighbors[uid]
179
+ # Shared utility callers are broad hubs, not reliable task evidence.
180
+ if len(neighbors) > 24:
181
+ neighbors = outgoing
182
+ for neighbor in neighbors:
183
+ if neighbor not in candidates:
184
+ confidence = max(scores[q, uid] for q in range(len(queries)))
185
+ support[neighbor] = max(support[neighbor], confidence / max(1, len(neighbors)))
186
+ expanded = sorted(support, key=lambda uid: (-support[uid], uid))[:80]
187
+ candidates.update(expanded)
188
+ score(expanded)
189
+
190
+ completions = {i for uid in candidates for i in self.bundle(uid, budget)} - candidates
191
+ score(completions)
192
+ candidates.update(completions)
193
+
194
+ ranks = {}
195
+ for q in range(len(queries)):
196
+ for role in ('implementation', 'test', 'documentation'):
197
+ order = sorted((i for i in candidates if self.roles[i] == role),
198
+ key=lambda uid: (-scores[q, uid], uid))
199
+ ranks.update({(q, uid): rank for rank, uid in enumerate(order)})
200
+
201
+ def value(q, uid):
202
+ relevance = scores[q, uid]
203
+ if relevance < .1:
204
+ return 0.0
205
+ # Rank separates saturated probability-like scores. No inverse cost
206
+ # reward: a short test must not displace a better-ranked body.
207
+ return (.5 * relevance + .5 * math.exp(-ranks[q, uid] / 8)) * substance(uid) * max(0.0, scores[0, uid]) * (1.0 if scoped[uid] else .25)
208
+
209
+ # Tests often describe behavior more explicitly than the implementation.
210
+ # Propagate relevance only along resolved source references, bounded by
211
+ # the test's fan-out. This is evidence support, not invented call edges.
212
+ corroboration = defaultdict(float)
213
+ for uid in candidates:
214
+ if self.roles[uid] != 'test':
215
+ continue
216
+ targets = [i for i in self.callees[uid] if i in candidates and self.roles[i] == 'implementation']
217
+ if not targets or len(targets) > 8:
218
+ continue
219
+ for q in range(len(queries)):
220
+ if scores[q, uid] < .1:
221
+ continue
222
+ strength = (.5 * scores[q, uid] + .5 * math.exp(-ranks[q, uid] / 8)) / math.sqrt(len(targets))
223
+ for target in targets:
224
+ corroboration[q, target] = max(corroboration[q, target], strength)
225
+ selected, selected_set, trace = [], set(), []
226
+ spent = 0
227
+ path_counts, covered = defaultdict(int), np.zeros(len(queries))
228
+ symbol_counts = defaultdict(int)
229
+
230
+ def choose(ids, q=None, phase='fill', limit=None):
231
+ nonlocal spent
232
+ choices = []
233
+ for uid in ids:
234
+ if uid in selected_set:
235
+ continue
236
+ bundle = [uid]
237
+ if self.roles[uid] == primary == 'implementation':
238
+ # Relevant local helpers are part of the entry point explanation.
239
+ # Keep their source evidence with the calling implementation.
240
+ references = {r['target'] for r in self.units[uid].get('relations', [])
241
+ if r.get('resolution') == 'explicit-doc-reference'}
242
+ # Resolved local callees explain the selected entry point.
243
+ # Bound fan-out, cost and relevance so utility hubs cannot
244
+ # pull arbitrary dependencies into the answer.
245
+ local_calls = {i for i in self.callees[uid]
246
+ if self.units[i]['path'] == self.units[uid]['path']}
247
+ if len({self.units[i]['symbol'] for i in local_calls}) <= 8:
248
+ references.update(local_calls)
249
+ extras = [i for i in references if i in candidates and i not in selected_set and i != uid
250
+ and self.roles[i] == 'implementation' and substance(i) >= .5
251
+ and self.costs[i] <= 1000 and scores[0, i] >= .5]
252
+ extras.sort(key=lambda i: (-scores[0, i], i))
253
+ extra_cost = 0
254
+ for i in extras[:3]:
255
+ if extra_cost + self.costs[i] <= 1600:
256
+ bundle.append(i)
257
+ extra_cost += self.costs[i]
258
+ cost = sum(self.costs[i] for i in bundle)
259
+ if spent + cost > (budget if limit is None else limit):
260
+ bundle, cost = [uid], self.costs[uid]
261
+ if spent + cost > (budget if limit is None else limit):
262
+ continue
263
+ values = [value(col, uid) + .3 * corroboration[col, uid] for col in range(len(queries))]
264
+ merit = values[q] if q is not None else max(v / (1 + covered[col]) for col, v in enumerate(values))
265
+ merit /= 1 + .1 * path_counts[self.units[uid]['path']] + .8 * symbol_counts[self.units[uid]['symbol']]
266
+ if merit > 0:
267
+ choices.append((merit, -uid, uid, bundle, cost, values))
268
+ if not choices:
269
+ return False
270
+ _, _, uid, bundle, cost, values = max(choices)
271
+ selected.extend(bundle); selected_set.update(bundle); spent += cost
272
+ covered[:] += values
273
+ path_counts[self.units[uid]['path']] += 1
274
+ symbol_counts[self.units[uid]['symbol']] += 1
275
+ trace.append({'id': uid, 'bundle': bundle, 'phase': phase, 'role': self.roles[uid], 'tokens': cost,
276
+ 'scores': [scores[q, uid] for q in range(len(queries))]})
277
+ return True
278
+
279
+ primary_ids = [i for i in candidates if self.roles[i] == primary and substance(i) >= .5]
280
+ facets = list(range(1, len(queries))) or [0]
281
+ # Build a stable 8k core, then add evidence. Candidate scoring and family
282
+ # completion do not depend on the output budget.
283
+ for stage in range(8000, max(8000, budget) + 8000, 8000):
284
+ ceiling = min(stage, budget)
285
+ for _ in range(3):
286
+ for q in facets:
287
+ choose(primary_ids, q, 'primary', ceiling * .82)
288
+ for _ in range(2):
289
+ companions = {i for uid in selected for i in self.callees[uid]
290
+ if i in candidates and self.roles[i] == primary and substance(i) >= .5}
291
+ choose(companions, phase='dependency', limit=ceiling * .82)
292
+ continuations = {i for uid in selected for i in self.bundle(uid, budget)}
293
+ while choose(continuations, phase='continuation', limit=ceiling * .85):
294
+ pass
295
+ for q in facets:
296
+ choose(primary_ids, q, 'primary', ceiling * .85)
297
+ if primary == 'implementation' and re.search(r'\btests?\b|测试|回归', intent, re.I):
298
+ tests = [i for i in candidates if self.roles[i] == 'test']
299
+ related = [i for i in tests if self.callees[i] & selected_set]
300
+ for q in facets:
301
+ choose(related or tests, q, 'support', ceiling)
302
+ while choose(candidates, limit=ceiling):
303
+ pass
304
+ context = self.render_selection(selected)
305
+ tokens = len(self.encoding.encode_ordinary(context))
306
+ if tokens > budget:
307
+ raise ValueError('Evidence packing exceeded token budget')
308
+ return context, {'version': VERSION, 'elapsedMs': round((time.monotonic() - started) * 1000),
309
+ 'tokens': tokens, 'queryCache': False, 'candidateCount': len(candidates),
310
+ 'rerankedCount': len(candidates), 'expandedCount': len(expanded), 'queries': queries,
311
+ 'modelRequests': {'embedding': 1, 'rerank': sum(w['requests'] for w in waves)},
312
+ 'waves': waves, 'selected': trace, 'recall': recall}
@@ -8,7 +8,7 @@ import sys
8
8
 
9
9
  from . import python, typescript, text, go
10
10
  from .files import path_exclusion, read_text
11
- from .schema import SCHEMA_VERSION, SourceFile, validate_units
11
+ from .schema import SCHEMA_VERSION, SourceFile, SourceSyntaxError, validate_units
12
12
 
13
13
  ADAPTERS = {'python': python, 'typescript': typescript, 'javascript': typescript, 'go': go, 'text': text}
14
14
  EXTENSIONS = {'.py': 'python', '.ts': 'typescript', '.tsx': 'typescript',
@@ -25,9 +25,12 @@ def adapter_manifest(files, language_options=None):
25
25
  options = language_options or {}
26
26
  if not isinstance(options, dict) or set(options) - set(ADAPTERS):
27
27
  raise ValueError('Unknown language options')
28
- paths = [Path(__file__), Path(__file__).with_name('schema.py'), Path(__file__).with_name('files.py')]
28
+ paths = [Path(__file__), Path(__file__).with_name('schema.py'), Path(__file__).with_name('files.py'),
29
+ Path(text.__file__)]
29
30
  for language in languages:
30
31
  paths.append(Path(ADAPTERS[language].__file__))
32
+ if language == 'python':
33
+ paths.append(Path(python.__file__).with_name('python_calls.py'))
31
34
  if language in {'typescript', 'javascript'}:
32
35
  paths.append(Path(typescript.__file__).with_suffix('.mjs'))
33
36
  if language == 'go':
@@ -42,6 +45,29 @@ def adapter_manifest(files, language_options=None):
42
45
  'sourceSha256': {path.name: hashlib.sha256(path.read_bytes()).hexdigest() for path in paths}}
43
46
 
44
47
 
48
+ def extract_with_fallback(adapter, sources, max_lines, settings):
49
+ try:
50
+ return adapter.extract(sources, max_lines, settings)
51
+ except SourceSyntaxError as error:
52
+ diagnostics = {item['path']: item for item in error.diagnostics}
53
+ if not diagnostics or not set(diagnostics) <= {source.path for source in sources}:
54
+ raise ValueError('Invalid syntax diagnostic paths') from error
55
+ # Rebuild relations using only valid source. Never preserve partial AST
56
+ # output or let a broken module supply symbols to its healthy callers.
57
+ valid = [source for source in sources if source.path not in diagnostics]
58
+ units = adapter.extract(valid, max_lines, settings) if valid else []
59
+ for source in sources:
60
+ if source.path not in diagnostics:
61
+ continue
62
+ fallback = text.extract([source], max_lines)
63
+ for unit in fallback:
64
+ unit['id'] += len(units)
65
+ if fallback:
66
+ fallback[0]['parseDiagnostic'] = {**diagnostics[source.path], 'fallback': 'text'}
67
+ units.extend(fallback)
68
+ return units
69
+
70
+
45
71
  def source_units(root, files, max_lines=65, language_options=None, report=None, cache=None):
46
72
  if type(max_lines) is not int or max_lines < 1:
47
73
  raise ValueError('max_lines must be a positive integer')
@@ -98,7 +124,7 @@ def source_units(root, files, max_lines=65, language_options=None, report=None,
98
124
  ], sort_keys=True).encode()).hexdigest() if cache is not None else None)
99
125
  previous = cache.get(key) if cache is not None else None
100
126
  extracted = (deepcopy(previous[1]) if previous and previous[0] == fingerprint
101
- else ADAPTERS[language].extract(batch, max_lines, settings))
127
+ else extract_with_fallback(ADAPTERS[language], batch, max_lines, settings))
102
128
  if cache is not None:
103
129
  pending_cache[key] = (fingerprint, deepcopy(extracted))
104
130
  offset = len(units)
@@ -124,6 +150,9 @@ def source_units(root, files, max_lines=65, language_options=None, report=None,
124
150
  cache.clear()
125
151
  cache.update(pending_cache)
126
152
  if report is not None:
153
+ diagnostics = [unit['parseDiagnostic'] for unit in units if 'parseDiagnostic' in unit]
127
154
  report.update(inputFiles=len(files), acceptedFiles=len(sources), excluded=excluded,
128
- fallbackFiles=sum(language_for(source.path) == 'text' for source in sources))
155
+ fallbackFiles=len({source.path for source in sources if language_for(source.path) == 'text'}
156
+ | {item['path'] for item in diagnostics}),
157
+ degradedFiles=len(diagnostics), parseDiagnostics=diagnostics)
129
158
  return units
@@ -8,9 +8,10 @@ import re
8
8
  from pathlib import Path
9
9
  import shutil
10
10
  import subprocess
11
+ import sys
11
12
  import tempfile
12
13
 
13
- from .schema import physical_lines
14
+ from .schema import physical_lines, SourceSyntaxError
14
15
 
15
16
 
16
17
  @lru_cache(maxsize=8)
@@ -54,23 +55,24 @@ def parse(sources, options):
54
55
  if cache.is_symlink() or (os.name == 'posix' and
55
56
  (cache.stat().st_uid != os.getuid() or cache.stat().st_mode & 0o077)):
56
57
  raise RuntimeError('Go parser cache must be private to the current user')
57
- executable = cache/identity
58
+ suffix = '.exe' if sys.platform == 'win32' else ''
59
+ executable = cache/(identity+suffix)
58
60
  if executable.is_symlink():
59
61
  raise RuntimeError('Invalid Go parser cache entry')
60
62
  if not executable.exists():
61
63
  with tempfile.TemporaryDirectory(dir=cache) as directory:
62
- target = Path(directory)/'parser'
64
+ target = Path(directory)/('parser'+suffix)
63
65
  env = {**os.environ, 'GOENV': 'off', 'GOWORK': 'off', 'GOTOOLCHAIN': 'local',
64
66
  'GOPROXY': 'off', 'GO111MODULE': 'off', 'CGO_ENABLED': '0', 'GOFLAGS': '',
65
67
  'GOCACHE': str(cache/'build')}
66
68
  built = subprocess.run([binary, 'build', '-o', str(target), str(source), str(semantic)],
67
- env=env, capture_output=True, text=True, timeout=120)
69
+ env=env, capture_output=True, encoding='utf-8', timeout=120)
68
70
  if built.returncode:
69
71
  raise RuntimeError('Go parser build failed: '+built.stderr[:2000])
70
72
  os.replace(target, executable)
71
73
  result = subprocess.run([str(executable)], input=json.dumps({'options': options, 'files': [
72
74
  {'path': source.path, 'text': source.text} for source in sources]}),
73
- text=True, capture_output=True, timeout=120)
75
+ encoding='utf-8', capture_output=True, timeout=120)
74
76
  if result.returncode:
75
77
  raise ValueError('Go adapter failed: '+result.stderr.strip()[:2000])
76
78
  return json.loads(result.stdout)
@@ -79,12 +81,14 @@ def parse(sources, options):
79
81
  def extract(sources, max_lines=65, options=None):
80
82
  options = settings(options)
81
83
  parsed = parse(sources, options)
84
+ if parsed.get('syntaxErrors'):
85
+ raise SourceSyntaxError(parsed['syntaxErrors'])
82
86
  units, records, targets = [], {}, defaultdict(list)
83
87
  source_by_path = {source.path: source for source in sources}
84
88
  for file in parsed['files']:
85
89
  path = file['path']
86
90
  lines = physical_lines(source_by_path[path].text)
87
- module = str(Path(path).parent)+'::'+file['package']
91
+ module = Path(path).parent.as_posix()+'::'+file['package']
88
92
  root = {'name': '', 'kind': 'module', 'start': 1, 'end': len(lines), 'decl': 0}
89
93
  owners = [root]*len(lines)
90
94
  for entry in file['entries']:
@@ -6,6 +6,7 @@ import (
6
6
  "fmt"
7
7
  "go/ast"
8
8
  "go/parser"
9
+ "go/scanner"
9
10
  "go/token"
10
11
  "os"
11
12
  "path"
@@ -74,10 +75,17 @@ func run() error {
74
75
  objects := map[*ast.Object]Target{}
75
76
  functions := map[string][]Target{}
76
77
  line := func(p token.Pos) int { return fset.PositionFor(p, false).Line }
78
+ diagnostics := []map[string]any{}
77
79
  for _, src := range input.Files {
78
80
  text := strings.ReplaceAll(strings.ReplaceAll(src.Text, "\r\n", "\n"), "\r", "\n")
79
81
  tree, err := parser.ParseFile(fset, src.Path, text, parser.ParseComments|parser.AllErrors)
80
82
  if err != nil {
83
+ if syntax, ok := err.(scanner.ErrorList); ok && len(syntax) > 0 {
84
+ pos := syntax[0].Pos
85
+ diagnostics = append(diagnostics, map[string]any{"path": src.Path, "language": "go",
86
+ "errorType": "SyntaxError", "line": pos.Line, "column": pos.Column})
87
+ continue
88
+ }
81
89
  return err
82
90
  }
83
91
  trees[src.Path] = tree
@@ -93,6 +101,9 @@ func run() error {
93
101
  }
94
102
  }
95
103
  }
104
+ if len(diagnostics) > 0 {
105
+ return json.NewEncoder(os.Stdout).Encode(map[string]any{"compilerVersion": runtime.Version(), "syntaxErrors": diagnostics})
106
+ }
96
107
  var typed TypeEvidence
97
108
  if input.Options.Mode == "types" {
98
109
  typed = resolveTypes(fset, trees, texts, input.Options.ModulePath, input.Options.GOOS, input.Options.GOARCH)
@@ -73,6 +73,8 @@ func resolveTypes(fset *token.FileSet, trees map[string]*ast.File, texts map[str
73
73
  context.GOOS = goos
74
74
  context.GOARCH = goarch
75
75
  context.CgoEnabled = false
76
+ // Snapshot paths are slash-separated regardless of the helper's host OS.
77
+ context.JoinPath = path.Join
76
78
  context.OpenFile = func(name string) (io.ReadCloser, error) {
77
79
  if text, ok := texts[path.Clean(name)]; ok {
78
80
  return io.NopCloser(strings.NewReader(text)), nil
@@ -4,20 +4,32 @@ Name-resolved calls/inheritance remain heuristic (shadowing/dynamic dispatch are
4
4
  not fully resolved); confidence is a provenance category, not a probability.
5
5
  """
6
6
  import ast
7
+ import re
7
8
  from collections import defaultdict
9
+ from .schema import SourceSyntaxError
10
+ from .python_calls import resolved_links, symbol_resolver
8
11
 
9
12
  def extract(sources, max_lines=65, options=None):
10
13
  if options:
11
14
  raise ValueError("Python adapter does not accept language options")
12
15
  units, trees, aliases, class_bases = [], {}, {}, {}
16
+ diagnostics = []
17
+ scoped_calls = {}
13
18
  for source in sources:
14
19
  path, text = source.path, source.text
15
20
  lines = text.splitlines()
16
21
  module = path.removesuffix('.py').replace('/', '.')
17
22
  if module.endswith('.__init__'):
18
23
  module = module[:-9]
19
- tree = ast.parse(text, filename=path)
24
+ try:
25
+ tree = ast.parse(text, filename=path)
26
+ except SyntaxError as error:
27
+ diagnostics.append({'path': path, 'language': 'python', 'errorType': type(error).__name__,
28
+ 'line': error.lineno or 1, 'column': error.offset or 1})
29
+ continue
20
30
  trees[module] = tree
31
+ package = module if path.endswith('/__init__.py') or path == '__init__.py' else module.rpartition('.')[0]
32
+ scoped_calls[module] = resolved_links(tree, module, package)
21
33
  imports = {}
22
34
  for n in tree.body:
23
35
  if isinstance(n, ast.Import):
@@ -74,9 +86,12 @@ def extract(sources, max_lines=65, options=None):
74
86
 
75
87
  definitions(tree.body)
76
88
 
89
+ if diagnostics:
90
+ raise SourceSyntaxError(diagnostics)
77
91
  symbols = defaultdict(list)
78
92
  for u in units:
79
93
  symbols[u['symbol']].append(u['id'])
94
+ resolve = symbol_resolver(set(trees), symbols)
80
95
  for u in units:
81
96
  targets, relations = [], []
82
97
 
@@ -86,18 +101,13 @@ def extract(sources, max_lines=65, options=None):
86
101
  'resolution': resolution} for target in ids if target != u['id'])
87
102
  if u['owner']:
88
103
  relate(symbols.get(u['owner'], []), 'member_of', 1.0, 'syntax')
89
- for call in u['calls']:
90
- pieces = call.split('.')
91
- first = pieces[0]
92
- target = None
93
- if first in ('self', 'cls') and u['owner']:
94
- target = u['owner'] + '.' + '.'.join(pieces[1:])
95
- elif first in aliases[u['module']]:
96
- target = aliases[u['module']][first] + ('.' + '.'.join(pieces[1:]) if len(pieces) > 1 else '')
97
- elif len(pieces) == 1:
98
- target = u['module'] + '.' + call
99
- if target:
100
- relate(symbols.get(target, []), 'calls', .7, 'static-name')
104
+ for kind, links in scoped_calls[u['module']].items():
105
+ for line, target in links:
106
+ if u['start'] <= line <= u['end']:
107
+ relate(resolve(target), kind, .7, 'scoped-import')
108
+ for reference in re.findall(r':(?:func|meth):`~?([A-Za-z_][\w.]*)`', u['text']):
109
+ targets_ = resolve(reference) or resolve(u['module'] + '.' + reference)
110
+ relate(targets_, 'references_value', .6, 'explicit-doc-reference')
101
111
  for base in class_bases.get(u['owner'] or u['symbol'], []):
102
112
  pieces = base.split('.')
103
113
  target = aliases[u['module']].get(pieces[0], u['module'] + '.' + pieces[0])