open-context-engine 0.1.3 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/README.zh-CN.md +1 -1
- package/docs/QUICKSTART.md +28 -4
- package/package.json +3 -2
- package/scripts/retrieval-server.py +65 -16
- package/src/client.mjs +6 -2
- package/src/config.mjs +1 -1
- package/src/eval/remote-models.mjs +1 -1
- package/src/mcp.mjs +5 -3
- package/src/retrieval/batched.py +1 -1
- package/src/retrieval/cascade.py +1 -1
- package/src/retrieval/engine.py +2 -2
- package/src/retrieval/entities.py +1 -1
- package/src/retrieval/evidence.py +312 -0
- package/src/retrieval/languages/__init__.py +33 -4
- package/src/retrieval/languages/go.py +3 -1
- package/src/retrieval/languages/go_ast.go +11 -0
- package/src/retrieval/languages/python.py +23 -13
- package/src/retrieval/languages/python_calls.py +188 -0
- package/src/retrieval/languages/schema.py +9 -2
- package/src/retrieval/languages/typescript.mjs +10 -4
- package/src/retrieval/languages/typescript.py +3 -0
- package/src/retrieval/live.py +11 -5
- package/src/retrieval/planning.py +15 -0
- package/src/retrieval/reranker.py +12 -6
- package/src/retrieval/shared_worker.py +115 -0
- package/src/retrieval/writer_lock.py +5 -1
- package/src/service.mjs +19 -3
- package/src/setup.mjs +2 -0
- package/src/shared-service.mjs +142 -0
- package/src/workspaces.mjs +7 -5
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""Evidence-oriented retrieval with per-question scoring and complete snippets.
|
|
2
|
+
|
|
3
|
+
The historical engines remain available for frozen evaluations. This engine
|
|
4
|
+
keeps independent recall channels, scores actual subquestions, and selects
|
|
5
|
+
implementation evidence before supporting tests when that is what was asked.
|
|
6
|
+
"""
|
|
7
|
+
from collections import defaultdict
|
|
8
|
+
import math
|
|
9
|
+
import re
|
|
10
|
+
import time
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
|
|
14
|
+
from engine import Engine, document, post
|
|
15
|
+
from reranker import rerank_pairs
|
|
16
|
+
|
|
17
|
+
VERSION = 'evidence-v2'
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def source_role(unit):
|
|
21
|
+
path = unit['path'].lower()
|
|
22
|
+
if re.search(r'(^|/)(tests?|__tests__|fixtures)(/|$)|(^|/)test_[^/]+|[._](test|spec)\.', path):
|
|
23
|
+
return 'test'
|
|
24
|
+
if unit.get('language') == 'text' or re.search(r'(^|/)(docs?|examples?)/|\.(md|rst|txt)$', path):
|
|
25
|
+
return 'documentation'
|
|
26
|
+
return 'implementation'
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def requested_role(query):
|
|
30
|
+
if re.search(r'^(?:find|show|locate|list)\s+(?:the\s+)?(?:(?:unit|regression)\s+)?tests?\b|^(?:查找|找到|列出|展示).{0,5}(?:测试|用例)', query, re.I):
|
|
31
|
+
return 'test'
|
|
32
|
+
if re.search(r'^(?:find|show|locate|list)\s+(?:the\s+)?(?:docs?|documentation|readme)\b|^(?:查找|找到|列出|展示).{0,5}文档', query, re.I):
|
|
33
|
+
return 'documentation'
|
|
34
|
+
return 'implementation'
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def requested_languages(query):
|
|
38
|
+
"""Honor explicit language scope without guessing from repository names."""
|
|
39
|
+
languages = set()
|
|
40
|
+
for pattern, values in [
|
|
41
|
+
(r'\bpython\b', {'python'}),
|
|
42
|
+
(r'\bgolang\b', {'go'}),
|
|
43
|
+
(r'\btypescript\b', {'typescript'}),
|
|
44
|
+
(r'\bjavascript\b', {'javascript', 'typescript'}),
|
|
45
|
+
(r'\b(?:frontend|front-end|react)\b|前端', {'javascript', 'typescript'}),
|
|
46
|
+
]:
|
|
47
|
+
if re.search(pattern, query, re.I):
|
|
48
|
+
languages.update(values)
|
|
49
|
+
if re.search(r'\bGo\b|\bgo\s+(?:code|service|backend|worker|implementation)\b', query):
|
|
50
|
+
languages.add('go')
|
|
51
|
+
return languages
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class EvidenceEngine(Engine):
|
|
55
|
+
version = VERSION
|
|
56
|
+
|
|
57
|
+
def __init__(self, *args, **kwargs):
|
|
58
|
+
super().__init__(*args, **kwargs)
|
|
59
|
+
self.roles = [source_role(u) for u in self.units]
|
|
60
|
+
self.symbols = defaultdict(list)
|
|
61
|
+
self.neighbors = [set() for _ in self.units]
|
|
62
|
+
self.callees = [set() for _ in self.units]
|
|
63
|
+
for u in self.units:
|
|
64
|
+
if u.get('name'):
|
|
65
|
+
self.symbols[u['symbol']].append(u['id'])
|
|
66
|
+
for relation in u.get('relations', []):
|
|
67
|
+
if relation['kind'] not in {'calls', 'same_symbol', 'references_value'}:
|
|
68
|
+
continue
|
|
69
|
+
target = relation['target']
|
|
70
|
+
self.neighbors[u['id']].add(target)
|
|
71
|
+
self.neighbors[target].add(u['id'])
|
|
72
|
+
if relation['kind'] in {'calls', 'references_value'}:
|
|
73
|
+
self.callees[u['id']].add(target)
|
|
74
|
+
|
|
75
|
+
def substance(self, uid):
|
|
76
|
+
unit = self.units[uid]
|
|
77
|
+
if self.roles[uid] != 'implementation':
|
|
78
|
+
return 1.0
|
|
79
|
+
if unit['kind'] in {'property', 'parameter', 'type', 'interface', 'type-alias'}:
|
|
80
|
+
return .25
|
|
81
|
+
lines = [line.strip() for line in unit['text'].splitlines() if line.strip()]
|
|
82
|
+
executable = [line for line in lines if not line.startswith(
|
|
83
|
+
('//', '/*', '*', '#', 'import ', 'from ', 'export type ', '}'))]
|
|
84
|
+
if not executable:
|
|
85
|
+
return .1
|
|
86
|
+
if unit['kind'] in {'function', 'method'}:
|
|
87
|
+
return 1.0 if len(executable) >= 2 else .5
|
|
88
|
+
return 1.0 if len(executable) >= 5 else .25
|
|
89
|
+
|
|
90
|
+
def bundle(self, uid, budget):
|
|
91
|
+
unit = self.units[uid]
|
|
92
|
+
family = self.symbols.get(unit['symbol'], [uid]) if unit.get('name') else [uid]
|
|
93
|
+
if unit['kind'] not in {'function', 'method'}:
|
|
94
|
+
return [uid]
|
|
95
|
+
if sum(self.costs[i] for i in family) <= 3200:
|
|
96
|
+
return sorted(family, key=lambda i: self.units[i]['start'])
|
|
97
|
+
return [uid]
|
|
98
|
+
|
|
99
|
+
def render_selection(self, selected):
|
|
100
|
+
# Coalesce only returned source; never invent lines between snippets.
|
|
101
|
+
paths = {}
|
|
102
|
+
for uid in selected:
|
|
103
|
+
unit = self.units[uid]
|
|
104
|
+
lines = paths.setdefault(unit['path'], {})
|
|
105
|
+
for number, line in enumerate(unit['text'].split('\n'), unit['start']):
|
|
106
|
+
lines[number] = line
|
|
107
|
+
blocks = []
|
|
108
|
+
for path, lines in paths.items():
|
|
109
|
+
previous = None
|
|
110
|
+
for number in sorted(lines):
|
|
111
|
+
if previous is None or number != previous + 1:
|
|
112
|
+
blocks.append('Path: ' + path)
|
|
113
|
+
blocks.append(f'{number}\t{lines[number]}')
|
|
114
|
+
previous = number
|
|
115
|
+
blocks.append('')
|
|
116
|
+
return '\n'.join(blocks)
|
|
117
|
+
|
|
118
|
+
def search(self, plan, budget=8000):
|
|
119
|
+
started = time.monotonic()
|
|
120
|
+
declarations = bool(re.search(r'\b(interface|type alias|schema)\b|类型定义|接口类型', plan['intent'], re.I))
|
|
121
|
+
def substance(uid):
|
|
122
|
+
if declarations and self.units[uid]['kind'] in {'type', 'interface', 'type-alias'}:
|
|
123
|
+
return 1.0
|
|
124
|
+
return self.substance(uid)
|
|
125
|
+
intent = plan['intent']
|
|
126
|
+
languages = requested_languages(intent)
|
|
127
|
+
scoped = np.asarray([not languages or u.get('language') in languages for u in self.units])
|
|
128
|
+
facets = list(dict.fromkeys(f['question'] for f in plan['facets'] if f['question'] != intent))[:4]
|
|
129
|
+
queries = [intent] + [facet + '\nContext for this part of the request: ' + intent for facet in facets]
|
|
130
|
+
result = post(self.embed_url + '/embeddings', {'model': self.embedding_model,
|
|
131
|
+
'input': ['Instruct: Retrieve source code implementing the requested behavior.\nQuery: ' + q for q in queries]}, self.embedding_key)
|
|
132
|
+
rows = sorted(result['data'], key=lambda row: row['index'])
|
|
133
|
+
if [row['index'] for row in rows] != list(range(len(queries))):
|
|
134
|
+
raise ValueError('Invalid query embedding indices')
|
|
135
|
+
matrix = np.asarray([row['embedding'] for row in rows], dtype=np.float32)
|
|
136
|
+
if matrix.shape != (len(queries), self.vectors.shape[1]) or not np.isfinite(matrix).all():
|
|
137
|
+
raise ValueError('Invalid query embedding vectors')
|
|
138
|
+
dense = self.vectors @ matrix.T
|
|
139
|
+
candidates, recall = set(), []
|
|
140
|
+
for col, query in enumerate(queries):
|
|
141
|
+
channels = []
|
|
142
|
+
for scores in (dense[:, col], self.lexical(query)):
|
|
143
|
+
order = np.argsort(-scores, kind='stable')
|
|
144
|
+
# Candidate depth is independent of the response token budget.
|
|
145
|
+
# Keep global recall and add a separate explicit-language lane.
|
|
146
|
+
ids = [int(uid) for uid in order[:40] if scores[uid] > 0]
|
|
147
|
+
if languages:
|
|
148
|
+
local = [int(uid) for uid in order if scoped[uid] and scores[uid] > 0][:80]
|
|
149
|
+
ids = list(dict.fromkeys(ids + local))
|
|
150
|
+
candidates.update(ids)
|
|
151
|
+
channels.append(ids)
|
|
152
|
+
recall.append(channels)
|
|
153
|
+
scores, waves = {}, []
|
|
154
|
+
|
|
155
|
+
def score(ids):
|
|
156
|
+
ids = sorted(ids)
|
|
157
|
+
pairs = [(q, j) for q in range(len(queries)) for j, uid in enumerate(ids) if (q, uid) not in scores]
|
|
158
|
+
if not pairs:
|
|
159
|
+
return
|
|
160
|
+
before = time.monotonic()
|
|
161
|
+
result = rerank_pairs(self.reranker, queries, [document(self.units[i], 5000) for i in ids], pairs, post)
|
|
162
|
+
for row in result['results']:
|
|
163
|
+
q, j = pairs[row['index']]
|
|
164
|
+
scores[q, ids[j]] = row['relevance_score']
|
|
165
|
+
waves.append({'pairs': len(pairs), 'requests': result['meta']['request_count'],
|
|
166
|
+
'elapsedMs': round((time.monotonic() - before) * 1000)})
|
|
167
|
+
|
|
168
|
+
score(candidates)
|
|
169
|
+
primary = requested_role(plan['intent'])
|
|
170
|
+
seeds = set()
|
|
171
|
+
for q in range(len(queries)):
|
|
172
|
+
order = sorted(candidates, key=lambda uid: (-scores[q, uid], uid))
|
|
173
|
+
seeds.update(order[:8])
|
|
174
|
+
seeds.update([i for i in order if self.roles[i] == primary and substance(i) >= .5][:6])
|
|
175
|
+
support = defaultdict(float)
|
|
176
|
+
for uid in sorted(seeds):
|
|
177
|
+
outgoing = self.callees[uid]
|
|
178
|
+
neighbors = outgoing | self.neighbors[uid]
|
|
179
|
+
# Shared utility callers are broad hubs, not reliable task evidence.
|
|
180
|
+
if len(neighbors) > 24:
|
|
181
|
+
neighbors = outgoing
|
|
182
|
+
for neighbor in neighbors:
|
|
183
|
+
if neighbor not in candidates:
|
|
184
|
+
confidence = max(scores[q, uid] for q in range(len(queries)))
|
|
185
|
+
support[neighbor] = max(support[neighbor], confidence / max(1, len(neighbors)))
|
|
186
|
+
expanded = sorted(support, key=lambda uid: (-support[uid], uid))[:80]
|
|
187
|
+
candidates.update(expanded)
|
|
188
|
+
score(expanded)
|
|
189
|
+
|
|
190
|
+
completions = {i for uid in candidates for i in self.bundle(uid, budget)} - candidates
|
|
191
|
+
score(completions)
|
|
192
|
+
candidates.update(completions)
|
|
193
|
+
|
|
194
|
+
ranks = {}
|
|
195
|
+
for q in range(len(queries)):
|
|
196
|
+
for role in ('implementation', 'test', 'documentation'):
|
|
197
|
+
order = sorted((i for i in candidates if self.roles[i] == role),
|
|
198
|
+
key=lambda uid: (-scores[q, uid], uid))
|
|
199
|
+
ranks.update({(q, uid): rank for rank, uid in enumerate(order)})
|
|
200
|
+
|
|
201
|
+
def value(q, uid):
|
|
202
|
+
relevance = scores[q, uid]
|
|
203
|
+
if relevance < .1:
|
|
204
|
+
return 0.0
|
|
205
|
+
# Rank separates saturated probability-like scores. No inverse cost
|
|
206
|
+
# reward: a short test must not displace a better-ranked body.
|
|
207
|
+
return (.5 * relevance + .5 * math.exp(-ranks[q, uid] / 8)) * substance(uid) * max(0.0, scores[0, uid]) * (1.0 if scoped[uid] else .25)
|
|
208
|
+
|
|
209
|
+
# Tests often describe behavior more explicitly than the implementation.
|
|
210
|
+
# Propagate relevance only along resolved source references, bounded by
|
|
211
|
+
# the test's fan-out. This is evidence support, not invented call edges.
|
|
212
|
+
corroboration = defaultdict(float)
|
|
213
|
+
for uid in candidates:
|
|
214
|
+
if self.roles[uid] != 'test':
|
|
215
|
+
continue
|
|
216
|
+
targets = [i for i in self.callees[uid] if i in candidates and self.roles[i] == 'implementation']
|
|
217
|
+
if not targets or len(targets) > 8:
|
|
218
|
+
continue
|
|
219
|
+
for q in range(len(queries)):
|
|
220
|
+
if scores[q, uid] < .1:
|
|
221
|
+
continue
|
|
222
|
+
strength = (.5 * scores[q, uid] + .5 * math.exp(-ranks[q, uid] / 8)) / math.sqrt(len(targets))
|
|
223
|
+
for target in targets:
|
|
224
|
+
corroboration[q, target] = max(corroboration[q, target], strength)
|
|
225
|
+
selected, selected_set, trace = [], set(), []
|
|
226
|
+
spent = 0
|
|
227
|
+
path_counts, covered = defaultdict(int), np.zeros(len(queries))
|
|
228
|
+
symbol_counts = defaultdict(int)
|
|
229
|
+
|
|
230
|
+
def choose(ids, q=None, phase='fill', limit=None):
|
|
231
|
+
nonlocal spent
|
|
232
|
+
choices = []
|
|
233
|
+
for uid in ids:
|
|
234
|
+
if uid in selected_set:
|
|
235
|
+
continue
|
|
236
|
+
bundle = [uid]
|
|
237
|
+
if self.roles[uid] == primary == 'implementation':
|
|
238
|
+
# Relevant local helpers are part of the entry point explanation.
|
|
239
|
+
# Keep their source evidence with the calling implementation.
|
|
240
|
+
references = {r['target'] for r in self.units[uid].get('relations', [])
|
|
241
|
+
if r.get('resolution') == 'explicit-doc-reference'}
|
|
242
|
+
# Resolved local callees explain the selected entry point.
|
|
243
|
+
# Bound fan-out, cost and relevance so utility hubs cannot
|
|
244
|
+
# pull arbitrary dependencies into the answer.
|
|
245
|
+
local_calls = {i for i in self.callees[uid]
|
|
246
|
+
if self.units[i]['path'] == self.units[uid]['path']}
|
|
247
|
+
if len({self.units[i]['symbol'] for i in local_calls}) <= 8:
|
|
248
|
+
references.update(local_calls)
|
|
249
|
+
extras = [i for i in references if i in candidates and i not in selected_set and i != uid
|
|
250
|
+
and self.roles[i] == 'implementation' and substance(i) >= .5
|
|
251
|
+
and self.costs[i] <= 1000 and scores[0, i] >= .5]
|
|
252
|
+
extras.sort(key=lambda i: (-scores[0, i], i))
|
|
253
|
+
extra_cost = 0
|
|
254
|
+
for i in extras[:3]:
|
|
255
|
+
if extra_cost + self.costs[i] <= 1600:
|
|
256
|
+
bundle.append(i)
|
|
257
|
+
extra_cost += self.costs[i]
|
|
258
|
+
cost = sum(self.costs[i] for i in bundle)
|
|
259
|
+
if spent + cost > (budget if limit is None else limit):
|
|
260
|
+
bundle, cost = [uid], self.costs[uid]
|
|
261
|
+
if spent + cost > (budget if limit is None else limit):
|
|
262
|
+
continue
|
|
263
|
+
values = [value(col, uid) + .3 * corroboration[col, uid] for col in range(len(queries))]
|
|
264
|
+
merit = values[q] if q is not None else max(v / (1 + covered[col]) for col, v in enumerate(values))
|
|
265
|
+
merit /= 1 + .1 * path_counts[self.units[uid]['path']] + .8 * symbol_counts[self.units[uid]['symbol']]
|
|
266
|
+
if merit > 0:
|
|
267
|
+
choices.append((merit, -uid, uid, bundle, cost, values))
|
|
268
|
+
if not choices:
|
|
269
|
+
return False
|
|
270
|
+
_, _, uid, bundle, cost, values = max(choices)
|
|
271
|
+
selected.extend(bundle); selected_set.update(bundle); spent += cost
|
|
272
|
+
covered[:] += values
|
|
273
|
+
path_counts[self.units[uid]['path']] += 1
|
|
274
|
+
symbol_counts[self.units[uid]['symbol']] += 1
|
|
275
|
+
trace.append({'id': uid, 'bundle': bundle, 'phase': phase, 'role': self.roles[uid], 'tokens': cost,
|
|
276
|
+
'scores': [scores[q, uid] for q in range(len(queries))]})
|
|
277
|
+
return True
|
|
278
|
+
|
|
279
|
+
primary_ids = [i for i in candidates if self.roles[i] == primary and substance(i) >= .5]
|
|
280
|
+
facets = list(range(1, len(queries))) or [0]
|
|
281
|
+
# Build a stable 8k core, then add evidence. Candidate scoring and family
|
|
282
|
+
# completion do not depend on the output budget.
|
|
283
|
+
for stage in range(8000, max(8000, budget) + 8000, 8000):
|
|
284
|
+
ceiling = min(stage, budget)
|
|
285
|
+
for _ in range(3):
|
|
286
|
+
for q in facets:
|
|
287
|
+
choose(primary_ids, q, 'primary', ceiling * .82)
|
|
288
|
+
for _ in range(2):
|
|
289
|
+
companions = {i for uid in selected for i in self.callees[uid]
|
|
290
|
+
if i in candidates and self.roles[i] == primary and substance(i) >= .5}
|
|
291
|
+
choose(companions, phase='dependency', limit=ceiling * .82)
|
|
292
|
+
continuations = {i for uid in selected for i in self.bundle(uid, budget)}
|
|
293
|
+
while choose(continuations, phase='continuation', limit=ceiling * .85):
|
|
294
|
+
pass
|
|
295
|
+
for q in facets:
|
|
296
|
+
choose(primary_ids, q, 'primary', ceiling * .85)
|
|
297
|
+
if primary == 'implementation' and re.search(r'\btests?\b|测试|回归', intent, re.I):
|
|
298
|
+
tests = [i for i in candidates if self.roles[i] == 'test']
|
|
299
|
+
related = [i for i in tests if self.callees[i] & selected_set]
|
|
300
|
+
for q in facets:
|
|
301
|
+
choose(related or tests, q, 'support', ceiling)
|
|
302
|
+
while choose(candidates, limit=ceiling):
|
|
303
|
+
pass
|
|
304
|
+
context = self.render_selection(selected)
|
|
305
|
+
tokens = len(self.encoding.encode_ordinary(context))
|
|
306
|
+
if tokens > budget:
|
|
307
|
+
raise ValueError('Evidence packing exceeded token budget')
|
|
308
|
+
return context, {'version': VERSION, 'elapsedMs': round((time.monotonic() - started) * 1000),
|
|
309
|
+
'tokens': tokens, 'queryCache': False, 'candidateCount': len(candidates),
|
|
310
|
+
'rerankedCount': len(candidates), 'expandedCount': len(expanded), 'queries': queries,
|
|
311
|
+
'modelRequests': {'embedding': 1, 'rerank': sum(w['requests'] for w in waves)},
|
|
312
|
+
'waves': waves, 'selected': trace, 'recall': recall}
|
|
@@ -8,7 +8,7 @@ import sys
|
|
|
8
8
|
|
|
9
9
|
from . import python, typescript, text, go
|
|
10
10
|
from .files import path_exclusion, read_text
|
|
11
|
-
from .schema import SCHEMA_VERSION, SourceFile, validate_units
|
|
11
|
+
from .schema import SCHEMA_VERSION, SourceFile, SourceSyntaxError, validate_units
|
|
12
12
|
|
|
13
13
|
ADAPTERS = {'python': python, 'typescript': typescript, 'javascript': typescript, 'go': go, 'text': text}
|
|
14
14
|
EXTENSIONS = {'.py': 'python', '.ts': 'typescript', '.tsx': 'typescript',
|
|
@@ -25,9 +25,12 @@ def adapter_manifest(files, language_options=None):
|
|
|
25
25
|
options = language_options or {}
|
|
26
26
|
if not isinstance(options, dict) or set(options) - set(ADAPTERS):
|
|
27
27
|
raise ValueError('Unknown language options')
|
|
28
|
-
paths = [Path(__file__), Path(__file__).with_name('schema.py'), Path(__file__).with_name('files.py')
|
|
28
|
+
paths = [Path(__file__), Path(__file__).with_name('schema.py'), Path(__file__).with_name('files.py'),
|
|
29
|
+
Path(text.__file__)]
|
|
29
30
|
for language in languages:
|
|
30
31
|
paths.append(Path(ADAPTERS[language].__file__))
|
|
32
|
+
if language == 'python':
|
|
33
|
+
paths.append(Path(python.__file__).with_name('python_calls.py'))
|
|
31
34
|
if language in {'typescript', 'javascript'}:
|
|
32
35
|
paths.append(Path(typescript.__file__).with_suffix('.mjs'))
|
|
33
36
|
if language == 'go':
|
|
@@ -42,6 +45,29 @@ def adapter_manifest(files, language_options=None):
|
|
|
42
45
|
'sourceSha256': {path.name: hashlib.sha256(path.read_bytes()).hexdigest() for path in paths}}
|
|
43
46
|
|
|
44
47
|
|
|
48
|
+
def extract_with_fallback(adapter, sources, max_lines, settings):
|
|
49
|
+
try:
|
|
50
|
+
return adapter.extract(sources, max_lines, settings)
|
|
51
|
+
except SourceSyntaxError as error:
|
|
52
|
+
diagnostics = {item['path']: item for item in error.diagnostics}
|
|
53
|
+
if not diagnostics or not set(diagnostics) <= {source.path for source in sources}:
|
|
54
|
+
raise ValueError('Invalid syntax diagnostic paths') from error
|
|
55
|
+
# Rebuild relations using only valid source. Never preserve partial AST
|
|
56
|
+
# output or let a broken module supply symbols to its healthy callers.
|
|
57
|
+
valid = [source for source in sources if source.path not in diagnostics]
|
|
58
|
+
units = adapter.extract(valid, max_lines, settings) if valid else []
|
|
59
|
+
for source in sources:
|
|
60
|
+
if source.path not in diagnostics:
|
|
61
|
+
continue
|
|
62
|
+
fallback = text.extract([source], max_lines)
|
|
63
|
+
for unit in fallback:
|
|
64
|
+
unit['id'] += len(units)
|
|
65
|
+
if fallback:
|
|
66
|
+
fallback[0]['parseDiagnostic'] = {**diagnostics[source.path], 'fallback': 'text'}
|
|
67
|
+
units.extend(fallback)
|
|
68
|
+
return units
|
|
69
|
+
|
|
70
|
+
|
|
45
71
|
def source_units(root, files, max_lines=65, language_options=None, report=None, cache=None):
|
|
46
72
|
if type(max_lines) is not int or max_lines < 1:
|
|
47
73
|
raise ValueError('max_lines must be a positive integer')
|
|
@@ -98,7 +124,7 @@ def source_units(root, files, max_lines=65, language_options=None, report=None,
|
|
|
98
124
|
], sort_keys=True).encode()).hexdigest() if cache is not None else None)
|
|
99
125
|
previous = cache.get(key) if cache is not None else None
|
|
100
126
|
extracted = (deepcopy(previous[1]) if previous and previous[0] == fingerprint
|
|
101
|
-
else ADAPTERS[language]
|
|
127
|
+
else extract_with_fallback(ADAPTERS[language], batch, max_lines, settings))
|
|
102
128
|
if cache is not None:
|
|
103
129
|
pending_cache[key] = (fingerprint, deepcopy(extracted))
|
|
104
130
|
offset = len(units)
|
|
@@ -124,6 +150,9 @@ def source_units(root, files, max_lines=65, language_options=None, report=None,
|
|
|
124
150
|
cache.clear()
|
|
125
151
|
cache.update(pending_cache)
|
|
126
152
|
if report is not None:
|
|
153
|
+
diagnostics = [unit['parseDiagnostic'] for unit in units if 'parseDiagnostic' in unit]
|
|
127
154
|
report.update(inputFiles=len(files), acceptedFiles=len(sources), excluded=excluded,
|
|
128
|
-
fallbackFiles=
|
|
155
|
+
fallbackFiles=len({source.path for source in sources if language_for(source.path) == 'text'}
|
|
156
|
+
| {item['path'] for item in diagnostics}),
|
|
157
|
+
degradedFiles=len(diagnostics), parseDiagnostics=diagnostics)
|
|
129
158
|
return units
|
|
@@ -11,7 +11,7 @@ import subprocess
|
|
|
11
11
|
import sys
|
|
12
12
|
import tempfile
|
|
13
13
|
|
|
14
|
-
from .schema import physical_lines
|
|
14
|
+
from .schema import physical_lines, SourceSyntaxError
|
|
15
15
|
|
|
16
16
|
|
|
17
17
|
@lru_cache(maxsize=8)
|
|
@@ -81,6 +81,8 @@ def parse(sources, options):
|
|
|
81
81
|
def extract(sources, max_lines=65, options=None):
|
|
82
82
|
options = settings(options)
|
|
83
83
|
parsed = parse(sources, options)
|
|
84
|
+
if parsed.get('syntaxErrors'):
|
|
85
|
+
raise SourceSyntaxError(parsed['syntaxErrors'])
|
|
84
86
|
units, records, targets = [], {}, defaultdict(list)
|
|
85
87
|
source_by_path = {source.path: source for source in sources}
|
|
86
88
|
for file in parsed['files']:
|
|
@@ -6,6 +6,7 @@ import (
|
|
|
6
6
|
"fmt"
|
|
7
7
|
"go/ast"
|
|
8
8
|
"go/parser"
|
|
9
|
+
"go/scanner"
|
|
9
10
|
"go/token"
|
|
10
11
|
"os"
|
|
11
12
|
"path"
|
|
@@ -74,10 +75,17 @@ func run() error {
|
|
|
74
75
|
objects := map[*ast.Object]Target{}
|
|
75
76
|
functions := map[string][]Target{}
|
|
76
77
|
line := func(p token.Pos) int { return fset.PositionFor(p, false).Line }
|
|
78
|
+
diagnostics := []map[string]any{}
|
|
77
79
|
for _, src := range input.Files {
|
|
78
80
|
text := strings.ReplaceAll(strings.ReplaceAll(src.Text, "\r\n", "\n"), "\r", "\n")
|
|
79
81
|
tree, err := parser.ParseFile(fset, src.Path, text, parser.ParseComments|parser.AllErrors)
|
|
80
82
|
if err != nil {
|
|
83
|
+
if syntax, ok := err.(scanner.ErrorList); ok && len(syntax) > 0 {
|
|
84
|
+
pos := syntax[0].Pos
|
|
85
|
+
diagnostics = append(diagnostics, map[string]any{"path": src.Path, "language": "go",
|
|
86
|
+
"errorType": "SyntaxError", "line": pos.Line, "column": pos.Column})
|
|
87
|
+
continue
|
|
88
|
+
}
|
|
81
89
|
return err
|
|
82
90
|
}
|
|
83
91
|
trees[src.Path] = tree
|
|
@@ -93,6 +101,9 @@ func run() error {
|
|
|
93
101
|
}
|
|
94
102
|
}
|
|
95
103
|
}
|
|
104
|
+
if len(diagnostics) > 0 {
|
|
105
|
+
return json.NewEncoder(os.Stdout).Encode(map[string]any{"compilerVersion": runtime.Version(), "syntaxErrors": diagnostics})
|
|
106
|
+
}
|
|
96
107
|
var typed TypeEvidence
|
|
97
108
|
if input.Options.Mode == "types" {
|
|
98
109
|
typed = resolveTypes(fset, trees, texts, input.Options.ModulePath, input.Options.GOOS, input.Options.GOARCH)
|
|
@@ -4,20 +4,32 @@ Name-resolved calls/inheritance remain heuristic (shadowing/dynamic dispatch are
|
|
|
4
4
|
not fully resolved); confidence is a provenance category, not a probability.
|
|
5
5
|
"""
|
|
6
6
|
import ast
|
|
7
|
+
import re
|
|
7
8
|
from collections import defaultdict
|
|
9
|
+
from .schema import SourceSyntaxError
|
|
10
|
+
from .python_calls import resolved_links, symbol_resolver
|
|
8
11
|
|
|
9
12
|
def extract(sources, max_lines=65, options=None):
|
|
10
13
|
if options:
|
|
11
14
|
raise ValueError("Python adapter does not accept language options")
|
|
12
15
|
units, trees, aliases, class_bases = [], {}, {}, {}
|
|
16
|
+
diagnostics = []
|
|
17
|
+
scoped_calls = {}
|
|
13
18
|
for source in sources:
|
|
14
19
|
path, text = source.path, source.text
|
|
15
20
|
lines = text.splitlines()
|
|
16
21
|
module = path.removesuffix('.py').replace('/', '.')
|
|
17
22
|
if module.endswith('.__init__'):
|
|
18
23
|
module = module[:-9]
|
|
19
|
-
|
|
24
|
+
try:
|
|
25
|
+
tree = ast.parse(text, filename=path)
|
|
26
|
+
except SyntaxError as error:
|
|
27
|
+
diagnostics.append({'path': path, 'language': 'python', 'errorType': type(error).__name__,
|
|
28
|
+
'line': error.lineno or 1, 'column': error.offset or 1})
|
|
29
|
+
continue
|
|
20
30
|
trees[module] = tree
|
|
31
|
+
package = module if path.endswith('/__init__.py') or path == '__init__.py' else module.rpartition('.')[0]
|
|
32
|
+
scoped_calls[module] = resolved_links(tree, module, package)
|
|
21
33
|
imports = {}
|
|
22
34
|
for n in tree.body:
|
|
23
35
|
if isinstance(n, ast.Import):
|
|
@@ -74,9 +86,12 @@ def extract(sources, max_lines=65, options=None):
|
|
|
74
86
|
|
|
75
87
|
definitions(tree.body)
|
|
76
88
|
|
|
89
|
+
if diagnostics:
|
|
90
|
+
raise SourceSyntaxError(diagnostics)
|
|
77
91
|
symbols = defaultdict(list)
|
|
78
92
|
for u in units:
|
|
79
93
|
symbols[u['symbol']].append(u['id'])
|
|
94
|
+
resolve = symbol_resolver(set(trees), symbols)
|
|
80
95
|
for u in units:
|
|
81
96
|
targets, relations = [], []
|
|
82
97
|
|
|
@@ -86,18 +101,13 @@ def extract(sources, max_lines=65, options=None):
|
|
|
86
101
|
'resolution': resolution} for target in ids if target != u['id'])
|
|
87
102
|
if u['owner']:
|
|
88
103
|
relate(symbols.get(u['owner'], []), 'member_of', 1.0, 'syntax')
|
|
89
|
-
for
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
target = aliases[u['module']][first] + ('.' + '.'.join(pieces[1:]) if len(pieces) > 1 else '')
|
|
97
|
-
elif len(pieces) == 1:
|
|
98
|
-
target = u['module'] + '.' + call
|
|
99
|
-
if target:
|
|
100
|
-
relate(symbols.get(target, []), 'calls', .7, 'static-name')
|
|
104
|
+
for kind, links in scoped_calls[u['module']].items():
|
|
105
|
+
for line, target in links:
|
|
106
|
+
if u['start'] <= line <= u['end']:
|
|
107
|
+
relate(resolve(target), kind, .7, 'scoped-import')
|
|
108
|
+
for reference in re.findall(r':(?:func|meth):`~?([A-Za-z_][\w.]*)`', u['text']):
|
|
109
|
+
targets_ = resolve(reference) or resolve(u['module'] + '.' + reference)
|
|
110
|
+
relate(targets_, 'references_value', .6, 'explicit-doc-reference')
|
|
101
111
|
for base in class_bases.get(u['owner'] or u['symbol'], []):
|
|
102
112
|
pieces = base.split('.')
|
|
103
113
|
target = aliases[u['module']].get(pieces[0], u['module'] + '.' + pieces[0])
|