cseq 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cseq/__init__.py +9 -0
- cseq/acceptance.py +162 -0
- cseq/api.py +328 -0
- cseq/binary.py +122 -0
- cseq/cache.py +420 -0
- cseq/cli.py +895 -0
- cseq/compile_db.py +98 -0
- cseq/config.py +138 -0
- cseq/cpu_rules.py +48 -0
- cseq/dependencies.py +202 -0
- cseq/docs.py +21 -0
- cseq/dwarf.py +115 -0
- cseq/event_store.py +378 -0
- cseq/explain.py +113 -0
- cseq/fast_parser.py +286 -0
- cseq/html.py +485 -0
- cseq/html_bundle.py +464 -0
- cseq/hybrid_parser.py +120 -0
- cseq/linker.py +117 -0
- cseq/marker.py +294 -0
- cseq/model.py +367 -0
- cseq/parser.py +797 -0
- cseq/parser_contract_cases.json +69 -0
- cseq/parser_dependency_lock.json +59 -0
- cseq/parser_environment.py +186 -0
- cseq/parser_migration.py +126 -0
- cseq/plugin.py +216 -0
- cseq/project.py +305 -0
- cseq/query.py +62 -0
- cseq/resources/README.md +171 -0
- cseq/resources/design.md +6197 -0
- cseq/resources/verification_report.html +26 -0
- cseq/runtime.py +177 -0
- cseq/runtime_address.py +79 -0
- cseq/runtime_cpu.py +87 -0
- cseq/scanner.py +54 -0
- cseq/sequence.py +281 -0
- cseq/server.py +104 -0
- cseq/source_index.py +227 -0
- cseq/static_store.py +280 -0
- cseq/trace_analysis.py +336 -0
- cseq/trace_diff.py +56 -0
- cseq/tree_sitter_parser.py +468 -0
- cseq/valueflow.py +179 -0
- cseq-0.0.1.dist-info/METADATA +181 -0
- cseq-0.0.1.dist-info/RECORD +49 -0
- cseq-0.0.1.dist-info/WHEEL +5 -0
- cseq-0.0.1.dist-info/entry_points.txt +2 -0
- cseq-0.0.1.dist-info/top_level.txt +1 -0
cseq/sequence.py
ADDED
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field, replace
|
|
4
|
+
|
|
5
|
+
from .model import CallSiteRecord, CallTargetKind, ProjectIndex
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True, slots=True)
|
|
9
|
+
class SequenceMessage:
|
|
10
|
+
caller: str
|
|
11
|
+
callee: str
|
|
12
|
+
label: str
|
|
13
|
+
external_as_self: bool = False
|
|
14
|
+
control_context: tuple[str, ...] = ()
|
|
15
|
+
recursive: bool = False
|
|
16
|
+
guard_expression: str | None = None
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True, slots=True)
|
|
20
|
+
class SequenceRegion:
|
|
21
|
+
context: tuple[str, ...]
|
|
22
|
+
start_message: int
|
|
23
|
+
end_message: int
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def kind(self) -> str:
|
|
27
|
+
return self.context[-1] if self.context else "ROOT"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True, slots=True)
|
|
31
|
+
class FoldSummary:
|
|
32
|
+
start_message: int
|
|
33
|
+
end_message: int
|
|
34
|
+
hidden_count: int
|
|
35
|
+
reason: str = "budget"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True, slots=True)
|
|
39
|
+
class FoldPlan:
|
|
40
|
+
visible_indices: tuple[int, ...]
|
|
41
|
+
collapsed: tuple[FoldSummary, ...]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(slots=True)
|
|
45
|
+
class SequenceModel:
|
|
46
|
+
entry: str
|
|
47
|
+
messages: list[SequenceMessage] = field(default_factory=list)
|
|
48
|
+
regions: list[SequenceRegion] = field(default_factory=list)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def resolve_direct_calls(index: ProjectIndex) -> None:
|
|
52
|
+
# Demand-driven symbol indexing: direct-call resolution only needs function
|
|
53
|
+
# names that actually occur at call sites. Building a dictionary for every
|
|
54
|
+
# function is needlessly expensive on multi-million-LOC projects with many
|
|
55
|
+
# uncalled/static functions.
|
|
56
|
+
needed_names = {c.callee_name for c in index.callsites if c.callee_name}
|
|
57
|
+
functions_by_name: dict[str, list] = {}
|
|
58
|
+
if needed_names:
|
|
59
|
+
for f in index.functions_named(needed_names):
|
|
60
|
+
if f.name in needed_names:
|
|
61
|
+
functions_by_name.setdefault(f.name, []).append(f)
|
|
62
|
+
|
|
63
|
+
resolved: list[CallSiteRecord] = []
|
|
64
|
+
for c in index.callsites:
|
|
65
|
+
matches = functions_by_name.get(c.callee_name or "", [])
|
|
66
|
+
same_tu_static = [m for m in matches if m.storage_class == "static" and m.source_path == c.source_path]
|
|
67
|
+
if same_tu_static:
|
|
68
|
+
matches = same_tu_static
|
|
69
|
+
else:
|
|
70
|
+
matches = [m for m in matches if m.storage_class != "static"]
|
|
71
|
+
if len(matches) == 1:
|
|
72
|
+
target = matches[0]
|
|
73
|
+
resolved.append(replace(c, target_kind=CallTargetKind.INTERNAL_EXACT,
|
|
74
|
+
target_function_id=target.qualified_id,
|
|
75
|
+
target_function_ids=(target.qualified_id,)))
|
|
76
|
+
elif len(matches) > 1:
|
|
77
|
+
resolved.append(replace(c, target_kind=CallTargetKind.INTERNAL_POSSIBLE,
|
|
78
|
+
target_function_ids=tuple(sorted(m.qualified_id for m in matches)),
|
|
79
|
+
unknown_possible=False))
|
|
80
|
+
elif c.callee_name:
|
|
81
|
+
resolved.append(replace(c, target_kind=CallTargetKind.EXTERNAL_DECLARED))
|
|
82
|
+
else:
|
|
83
|
+
resolved.append(c)
|
|
84
|
+
index.callsites[:] = resolved
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def build_sequence(index: ProjectIndex, entry: str) -> SequenceModel:
|
|
88
|
+
by_caller: dict[str, list[CallSiteRecord]] = {}
|
|
89
|
+
for c in index.callsites:
|
|
90
|
+
by_caller.setdefault(c.caller_name, []).append(c)
|
|
91
|
+
model = SequenceModel(entry=entry)
|
|
92
|
+
active: list[str] = []
|
|
93
|
+
|
|
94
|
+
def append_message(msg: SequenceMessage) -> None:
|
|
95
|
+
model.messages.append(msg)
|
|
96
|
+
|
|
97
|
+
def visit(name: str, inherited_context: tuple[str, ...] = ()) -> None:
|
|
98
|
+
active.append(name)
|
|
99
|
+
try:
|
|
100
|
+
for call in by_caller.get(name, []):
|
|
101
|
+
context = inherited_context + tuple(call.control_context)
|
|
102
|
+
if call.guard_expression:
|
|
103
|
+
context = context + (f"GUARD:{call.guard_expression}",)
|
|
104
|
+
if call.target_kind == CallTargetKind.INTERNAL_EXACT and call.callee_name:
|
|
105
|
+
recursive = call.callee_name in active
|
|
106
|
+
append_message(SequenceMessage(name, call.callee_name, call.raw_text,
|
|
107
|
+
control_context=context, recursive=recursive, guard_expression=call.guard_expression))
|
|
108
|
+
if not recursive:
|
|
109
|
+
visit(call.callee_name, context)
|
|
110
|
+
elif call.target_kind == CallTargetKind.INDIRECT_EXACT and call.target_function_ids:
|
|
111
|
+
target_id = call.target_function_ids[0]
|
|
112
|
+
target = next(iter(index.functions_by_ids({target_id})), None)
|
|
113
|
+
target_name = target.name if target else "<indirect>"
|
|
114
|
+
recursive = target_name in active
|
|
115
|
+
append_message(SequenceMessage(name, target_name, call.raw_text,
|
|
116
|
+
control_context=context, recursive=recursive, guard_expression=call.guard_expression))
|
|
117
|
+
if target and not recursive:
|
|
118
|
+
visit(target.name, context)
|
|
119
|
+
elif call.target_kind == CallTargetKind.INDIRECT_POSSIBLE and call.target_function_ids:
|
|
120
|
+
names = []
|
|
121
|
+
by_id = {fn.qualified_id: fn for fn in index.functions_by_ids(set(call.target_function_ids))}
|
|
122
|
+
for tid in call.target_function_ids:
|
|
123
|
+
f = by_id.get(tid)
|
|
124
|
+
if f and f.name not in names:
|
|
125
|
+
names.append(f.name)
|
|
126
|
+
append_message(SequenceMessage(name, name, f"? {call.raw_text} -> {{{', '.join(names)}}}",
|
|
127
|
+
external_as_self=True, control_context=context, guard_expression=call.guard_expression))
|
|
128
|
+
elif call.target_kind == CallTargetKind.EXTERNAL_DECLARED:
|
|
129
|
+
append_message(SequenceMessage(name, name, call.raw_text, external_as_self=True,
|
|
130
|
+
control_context=context, guard_expression=call.guard_expression))
|
|
131
|
+
else:
|
|
132
|
+
append_message(SequenceMessage(name, name, f"? {call.raw_text}", external_as_self=True,
|
|
133
|
+
control_context=context, guard_expression=call.guard_expression))
|
|
134
|
+
finally:
|
|
135
|
+
active.pop()
|
|
136
|
+
|
|
137
|
+
visit(entry)
|
|
138
|
+
model.regions = derive_regions(model.messages)
|
|
139
|
+
return model
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def derive_regions(messages: list[SequenceMessage]) -> list[SequenceRegion]:
|
|
143
|
+
"""Derive contiguous control regions from message context paths.
|
|
144
|
+
|
|
145
|
+
Region information is presentation-neutral metadata. It records where a
|
|
146
|
+
context is active without expanding loops or recursion.
|
|
147
|
+
"""
|
|
148
|
+
open_at: dict[tuple[str, ...], int] = {}
|
|
149
|
+
regions: list[SequenceRegion] = []
|
|
150
|
+
previous: set[tuple[str, ...]] = set()
|
|
151
|
+
for i, msg in enumerate(messages):
|
|
152
|
+
current = {msg.control_context[:depth] for depth in range(1, len(msg.control_context) + 1)}
|
|
153
|
+
for ctx in sorted(previous - current, key=len, reverse=True):
|
|
154
|
+
start = open_at.pop(ctx, i)
|
|
155
|
+
regions.append(SequenceRegion(ctx, start, i - 1))
|
|
156
|
+
for ctx in sorted(current - previous, key=len):
|
|
157
|
+
open_at[ctx] = i
|
|
158
|
+
previous = current
|
|
159
|
+
end = len(messages) - 1
|
|
160
|
+
for ctx, start in sorted(open_at.items(), key=lambda x: len(x[0]), reverse=True):
|
|
161
|
+
regions.append(SequenceRegion(ctx, start, end))
|
|
162
|
+
return sorted(regions, key=lambda r: (r.start_message, len(r.context), r.end_message))
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def build_fold_plan(model: SequenceModel, max_messages: int = 200) -> FoldPlan:
|
|
166
|
+
"""Create a deterministic semantic fold plan without changing Sequence IR.
|
|
167
|
+
|
|
168
|
+
The budget is spent on semantically valuable events first. This avoids the
|
|
169
|
+
failure mode where a uniformly sampled trace hides the one unresolved call,
|
|
170
|
+
recursion edge, guard boundary, or CPU/control boundary the user actually
|
|
171
|
+
needs to investigate.
|
|
172
|
+
"""
|
|
173
|
+
total = len(model.messages)
|
|
174
|
+
budget = max(1, int(max_messages))
|
|
175
|
+
if total <= budget:
|
|
176
|
+
return FoldPlan(tuple(range(total)), ())
|
|
177
|
+
|
|
178
|
+
def score(i: int, msg: SequenceMessage) -> int:
|
|
179
|
+
value = 0
|
|
180
|
+
if i in {0, total - 1}:
|
|
181
|
+
value += 10_000
|
|
182
|
+
if msg.recursive:
|
|
183
|
+
value += 1_000
|
|
184
|
+
if msg.label.startswith('?'):
|
|
185
|
+
value += 900
|
|
186
|
+
if msg.external_as_self:
|
|
187
|
+
value += 700
|
|
188
|
+
if msg.guard_expression:
|
|
189
|
+
value += 650
|
|
190
|
+
contexts = tuple(x.upper() for x in msg.control_context)
|
|
191
|
+
if any(x.startswith('CASE:') or x == 'DEFAULT' for x in contexts):
|
|
192
|
+
value += 600
|
|
193
|
+
if any(x.startswith('SWITCH:') for x in contexts):
|
|
194
|
+
value += 550
|
|
195
|
+
if any(x.startswith('LOOP:') for x in contexts):
|
|
196
|
+
value += 500
|
|
197
|
+
# Context transitions are important entry/exit landmarks.
|
|
198
|
+
before = model.messages[i - 1].control_context if i > 0 else ()
|
|
199
|
+
after = model.messages[i + 1].control_context if i + 1 < total else ()
|
|
200
|
+
if msg.control_context != before:
|
|
201
|
+
value += 450
|
|
202
|
+
if msg.control_context != after:
|
|
203
|
+
value += 400
|
|
204
|
+
return value
|
|
205
|
+
|
|
206
|
+
ranked = sorted(range(total), key=lambda i: (-score(i, model.messages[i]), i))
|
|
207
|
+
selected = sorted(ranked[:budget])
|
|
208
|
+
|
|
209
|
+
# If spare budget remains after de-duplication (defensive), distribute it
|
|
210
|
+
# across the largest currently hidden spans rather than clustering at start.
|
|
211
|
+
selected_set = set(selected)
|
|
212
|
+
while len(selected_set) < budget:
|
|
213
|
+
hidden = [i for i in range(total) if i not in selected_set]
|
|
214
|
+
if not hidden:
|
|
215
|
+
break
|
|
216
|
+
# Pick the point farthest from any visible event.
|
|
217
|
+
pick = max(hidden, key=lambda i: (min(abs(i-j) for j in selected_set), -i))
|
|
218
|
+
selected_set.add(pick)
|
|
219
|
+
selected = sorted(selected_set)
|
|
220
|
+
|
|
221
|
+
collapsed: list[FoldSummary] = []
|
|
222
|
+
cursor = 0
|
|
223
|
+
visible = set(selected)
|
|
224
|
+
while cursor < total:
|
|
225
|
+
if cursor in visible:
|
|
226
|
+
cursor += 1
|
|
227
|
+
continue
|
|
228
|
+
start = cursor
|
|
229
|
+
while cursor < total and cursor not in visible:
|
|
230
|
+
cursor += 1
|
|
231
|
+
end = cursor - 1
|
|
232
|
+
hidden_msgs = model.messages[start:end + 1]
|
|
233
|
+
repeated = len({(m.caller, m.callee, m.label, m.control_context) for m in hidden_msgs}) == 1
|
|
234
|
+
one_region = len({m.control_context for m in hidden_msgs}) == 1
|
|
235
|
+
reason = 'repetition' if repeated else ('within-region-budget' if one_region else 'budget')
|
|
236
|
+
collapsed.append(FoldSummary(start, end, end - start + 1, reason))
|
|
237
|
+
return FoldPlan(tuple(selected), tuple(collapsed))
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def render_plantuml(model: SequenceModel) -> str:
|
|
241
|
+
participants: list[str] = []
|
|
242
|
+
for m in model.messages:
|
|
243
|
+
for name in (m.caller, m.callee):
|
|
244
|
+
if name not in participants:
|
|
245
|
+
participants.append(name)
|
|
246
|
+
lines = ["@startuml", "title cseq sequence"]
|
|
247
|
+
for name in participants:
|
|
248
|
+
lines.append(f'participant "{_escape(name)}" as {_ident(name)}')
|
|
249
|
+
lines.append("")
|
|
250
|
+
open_context: tuple[str, ...] = ()
|
|
251
|
+
for m in model.messages:
|
|
252
|
+
common = 0
|
|
253
|
+
while common < len(open_context) and common < len(m.control_context) and open_context[common] == m.control_context[common]:
|
|
254
|
+
common += 1
|
|
255
|
+
for _ in range(len(open_context) - common):
|
|
256
|
+
lines.append("end")
|
|
257
|
+
for ctx in m.control_context[common:]:
|
|
258
|
+
if ctx.lower().startswith("loop:"):
|
|
259
|
+
label = ctx.split(':', 1)[1]
|
|
260
|
+
if label.startswith("[infinite] "):
|
|
261
|
+
lines.append(f"loop [infinite] {_escape(label[len('[infinite] '):])}")
|
|
262
|
+
else:
|
|
263
|
+
lines.append(f"loop {_escape(label)}")
|
|
264
|
+
else:
|
|
265
|
+
lines.append(f"group {_escape(ctx)}")
|
|
266
|
+
open_context = m.control_context
|
|
267
|
+
suffix = " <<recursive>>" if m.recursive else ""
|
|
268
|
+
lines.append(f"{_ident(m.caller)} -> {_ident(m.callee)} : {_escape(m.label)}{suffix}")
|
|
269
|
+
for _ in open_context:
|
|
270
|
+
lines.append("end")
|
|
271
|
+
lines.append("@enduml")
|
|
272
|
+
return "\n".join(lines) + "\n"
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _ident(name: str) -> str:
|
|
276
|
+
safe = "".join(c if c.isalnum() or c == "_" else "_" for c in name)
|
|
277
|
+
return f"p_{safe}"
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _escape(text: str) -> str:
|
|
281
|
+
return text.replace("\\", "\\\\").replace('"', '\\"').replace("\n", " ")
|
cseq/server.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from typing import BinaryIO
|
|
8
|
+
|
|
9
|
+
_RANGE_RE = re.compile(r"bytes=(\d*)-(\d*)$")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class CseqRequestHandler(SimpleHTTPRequestHandler):
|
|
13
|
+
"""Static-file handler with single-range support for large .cseqdata files."""
|
|
14
|
+
|
|
15
|
+
server_version = "cseq-http/1"
|
|
16
|
+
|
|
17
|
+
def end_headers(self) -> None:
|
|
18
|
+
self.send_header("Cache-Control", "no-store")
|
|
19
|
+
self.send_header("Accept-Ranges", "bytes")
|
|
20
|
+
super().end_headers()
|
|
21
|
+
|
|
22
|
+
def send_head(self):
|
|
23
|
+
path = Path(self.translate_path(self.path))
|
|
24
|
+
if path.is_dir():
|
|
25
|
+
return super().send_head()
|
|
26
|
+
try:
|
|
27
|
+
stat = path.stat()
|
|
28
|
+
except OSError:
|
|
29
|
+
self.send_error(404, "File not found")
|
|
30
|
+
return None
|
|
31
|
+
range_header = self.headers.get("Range")
|
|
32
|
+
if not range_header:
|
|
33
|
+
self._cseq_range = None
|
|
34
|
+
return super().send_head()
|
|
35
|
+
parsed = _parse_range(range_header, stat.st_size)
|
|
36
|
+
if parsed is None:
|
|
37
|
+
self.send_response(416)
|
|
38
|
+
self.send_header("Content-Range", f"bytes */{stat.st_size}")
|
|
39
|
+
self.send_header("Content-Length", "0")
|
|
40
|
+
self.end_headers()
|
|
41
|
+
return None
|
|
42
|
+
start, end = parsed
|
|
43
|
+
ctype = self.guess_type(str(path))
|
|
44
|
+
try:
|
|
45
|
+
f = path.open("rb")
|
|
46
|
+
except OSError:
|
|
47
|
+
self.send_error(404, "File not found")
|
|
48
|
+
return None
|
|
49
|
+
self.send_response(206)
|
|
50
|
+
self.send_header("Content-type", ctype)
|
|
51
|
+
self.send_header("Content-Range", f"bytes {start}-{end}/{stat.st_size}")
|
|
52
|
+
self.send_header("Content-Length", str(end - start + 1))
|
|
53
|
+
self.send_header("Last-Modified", self.date_time_string(stat.st_mtime))
|
|
54
|
+
self.end_headers()
|
|
55
|
+
f.seek(start)
|
|
56
|
+
self._cseq_range = (start, end)
|
|
57
|
+
return f
|
|
58
|
+
|
|
59
|
+
def copyfile(self, source: BinaryIO, outputfile: BinaryIO) -> None:
|
|
60
|
+
byte_range = getattr(self, "_cseq_range", None)
|
|
61
|
+
if byte_range is None:
|
|
62
|
+
return super().copyfile(source, outputfile)
|
|
63
|
+
start, end = byte_range
|
|
64
|
+
remaining = end - start + 1
|
|
65
|
+
while remaining > 0:
|
|
66
|
+
chunk = source.read(min(1024 * 1024, remaining))
|
|
67
|
+
if not chunk:
|
|
68
|
+
break
|
|
69
|
+
outputfile.write(chunk)
|
|
70
|
+
remaining -= len(chunk)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _parse_range(value: str, size: int) -> tuple[int, int] | None:
|
|
74
|
+
m = _RANGE_RE.fullmatch(value.strip())
|
|
75
|
+
if not m or size <= 0:
|
|
76
|
+
return None
|
|
77
|
+
a, b = m.groups()
|
|
78
|
+
if not a and not b:
|
|
79
|
+
return None
|
|
80
|
+
if not a:
|
|
81
|
+
length = int(b)
|
|
82
|
+
if length <= 0:
|
|
83
|
+
return None
|
|
84
|
+
start = max(0, size - length)
|
|
85
|
+
return start, size - 1
|
|
86
|
+
start = int(a)
|
|
87
|
+
end = int(b) if b else size - 1
|
|
88
|
+
if start >= size or start < 0 or end < start:
|
|
89
|
+
return None
|
|
90
|
+
return start, min(end, size - 1)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def make_server(root: str | Path, host: str = "127.0.0.1", port: int = 8000) -> ThreadingHTTPServer:
|
|
94
|
+
directory = str(Path(root).resolve())
|
|
95
|
+
|
|
96
|
+
class Handler(CseqRequestHandler):
|
|
97
|
+
def __init__(self, *args, **kwargs):
|
|
98
|
+
super().__init__(*args, directory=directory, **kwargs)
|
|
99
|
+
|
|
100
|
+
def log_message(self, format: str, *args) -> None:
|
|
101
|
+
# Keep CLI output clean; request logs can be added via a future verbose flag.
|
|
102
|
+
return
|
|
103
|
+
|
|
104
|
+
return ThreadingHTTPServer((host, port), Handler)
|
cseq/source_index.py
ADDED
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import sqlite3
|
|
6
|
+
import threading
|
|
7
|
+
import os
|
|
8
|
+
import time
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class SourceFingerprintIndex:
|
|
14
|
+
"""Persistent metadata/hash index for project source files.
|
|
15
|
+
|
|
16
|
+
The index is only an acceleration cache. A content hash is reused only when
|
|
17
|
+
the full stat identity used here is unchanged; otherwise bytes are hashed
|
|
18
|
+
again. Dependency snapshots are stored separately and validated against the
|
|
19
|
+
current file hashes before reuse.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
SCHEMA_VERSION = 1
|
|
23
|
+
|
|
24
|
+
def __init__(self, path: str | Path) -> None:
|
|
25
|
+
self.path = Path(path)
|
|
26
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
27
|
+
self.connection = sqlite3.connect(self.path, check_same_thread=False)
|
|
28
|
+
self._lock = threading.RLock()
|
|
29
|
+
self.connection.execute("PRAGMA journal_mode=OFF")
|
|
30
|
+
self.connection.execute("PRAGMA synchronous=OFF")
|
|
31
|
+
self.connection.execute("PRAGMA temp_store=MEMORY")
|
|
32
|
+
self.connection.executescript(
|
|
33
|
+
"""
|
|
34
|
+
CREATE TABLE IF NOT EXISTS meta(
|
|
35
|
+
key TEXT PRIMARY KEY,
|
|
36
|
+
value TEXT NOT NULL
|
|
37
|
+
);
|
|
38
|
+
CREATE TABLE IF NOT EXISTS files(
|
|
39
|
+
path TEXT PRIMARY KEY,
|
|
40
|
+
size INTEGER NOT NULL,
|
|
41
|
+
mtime_ns INTEGER NOT NULL,
|
|
42
|
+
ctime_ns INTEGER NOT NULL,
|
|
43
|
+
dev INTEGER NOT NULL,
|
|
44
|
+
ino INTEGER NOT NULL,
|
|
45
|
+
sha256 TEXT NOT NULL
|
|
46
|
+
) WITHOUT ROWID;
|
|
47
|
+
CREATE TABLE IF NOT EXISTS dependencies(
|
|
48
|
+
tu_path TEXT NOT NULL,
|
|
49
|
+
context_hash TEXT NOT NULL,
|
|
50
|
+
source_hash TEXT NOT NULL,
|
|
51
|
+
fingerprint TEXT NOT NULL,
|
|
52
|
+
records_json TEXT NOT NULL,
|
|
53
|
+
PRIMARY KEY(tu_path, context_hash)
|
|
54
|
+
) WITHOUT ROWID;
|
|
55
|
+
"""
|
|
56
|
+
)
|
|
57
|
+
row = self.connection.execute(
|
|
58
|
+
"SELECT value FROM meta WHERE key='schema_version'"
|
|
59
|
+
).fetchone()
|
|
60
|
+
old = int(row[0]) if row and str(row[0]).isdigit() else None
|
|
61
|
+
if old != self.SCHEMA_VERSION:
|
|
62
|
+
self.connection.execute("DELETE FROM files")
|
|
63
|
+
self.connection.execute("DELETE FROM dependencies")
|
|
64
|
+
self.connection.execute(
|
|
65
|
+
"INSERT OR REPLACE INTO meta(key,value) VALUES('schema_version',?)",
|
|
66
|
+
(str(self.SCHEMA_VERSION),),
|
|
67
|
+
)
|
|
68
|
+
self.connection.commit()
|
|
69
|
+
|
|
70
|
+
self.hash_hits = 0
|
|
71
|
+
self.hash_misses = 0
|
|
72
|
+
self.dependency_hits = 0
|
|
73
|
+
self.dependency_misses = 0
|
|
74
|
+
|
|
75
|
+
@staticmethod
|
|
76
|
+
def _stat_identity(path: Path) -> tuple[int, int, int, int, int]:
|
|
77
|
+
st = path.stat()
|
|
78
|
+
return (
|
|
79
|
+
int(st.st_size),
|
|
80
|
+
int(getattr(st, "st_mtime_ns", int(st.st_mtime * 1_000_000_000))),
|
|
81
|
+
int(getattr(st, "st_ctime_ns", int(st.st_ctime * 1_000_000_000))),
|
|
82
|
+
int(getattr(st, "st_dev", 0)),
|
|
83
|
+
int(getattr(st, "st_ino", 0)),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
@staticmethod
|
|
87
|
+
def _hash_file(path: Path) -> str:
|
|
88
|
+
h = hashlib.sha256()
|
|
89
|
+
with path.open("rb") as f:
|
|
90
|
+
for chunk in iter(lambda: f.read(1024 * 1024), b""):
|
|
91
|
+
h.update(chunk)
|
|
92
|
+
return h.hexdigest()
|
|
93
|
+
|
|
94
|
+
def content_hash(self, path: str | Path) -> str:
|
|
95
|
+
p = Path(path).resolve()
|
|
96
|
+
key = str(p)
|
|
97
|
+
size, mtime_ns, ctime_ns, dev, ino = self._stat_identity(p)
|
|
98
|
+
with self._lock:
|
|
99
|
+
row = self.connection.execute(
|
|
100
|
+
"SELECT size,mtime_ns,ctime_ns,dev,ino,sha256 FROM files WHERE path=?",
|
|
101
|
+
(key,),
|
|
102
|
+
).fetchone()
|
|
103
|
+
if row and tuple(row[:5]) == (size, mtime_ns, ctime_ns, dev, ino):
|
|
104
|
+
# Windows filesystems can expose unchanged stat timestamps for a
|
|
105
|
+
# same-size rewrite performed within one clock tick. Revalidate
|
|
106
|
+
# only very recent files; normal warm-cache scans remain metadata-
|
|
107
|
+
# only while immediate edits cannot silently reuse stale content.
|
|
108
|
+
recent_windows_file = (
|
|
109
|
+
os.name == "nt"
|
|
110
|
+
and 0 <= time.time_ns() - mtime_ns <= 250_000_000
|
|
111
|
+
)
|
|
112
|
+
if not recent_windows_file:
|
|
113
|
+
with self._lock:
|
|
114
|
+
self.hash_hits += 1
|
|
115
|
+
return str(row[5])
|
|
116
|
+
digest = self._hash_file(p)
|
|
117
|
+
if digest == str(row[5]):
|
|
118
|
+
with self._lock:
|
|
119
|
+
self.hash_hits += 1
|
|
120
|
+
return digest
|
|
121
|
+
with self._lock:
|
|
122
|
+
self.connection.execute(
|
|
123
|
+
"""INSERT OR REPLACE INTO files(path,size,mtime_ns,ctime_ns,dev,ino,sha256)
|
|
124
|
+
VALUES(?,?,?,?,?,?,?)""",
|
|
125
|
+
(key, size, mtime_ns, ctime_ns, dev, ino, digest),
|
|
126
|
+
)
|
|
127
|
+
self.hash_misses += 1
|
|
128
|
+
return digest
|
|
129
|
+
digest = self._hash_file(p)
|
|
130
|
+
with self._lock:
|
|
131
|
+
self.connection.execute(
|
|
132
|
+
"""INSERT OR REPLACE INTO files(path,size,mtime_ns,ctime_ns,dev,ino,sha256)
|
|
133
|
+
VALUES(?,?,?,?,?,?,?)""",
|
|
134
|
+
(key, size, mtime_ns, ctime_ns, dev, ino, digest),
|
|
135
|
+
)
|
|
136
|
+
self.hash_misses += 1
|
|
137
|
+
return digest
|
|
138
|
+
|
|
139
|
+
def remove_missing_files(self, current_paths: set[str], project_root: str | Path | None = None) -> int:
|
|
140
|
+
root = Path(project_root).resolve() if project_root is not None else None
|
|
141
|
+
with self._lock:
|
|
142
|
+
rows = [r[0] for r in self.connection.execute("SELECT path FROM files")]
|
|
143
|
+
stale: list[str] = []
|
|
144
|
+
for path_text in rows:
|
|
145
|
+
if path_text in current_paths:
|
|
146
|
+
continue
|
|
147
|
+
if root is not None:
|
|
148
|
+
try:
|
|
149
|
+
Path(path_text).resolve().relative_to(root)
|
|
150
|
+
except ValueError:
|
|
151
|
+
# External include hashes are also cached here. A project
|
|
152
|
+
# scan must not evict them merely because they are not
|
|
153
|
+
# project-owned source files.
|
|
154
|
+
continue
|
|
155
|
+
stale.append(path_text)
|
|
156
|
+
for path_text in stale:
|
|
157
|
+
self.connection.execute("DELETE FROM files WHERE path=?", (path_text,))
|
|
158
|
+
return len(stale)
|
|
159
|
+
|
|
160
|
+
def dependency_snapshot(
|
|
161
|
+
self, tu_path: str, context_hash: str, source_hash: str
|
|
162
|
+
) -> tuple[str, list[dict[str, Any]]] | None:
|
|
163
|
+
with self._lock:
|
|
164
|
+
row = self.connection.execute(
|
|
165
|
+
"""SELECT source_hash,fingerprint,records_json
|
|
166
|
+
FROM dependencies WHERE tu_path=? AND context_hash=?""",
|
|
167
|
+
(tu_path, context_hash),
|
|
168
|
+
).fetchone()
|
|
169
|
+
if not row or row[0] != source_hash:
|
|
170
|
+
with self._lock:
|
|
171
|
+
self.dependency_misses += 1
|
|
172
|
+
return None
|
|
173
|
+
try:
|
|
174
|
+
records = json.loads(row[2])
|
|
175
|
+
except (TypeError, json.JSONDecodeError):
|
|
176
|
+
with self._lock:
|
|
177
|
+
self.dependency_misses += 1
|
|
178
|
+
return None
|
|
179
|
+
if not isinstance(records, list):
|
|
180
|
+
with self._lock:
|
|
181
|
+
self.dependency_misses += 1
|
|
182
|
+
return None
|
|
183
|
+
return str(row[1]), records
|
|
184
|
+
|
|
185
|
+
def note_dependency_hit(self) -> None:
|
|
186
|
+
with self._lock:
|
|
187
|
+
self.dependency_hits += 1
|
|
188
|
+
|
|
189
|
+
def note_dependency_miss(self) -> None:
|
|
190
|
+
with self._lock:
|
|
191
|
+
self.dependency_misses += 1
|
|
192
|
+
|
|
193
|
+
def put_dependency_snapshot(
|
|
194
|
+
self,
|
|
195
|
+
tu_path: str,
|
|
196
|
+
context_hash: str,
|
|
197
|
+
source_hash: str,
|
|
198
|
+
fingerprint: str,
|
|
199
|
+
records: list[dict[str, Any]],
|
|
200
|
+
) -> None:
|
|
201
|
+
with self._lock:
|
|
202
|
+
self.connection.execute(
|
|
203
|
+
"""INSERT OR REPLACE INTO dependencies(
|
|
204
|
+
tu_path,context_hash,source_hash,fingerprint,records_json
|
|
205
|
+
) VALUES(?,?,?,?,?)""",
|
|
206
|
+
(
|
|
207
|
+
tu_path,
|
|
208
|
+
context_hash,
|
|
209
|
+
source_hash,
|
|
210
|
+
fingerprint,
|
|
211
|
+
json.dumps(records, ensure_ascii=False, sort_keys=True, separators=(",", ":")),
|
|
212
|
+
),
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
def commit(self) -> None:
|
|
216
|
+
with self._lock:
|
|
217
|
+
self.connection.commit()
|
|
218
|
+
|
|
219
|
+
def clear(self) -> None:
|
|
220
|
+
with self._lock:
|
|
221
|
+
self.connection.execute("DELETE FROM files")
|
|
222
|
+
self.connection.execute("DELETE FROM dependencies")
|
|
223
|
+
self.connection.commit()
|
|
224
|
+
|
|
225
|
+
def close(self) -> None:
|
|
226
|
+
with self._lock:
|
|
227
|
+
self.connection.close()
|