cseq 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cseq/cache.py ADDED
@@ -0,0 +1,420 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import asdict
4
+ from hashlib import sha256
5
+ import json
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from .model import (
10
+ CallSiteRecord, CallTargetKind, CompactFunctionRecord, CompactFunctionTable, Diagnostic, FileParseArtifact, FlowEdge, FlowEdgeKind,
11
+ FunctionRecord, GuardedSourceRegion, ParseStatus, SourceRange, TranslationUnit, ValueSlot, ValueSlotKind,
12
+ ProjectIndex, CpuEvidenceRecord, CpuEvidenceLayer, CpuValueKind,
13
+ )
14
+
15
+ CACHE_SCHEMA_VERSION = 7
16
+
17
+
18
+ def build_cache_key(tu: TranslationUnit, *, parser_backend: str, parser_version: str = "1", include_dependency_hash: str | None = None) -> str:
19
+ payload = {
20
+ "schema": CACHE_SCHEMA_VERSION,
21
+ "source_hash": tu.source.content_hash,
22
+ "source_path": tu.source.project_relative_path,
23
+ "arguments": list(tu.arguments),
24
+ "configuration": tu.configuration_name,
25
+ "parser_backend": parser_backend,
26
+ "parser_version": parser_version,
27
+ "include_dependency_hash": include_dependency_hash,
28
+ }
29
+ raw = json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
30
+ return sha256(raw).hexdigest()
31
+
32
+
33
+ class JsonCache:
34
+ def __init__(self, root: str | Path) -> None:
35
+ self.root = Path(root)
36
+ self.root.mkdir(parents=True, exist_ok=True)
37
+
38
+ def path_for(self, key: str) -> Path:
39
+ return self.root / key[:2] / f"{key}.json"
40
+
41
+ def put(self, key: str, payload: dict[str, Any]) -> Path:
42
+ path = self.path_for(key)
43
+ path.parent.mkdir(parents=True, exist_ok=True)
44
+ wrapper = {"schema": CACHE_SCHEMA_VERSION, "payload": payload}
45
+ tmp = path.with_suffix(".tmp")
46
+ tmp.write_text(json.dumps(wrapper, ensure_ascii=False, sort_keys=True), encoding="utf-8")
47
+ tmp.replace(path)
48
+ return path
49
+
50
+ def get(self, key: str) -> dict[str, Any] | None:
51
+ path = self.path_for(key)
52
+ if not path.exists():
53
+ return None
54
+ try:
55
+ wrapper = json.loads(path.read_text(encoding="utf-8"))
56
+ except (OSError, json.JSONDecodeError):
57
+ return None
58
+ if wrapper.get("schema") != CACHE_SCHEMA_VERSION:
59
+ return None
60
+ payload = wrapper.get("payload")
61
+ return payload if isinstance(payload, dict) else None
62
+
63
+ def remove(self, key: str) -> bool:
64
+ path = self.path_for(key)
65
+ if not path.exists():
66
+ return False
67
+ path.unlink()
68
+ return True
69
+
70
+
71
+
72
+ def _range_to_dict(r: SourceRange) -> dict[str, int | None]:
73
+ return {
74
+ "start_line": r.start_line,
75
+ "start_column": r.start_column,
76
+ "end_line": r.end_line,
77
+ "end_column": r.end_column,
78
+ }
79
+
80
+
81
+ def _range_from_dict(d: dict[str, Any] | None) -> SourceRange:
82
+ d = d or {}
83
+ return SourceRange(d.get("start_line"), d.get("start_column"), d.get("end_line"), d.get("end_column"))
84
+
85
+
86
+ def serialize_parse_artifact(artifact: FileParseArtifact) -> dict[str, Any]:
87
+ """Serialize parser output without embedding source bytes.
88
+
89
+ Fast lexical artifacts use a compact representation because their function
90
+ records are numerous and highly repetitive on generated/large projects.
91
+ The semantic representation remains explicit for debuggability.
92
+ """
93
+ if artifact.parser_backend.startswith("fast-lexical"):
94
+ funcs = artifact.functions
95
+ function_index = None if isinstance(funcs, CompactFunctionTable) else {f.qualified_id: i for i, f in enumerate(funcs)}
96
+ def caller_index(c):
97
+ if isinstance(funcs, CompactFunctionTable):
98
+ return funcs.index_of_qid(c.caller_id)
99
+ return function_index.get(c.caller_id, -1)
100
+ return {
101
+ "format": "fast-compact-v2",
102
+ "status": artifact.status.value,
103
+ "parser_backend": artifact.parser_backend,
104
+ "functions": [
105
+ [
106
+ f.name, f.source_range.start_line, f.source_range.start_column,
107
+ f.source_range.end_line, f.source_range.end_column,
108
+ f.storage_class, f.type_qualifier, list(f.parameters),
109
+ ]
110
+ for f in artifact.functions
111
+ ],
112
+ "callsites": [
113
+ [
114
+ c.caller_id, c.caller_name, c.callee_name, c.raw_text,
115
+ c.source_range.start_line, c.source_range.start_column,
116
+ c.source_range.end_line, c.source_range.end_column,
117
+ ]
118
+ for c in artifact.callsites
119
+ ],
120
+ }
121
+ return {
122
+ "status": artifact.status.value,
123
+ "parser_backend": artifact.parser_backend,
124
+ "functions": [
125
+ {
126
+ "name": f.name, "qualified_id": f.qualified_id, "source_path": f.source_path,
127
+ "source_range": _range_to_dict(f.source_range), "storage_class": f.storage_class,
128
+ "type_qualifier": f.type_qualifier, "parameters": list(f.parameters),
129
+ } for f in artifact.functions
130
+ ],
131
+ "callsites": [
132
+ {
133
+ "caller_id": c.caller_id, "caller_name": c.caller_name, "callee_name": c.callee_name,
134
+ "raw_text": c.raw_text, "source_path": c.source_path,
135
+ "source_range": _range_to_dict(c.source_range), "target_kind": c.target_kind.value,
136
+ "target_function_id": c.target_function_id, "target_function_ids": list(c.target_function_ids),
137
+ "unknown_possible": c.unknown_possible, "callee_slot_key": c.callee_slot_key,
138
+ "argument_slot_keys": list(c.argument_slot_keys), "control_context": list(c.control_context),
139
+ "macro_expanded": c.macro_expanded, "spelling_range": _range_to_dict(c.spelling_range) if c.spelling_range else None,
140
+ "expansion_range": _range_to_dict(c.expansion_range) if c.expansion_range else None, "guard_expression": c.guard_expression,
141
+ } for c in artifact.callsites
142
+ ],
143
+ "value_slots": [
144
+ {
145
+ "key": v.key, "kind": v.kind.value, "name": v.name, "function_id": v.function_id,
146
+ "field_name": v.field_name, "index": v.index,
147
+ } for v in artifact.value_slots
148
+ ],
149
+ "flow_edges": [
150
+ {
151
+ "source_key": e.source_key, "target_key": e.target_key, "kind": e.kind.value,
152
+ "source_path": e.source_path, "source_range": _range_to_dict(e.source_range),
153
+ } for e in artifact.flow_edges
154
+ ],
155
+ "diagnostics": [{"severity": d.severity, "message": d.message, "raw": d.raw} for d in artifact.diagnostics],
156
+ "opaque_regions": [_range_to_dict(r) for r in artifact.opaque_regions],
157
+ "guarded_regions": [
158
+ {"source_path": g.source_path, "source_range": _range_to_dict(g.source_range), "guard_expression": g.guard_expression}
159
+ for g in artifact.guarded_regions
160
+ ],
161
+ }
162
+
163
+
164
+ def serialize_parse_artifact_light(artifact: FileParseArtifact) -> dict[str, Any]:
165
+ """Serialize a parse artifact without serializing its function table at all.
166
+
167
+ This is intentionally separate from serialize_parse_artifact(): building a
168
+ full million-function JSON object only to delete ``functions`` defeats the
169
+ out-of-core design. Caller IDs are self-contained in cache schema v7.
170
+ """
171
+ if artifact.parser_backend.startswith("fast-lexical"):
172
+ return {
173
+ "format": "fast-compact-v2",
174
+ "status": artifact.status.value,
175
+ "parser_backend": artifact.parser_backend,
176
+ "functions": [],
177
+ "callsites": [
178
+ [
179
+ c.caller_id, c.caller_name, c.callee_name, c.raw_text,
180
+ c.source_range.start_line, c.source_range.start_column,
181
+ c.source_range.end_line, c.source_range.end_column,
182
+ ]
183
+ for c in artifact.callsites
184
+ ],
185
+ }
186
+ return {
187
+ "status": artifact.status.value,
188
+ "parser_backend": artifact.parser_backend,
189
+ "functions": [],
190
+ "callsites": [
191
+ {
192
+ "caller_id": c.caller_id, "caller_name": c.caller_name, "callee_name": c.callee_name,
193
+ "raw_text": c.raw_text, "source_path": c.source_path,
194
+ "source_range": _range_to_dict(c.source_range), "target_kind": c.target_kind.value,
195
+ "target_function_id": c.target_function_id, "target_function_ids": list(c.target_function_ids),
196
+ "unknown_possible": c.unknown_possible, "callee_slot_key": c.callee_slot_key,
197
+ "argument_slot_keys": list(c.argument_slot_keys), "control_context": list(c.control_context),
198
+ "macro_expanded": c.macro_expanded, "spelling_range": _range_to_dict(c.spelling_range) if c.spelling_range else None,
199
+ "expansion_range": _range_to_dict(c.expansion_range) if c.expansion_range else None, "guard_expression": c.guard_expression,
200
+ } for c in artifact.callsites
201
+ ],
202
+ "value_slots": [
203
+ {
204
+ "key": v.key, "kind": v.kind.value, "name": v.name, "function_id": v.function_id,
205
+ "field_name": v.field_name, "index": v.index,
206
+ } for v in artifact.value_slots
207
+ ],
208
+ "flow_edges": [
209
+ {
210
+ "source_key": e.source_key, "target_key": e.target_key, "kind": e.kind.value,
211
+ "source_path": e.source_path, "source_range": _range_to_dict(e.source_range),
212
+ } for e in artifact.flow_edges
213
+ ],
214
+ "diagnostics": [{"severity": d.severity, "message": d.message, "raw": d.raw} for d in artifact.diagnostics],
215
+ "opaque_regions": [_range_to_dict(r) for r in artifact.opaque_regions],
216
+ "guarded_regions": [
217
+ {"source_path": g.source_path, "source_range": _range_to_dict(g.source_range), "guard_expression": g.guard_expression}
218
+ for g in artifact.guarded_regions
219
+ ],
220
+ }
221
+
222
+
223
+ def strip_parse_artifact_functions(payload: dict[str, Any]) -> dict[str, Any]:
224
+ """Return a lightweight parse payload whose function table lives elsewhere.
225
+
226
+ CallSite caller IDs are self-contained in cache schema v7, so Static Symbol
227
+ Store reuse can restore every non-function parser artifact without loading
228
+ the often-dominant function array.
229
+ """
230
+ light = dict(payload)
231
+ light["functions"] = []
232
+ return light
233
+
234
+
235
+ def deserialize_parse_artifact(payload: dict[str, Any], tu: TranslationUnit, *, include_functions: bool = True) -> FileParseArtifact:
236
+ if payload.get("format") == "fast-compact-v2":
237
+ functions = CompactFunctionTable(tu.source.project_relative_path)
238
+ if include_functions:
239
+ for x in payload.get("functions", []):
240
+ name, sl, sc, el, ec, storage, type_qualifier, params = x
241
+ functions.append_fields(
242
+ name, sl, sc, el, ec, storage, type_qualifier, tuple(params or ()),
243
+ )
244
+ callsites: list[CallSiteRecord] = []
245
+ for x in payload.get("callsites", []):
246
+ caller_id, caller_name, callee_name, raw_text, sl, sc, el, ec = x
247
+ callsites.append(CallSiteRecord(
248
+ caller_id=caller_id, caller_name=caller_name, callee_name=callee_name,
249
+ raw_text=raw_text, source_path=tu.source.project_relative_path,
250
+ source_range=SourceRange(sl, sc, el, ec),
251
+ ))
252
+ return FileParseArtifact(
253
+ translation_unit=tu, status=ParseStatus(payload["status"]),
254
+ functions=functions, callsites=callsites,
255
+ parser_backend=str(payload.get("parser_backend") or "fast-lexical-v1"),
256
+ )
257
+ return FileParseArtifact(
258
+ translation_unit=tu,
259
+ status=ParseStatus(payload["status"]),
260
+ functions=[FunctionRecord(
261
+ name=x["name"], qualified_id=x["qualified_id"], source_path=x["source_path"],
262
+ source_range=_range_from_dict(x.get("source_range")), storage_class=x.get("storage_class"),
263
+ type_qualifier=x.get("type_qualifier"), parameters=tuple(x.get("parameters") or ()),
264
+ ) for x in payload.get("functions", [])] if include_functions else [],
265
+ callsites=[CallSiteRecord(
266
+ caller_id=x["caller_id"], caller_name=x["caller_name"], callee_name=x.get("callee_name"),
267
+ raw_text=x["raw_text"], source_path=x["source_path"], source_range=_range_from_dict(x.get("source_range")),
268
+ target_kind=CallTargetKind(x.get("target_kind", "UNKNOWN")), target_function_id=x.get("target_function_id"),
269
+ target_function_ids=tuple(x.get("target_function_ids") or ()), unknown_possible=bool(x.get("unknown_possible")),
270
+ callee_slot_key=x.get("callee_slot_key"), argument_slot_keys=tuple(x.get("argument_slot_keys") or ()), control_context=tuple(x.get("control_context") or ()),
271
+ macro_expanded=bool(x.get("macro_expanded")), spelling_range=_range_from_dict(x.get("spelling_range")) if x.get("spelling_range") else None,
272
+ expansion_range=_range_from_dict(x.get("expansion_range")) if x.get("expansion_range") else None, guard_expression=x.get("guard_expression"),
273
+ ) for x in payload.get("callsites", [])],
274
+ value_slots=[ValueSlot(
275
+ key=x["key"], kind=ValueSlotKind(x["kind"]), name=x["name"], function_id=x.get("function_id"),
276
+ field_name=x.get("field_name"), index=x.get("index"),
277
+ ) for x in payload.get("value_slots", [])],
278
+ flow_edges=[FlowEdge(
279
+ source_key=x["source_key"], target_key=x["target_key"], kind=FlowEdgeKind(x["kind"]),
280
+ source_path=x.get("source_path"), source_range=_range_from_dict(x.get("source_range")),
281
+ ) for x in payload.get("flow_edges", [])],
282
+ diagnostics=[Diagnostic(x["severity"], x["message"], x.get("raw")) for x in payload.get("diagnostics", [])],
283
+ opaque_regions=[_range_from_dict(x) for x in payload.get("opaque_regions", [])],
284
+ guarded_regions=[GuardedSourceRegion(
285
+ source_path=x.get("source_path") or tu.source.project_relative_path,
286
+ source_range=_range_from_dict(x.get("source_range")),
287
+ guard_expression=str(x.get("guard_expression") or "<unknown>"),
288
+ ) for x in payload.get("guarded_regions", [])],
289
+ parser_backend=str(payload.get("parser_backend") or "unknown"),
290
+ )
291
+
292
+
293
+ ANALYSIS_CACHE_SCHEMA_VERSION = 7
294
+
295
+
296
+ def _config_fingerprint_payload(config, defined_macros: tuple[str, ...]) -> dict[str, Any]:
297
+ return {
298
+ "cpu_groups": {k: list(v) for k, v in sorted(config.cpu_groups.items())},
299
+ "preprocessor_rules": [
300
+ [r.when_defined, r.assign_group, r.assign_cpu] for r in config.preprocessor_rules
301
+ ],
302
+ "path_rules": [[r.glob, r.assign_group, r.assign_cpu] for r in config.path_rules],
303
+ "override_rules": [[r.file, r.assign_group, r.assign_cpu] for r in config.override_rules],
304
+ "configurations": {k: list(v) for k, v in sorted(config.configurations.items())},
305
+ "defined_macros": list(defined_macros),
306
+ }
307
+
308
+
309
+ def build_analysis_cache_key(index: ProjectIndex, config, defined_macros: tuple[str, ...]) -> str:
310
+ """Fingerprint project-wide symbol/value-flow inputs.
311
+
312
+ Parser cache is TU-local. This key intentionally spans every parse artifact,
313
+ so a symbol change in another translation unit invalidates project-wide call
314
+ resolution without forcing unrelated files to reparse.
315
+ """
316
+ payload = {
317
+ "schema": ANALYSIS_CACHE_SCHEMA_VERSION,
318
+ "artifacts": [serialize_parse_artifact(a) for a in index.parse_artifacts],
319
+ "config": _config_fingerprint_payload(config, defined_macros),
320
+ }
321
+ raw = json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
322
+ return sha256(raw).hexdigest()
323
+
324
+
325
+ def build_analysis_cache_key_from_fingerprints(
326
+ artifact_fingerprints: list[str] | tuple[str, ...], config, defined_macros: tuple[str, ...]
327
+ ) -> str:
328
+ """Build the project-analysis key from TU parser fingerprints.
329
+
330
+ Each parser fingerprint already covers source content, compile arguments,
331
+ configuration, parser backend/schema and include dependencies. Re-serializing
332
+ every FunctionRecord/CallSite to build a second hash is redundant and becomes
333
+ very expensive on million-function projects.
334
+ """
335
+ payload = {
336
+ "schema": ANALYSIS_CACHE_SCHEMA_VERSION,
337
+ "artifact_fingerprints": list(artifact_fingerprints),
338
+ "config": _config_fingerprint_payload(config, defined_macros),
339
+ }
340
+ raw = json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
341
+ return sha256(raw).hexdigest()
342
+
343
+
344
+ def serialize_analysis_snapshot(index: ProjectIndex) -> dict[str, Any]:
345
+ return {
346
+ "analysis_schema": ANALYSIS_CACHE_SCHEMA_VERSION,
347
+ "callsites": [
348
+ {
349
+ "caller_id": c.caller_id,
350
+ "caller_name": c.caller_name,
351
+ "callee_name": c.callee_name,
352
+ "raw_text": c.raw_text,
353
+ "source_path": c.source_path,
354
+ "source_range": _range_to_dict(c.source_range),
355
+ "target_kind": c.target_kind.value,
356
+ "target_function_id": c.target_function_id,
357
+ "target_function_ids": list(c.target_function_ids),
358
+ "unknown_possible": c.unknown_possible,
359
+ "callee_slot_key": c.callee_slot_key,
360
+ "argument_slot_keys": list(c.argument_slot_keys), "control_context": list(c.control_context),
361
+ "macro_expanded": c.macro_expanded, "spelling_range": _range_to_dict(c.spelling_range) if c.spelling_range else None,
362
+ "expansion_range": _range_to_dict(c.expansion_range) if c.expansion_range else None, "guard_expression": c.guard_expression,
363
+ }
364
+ for c in index.callsites
365
+ ],
366
+ "flow_edges": [
367
+ {
368
+ "source_key": e.source_key,
369
+ "target_key": e.target_key,
370
+ "kind": e.kind.value,
371
+ "source_path": e.source_path,
372
+ "source_range": _range_to_dict(e.source_range),
373
+ }
374
+ for e in index.flow_edges
375
+ ],
376
+ "cpu_evidence": [
377
+ {
378
+ "target_function_id": e.target_function_id,
379
+ "layer": e.layer.value,
380
+ "value_kind": e.value_kind.value,
381
+ "value": e.value,
382
+ "provenance": e.provenance,
383
+ "explicit_override": e.explicit_override,
384
+ }
385
+ for e in index.cpu_evidence
386
+ ],
387
+ }
388
+
389
+
390
+ def restore_analysis_snapshot(index: ProjectIndex, payload: dict[str, Any]) -> bool:
391
+ if payload.get("analysis_schema") != ANALYSIS_CACHE_SCHEMA_VERSION:
392
+ return False
393
+ index.callsites[:] = [
394
+ CallSiteRecord(
395
+ caller_id=x["caller_id"], caller_name=x["caller_name"], callee_name=x.get("callee_name"),
396
+ raw_text=x["raw_text"], source_path=x["source_path"], source_range=_range_from_dict(x.get("source_range")),
397
+ target_kind=CallTargetKind(x.get("target_kind", "UNKNOWN")), target_function_id=x.get("target_function_id"),
398
+ target_function_ids=tuple(x.get("target_function_ids") or ()), unknown_possible=bool(x.get("unknown_possible")),
399
+ callee_slot_key=x.get("callee_slot_key"), argument_slot_keys=tuple(x.get("argument_slot_keys") or ()), control_context=tuple(x.get("control_context") or ()),
400
+ macro_expanded=bool(x.get("macro_expanded")), spelling_range=_range_from_dict(x.get("spelling_range")) if x.get("spelling_range") else None,
401
+ expansion_range=_range_from_dict(x.get("expansion_range")) if x.get("expansion_range") else None, guard_expression=x.get("guard_expression"),
402
+ )
403
+ for x in payload.get("callsites", [])
404
+ ]
405
+ index.flow_edges[:] = [
406
+ FlowEdge(
407
+ source_key=x["source_key"], target_key=x["target_key"], kind=FlowEdgeKind(x["kind"]),
408
+ source_path=x.get("source_path"), source_range=_range_from_dict(x.get("source_range")),
409
+ )
410
+ for x in payload.get("flow_edges", [])
411
+ ]
412
+ index.cpu_evidence[:] = [
413
+ CpuEvidenceRecord(
414
+ target_function_id=x["target_function_id"], layer=CpuEvidenceLayer(x["layer"]),
415
+ value_kind=CpuValueKind(x["value_kind"]), value=x["value"], provenance=x["provenance"],
416
+ explicit_override=bool(x.get("explicit_override")),
417
+ )
418
+ for x in payload.get("cpu_evidence", [])
419
+ ]
420
+ return True