cseq 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cseq/linker.py ADDED
@@ -0,0 +1,117 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, replace
4
+ from pathlib import Path
5
+ import re
6
+
7
+ from .model import CallTargetKind, LinkerEvidenceRecord, ProjectIndex
8
+
9
+ _SECTION_RE = re.compile(r"^\s*(\.[^\s]+)\s+0x([0-9A-Fa-f]+)\s+(?:0x[0-9A-Fa-f]+|\d+)\s+(\S+\.o(?:\([^)]*\))?)\s*$")
10
+ _SYMBOL_RE = re.compile(r"^\s*0x([0-9A-Fa-f]+)\s+([A-Za-z_.$][A-Za-z0-9_.$@]*)\s*$")
11
+ _COMPACT_RE = re.compile(r"^\s*0x([0-9A-Fa-f]+)\s+([A-Za-z_.$][A-Za-z0-9_.$@]*)\s+(\S+\.o(?:\([^)]*\))?)\s*$")
12
+
13
+
14
+ @dataclass(frozen=True, slots=True)
15
+ class LinkedSymbol:
16
+ name: str
17
+ address: int
18
+ object_file: str | None = None
19
+ section: str | None = None
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class LinkerMapIndex:
24
+ path: str
25
+ symbols: tuple[LinkedSymbol, ...]
26
+
27
+ def named(self, name: str) -> tuple[LinkedSymbol, ...]:
28
+ return tuple(s for s in self.symbols if s.name == name)
29
+
30
+
31
+ def parse_linker_map(path: str | Path) -> LinkerMapIndex:
32
+ """Parse a conservative GNU ld/lld-like map subset.
33
+
34
+ Unrecognized rows are ignored; no inference is made from them. The parser
35
+ supports section/object rows followed by symbol rows, plus a compact
36
+ ``0xADDR symbol object.o`` form useful for vendor adapters/tests.
37
+ """
38
+ p = Path(path)
39
+ current_object: str | None = None
40
+ current_section: str | None = None
41
+ symbols: list[LinkedSymbol] = []
42
+ for raw in p.read_text(encoding="utf-8", errors="replace").splitlines():
43
+ m = _SECTION_RE.match(raw)
44
+ if m:
45
+ current_section = m.group(1)
46
+ current_object = m.group(3)
47
+ continue
48
+ m = _COMPACT_RE.match(raw)
49
+ if m:
50
+ symbols.append(LinkedSymbol(m.group(2), int(m.group(1), 16), m.group(3), current_section))
51
+ continue
52
+ m = _SYMBOL_RE.match(raw)
53
+ if m:
54
+ symbols.append(LinkedSymbol(m.group(2), int(m.group(1), 16), current_object, current_section))
55
+ return LinkerMapIndex(str(p), tuple(symbols))
56
+
57
+
58
+ def apply_linker_map(index: ProjectIndex, linker: LinkerMapIndex) -> list[LinkerEvidenceRecord]:
59
+ """Add build-specific linker evidence and narrow duplicate direct targets.
60
+
61
+ A duplicate source definition is narrowed only when an object-file basename
62
+ from the linker map uniquely matches one source-file basename. Address/name
63
+ evidence without an object mapping is retained but never used to guess.
64
+ """
65
+ functions = list(index.functions())
66
+ by_name: dict[str, list] = {}
67
+ for fn in functions:
68
+ by_name.setdefault(fn.name, []).append(fn)
69
+
70
+ evidence: list[LinkerEvidenceRecord] = []
71
+ object_matches: dict[str, set[str]] = {}
72
+ for symbol in linker.symbols:
73
+ candidates = [f for f in by_name.get(symbol.name, []) if f.storage_class != "static"]
74
+ matched: list[str] = []
75
+ if symbol.object_file:
76
+ obj_stem = _object_stem(symbol.object_file)
77
+ matched = [f.qualified_id for f in candidates if Path(f.source_path).stem == obj_stem]
78
+ record = LinkerEvidenceRecord(
79
+ symbol_name=symbol.name,
80
+ address=symbol.address,
81
+ object_file=symbol.object_file,
82
+ section=symbol.section,
83
+ matched_function_ids=tuple(sorted(matched)),
84
+ provenance=f"linker-map:{linker.path}",
85
+ )
86
+ evidence.append(record)
87
+ if len(matched) == 1:
88
+ object_matches.setdefault(symbol.name, set()).add(matched[0])
89
+
90
+ refined = []
91
+ for call in index.callsites:
92
+ if call.target_kind != CallTargetKind.INTERNAL_POSSIBLE or not call.callee_name:
93
+ refined.append(call)
94
+ continue
95
+ linked = object_matches.get(call.callee_name, set())
96
+ candidates = set(call.target_function_ids)
97
+ narrowed = sorted(linked & candidates)
98
+ if len(narrowed) == 1:
99
+ refined.append(replace(
100
+ call,
101
+ target_kind=CallTargetKind.INTERNAL_EXACT,
102
+ target_function_id=narrowed[0],
103
+ target_function_ids=(narrowed[0],),
104
+ unknown_possible=False,
105
+ ))
106
+ else:
107
+ refined.append(call)
108
+ index.callsites[:] = refined
109
+ index.linker_evidence.extend(evidence)
110
+ return evidence
111
+
112
+
113
+ def _object_stem(value: str) -> str:
114
+ # Archive members such as libx.a(foo.o) are reduced to the member object.
115
+ m = re.search(r"\(([^()]+\.o)\)$", value)
116
+ leaf = m.group(1) if m else Path(value).name
117
+ return Path(leaf).stem
cseq/marker.py ADDED
@@ -0,0 +1,294 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, asdict
4
+ from enum import Enum
5
+ from hashlib import sha256
6
+ import json
7
+ from pathlib import Path
8
+ import re
9
+ from typing import Iterable
10
+
11
+
12
+ MARKER_PREFIX = "[[CSEQ:"
13
+ HUMAN_MARKER = "[[CSEQ]]"
14
+
15
+
16
+ class MarkerState(str, Enum):
17
+ SUGGESTED = "SUGGESTED"
18
+ APPLIED = "APPLIED"
19
+ OBSERVED = "OBSERVED"
20
+ STALE = "STALE"
21
+ ORPHANED = "ORPHANED"
22
+ REMOVED = "REMOVED"
23
+
24
+
25
+ @dataclass(frozen=True, slots=True)
26
+ class CStringLiteral:
27
+ start: int
28
+ end: int # exclusive byte offset
29
+ token: bytes
30
+ content: bytes
31
+
32
+
33
+ @dataclass(frozen=True, slots=True)
34
+ class MarkerRecord:
35
+ marker_id: str
36
+ source_path: str
37
+ byte_start: int
38
+ byte_end: int
39
+ original_literal: str
40
+ modified_literal: str
41
+ stable_fingerprint: str
42
+ state: MarkerState = MarkerState.APPLIED
43
+
44
+
45
+ @dataclass(frozen=True, slots=True)
46
+ class RewriteManifest:
47
+ source_path: str
48
+ before_hash: str
49
+ after_hash: str
50
+ marker: MarkerRecord
51
+ before_bytes_hex: str
52
+ after_bytes_hex: str
53
+
54
+ def to_json(self) -> str:
55
+ data = asdict(self)
56
+ data["marker"]["state"] = self.marker.state.value
57
+ return json.dumps(data, ensure_ascii=False, indent=2, sort_keys=True)
58
+
59
+ @classmethod
60
+ def from_json(cls, text: str) -> "RewriteManifest":
61
+ data = json.loads(text)
62
+ marker_data = dict(data["marker"])
63
+ marker_data["state"] = MarkerState(marker_data["state"])
64
+ return cls(
65
+ source_path=data["source_path"],
66
+ before_hash=data["before_hash"],
67
+ after_hash=data["after_hash"],
68
+ marker=MarkerRecord(**marker_data),
69
+ before_bytes_hex=data["before_bytes_hex"],
70
+ after_bytes_hex=data["after_bytes_hex"],
71
+ )
72
+
73
+
74
+ def file_hash(data: bytes) -> str:
75
+ return sha256(data).hexdigest()
76
+
77
+
78
+ def scan_c_string_literals(data: bytes) -> list[CStringLiteral]:
79
+ """Return ordinary/prefixed C string literal byte ranges.
80
+
81
+ The scanner deliberately works on bytes so source encoding, newlines and BOM
82
+ can be preserved exactly by rewrite operations. Comments and character
83
+ literals are skipped so quote-like bytes inside them are not rewritten.
84
+ """
85
+ out: list[CStringLiteral] = []
86
+ i = 0
87
+ n = len(data)
88
+ while i < n:
89
+ if data.startswith(b"//", i):
90
+ nl = data.find(b"\n", i + 2)
91
+ i = n if nl < 0 else nl + 1
92
+ continue
93
+ if data.startswith(b"/*", i):
94
+ close = data.find(b"*/", i + 2)
95
+ i = n if close < 0 else close + 2
96
+ continue
97
+ if data[i:i+1] == b"'":
98
+ i = _skip_quoted(data, i, ord("'"))
99
+ continue
100
+
101
+ prefix_len = _string_prefix_len(data, i)
102
+ quote = i + prefix_len
103
+ if quote < n and data[quote:quote+1] == b'"':
104
+ end = _skip_quoted(data, quote, ord('"'))
105
+ token_end = end
106
+ token = data[i:token_end]
107
+ # content excludes prefix and quotes; malformed unterminated literals
108
+ # are kept as scanned but never selected for rewrite.
109
+ if token_end <= n and token.endswith(b'"'):
110
+ content = data[quote + 1:token_end - 1]
111
+ out.append(CStringLiteral(i, token_end, token, content))
112
+ i = max(token_end, i + 1)
113
+ continue
114
+ i += 1
115
+ return out
116
+
117
+
118
+ def apply_marker(
119
+ path: str | Path,
120
+ *,
121
+ literal_contains: str,
122
+ occurrence: int = 0,
123
+ expected_hash: str | None = None,
124
+ project_root: str | Path | None = None,
125
+ marker_id: str | None = None,
126
+ manifest_path: str | Path | None = None,
127
+ ) -> RewriteManifest:
128
+ p = Path(path)
129
+ before = p.read_bytes()
130
+ before_hash = file_hash(before)
131
+ if expected_hash is not None and before_hash != expected_hash:
132
+ raise RuntimeError("source hash mismatch; refusing marker rewrite")
133
+
134
+ needle = literal_contains.encode("utf-8")
135
+ candidates = [lit for lit in scan_c_string_literals(before) if needle in lit.content]
136
+ if occurrence < 0 or occurrence >= len(candidates):
137
+ raise ValueError(f"matching string literal occurrence not found: {occurrence}")
138
+ lit = candidates[occurrence]
139
+
140
+ if HUMAN_MARKER.encode() in lit.content or b"[[CSEQ:" in lit.content:
141
+ raise ValueError("selected string literal already contains a CSEQ marker")
142
+
143
+ rel = _relative_name(p, project_root)
144
+ fingerprint = _fingerprint(rel, before, lit)
145
+ marker_id = marker_id or fingerprint[:12].upper()
146
+ marker_text = f"[[CSEQ:{marker_id}]] ".encode("ascii")
147
+
148
+ # Insert after the opening quote, preserving prefix (u8/L/u/U) exactly.
149
+ quote_rel = lit.token.find(b'"')
150
+ insert_at = lit.start + quote_rel + 1
151
+ after = before[:insert_at] + marker_text + before[insert_at:]
152
+ p.write_bytes(after)
153
+ after_hash = file_hash(after)
154
+
155
+ modified_token = lit.token[:quote_rel+1] + marker_text + lit.token[quote_rel+1:]
156
+ record = MarkerRecord(
157
+ marker_id=marker_id,
158
+ source_path=rel,
159
+ byte_start=lit.start,
160
+ byte_end=lit.end + len(marker_text),
161
+ original_literal=lit.token.decode("utf-8", errors="replace"),
162
+ modified_literal=modified_token.decode("utf-8", errors="replace"),
163
+ stable_fingerprint=fingerprint,
164
+ state=MarkerState.APPLIED,
165
+ )
166
+ manifest = RewriteManifest(
167
+ source_path=rel,
168
+ before_hash=before_hash,
169
+ after_hash=after_hash,
170
+ marker=record,
171
+ before_bytes_hex=before.hex(),
172
+ after_bytes_hex=after.hex(),
173
+ )
174
+ if manifest_path is not None:
175
+ mp = Path(manifest_path)
176
+ mp.parent.mkdir(parents=True, exist_ok=True)
177
+ mp.write_text(manifest.to_json(), encoding="utf-8")
178
+ return manifest
179
+
180
+
181
+ def apply_marker_overlay(
182
+ source_path: str | Path,
183
+ output_path: str | Path,
184
+ *,
185
+ literal_contains: str,
186
+ occurrence: int = 0,
187
+ expected_hash: str | None = None,
188
+ project_root: str | Path | None = None,
189
+ marker_id: str | None = None,
190
+ manifest_path: str | Path | None = None,
191
+ ) -> RewriteManifest:
192
+ """Create a marker-injected overlay without modifying the original source."""
193
+ source = Path(source_path)
194
+ output = Path(output_path)
195
+ original = source.read_bytes()
196
+ original_hash = file_hash(original)
197
+ if expected_hash is not None and original_hash != expected_hash:
198
+ raise RuntimeError("source hash mismatch; refusing marker overlay")
199
+ if source.resolve() == output.resolve():
200
+ raise ValueError("overlay output must differ from source path")
201
+ output.parent.mkdir(parents=True, exist_ok=True)
202
+ output.write_bytes(original)
203
+ try:
204
+ return apply_marker(
205
+ output,
206
+ literal_contains=literal_contains,
207
+ occurrence=occurrence,
208
+ expected_hash=original_hash,
209
+ project_root=project_root,
210
+ marker_id=marker_id,
211
+ manifest_path=manifest_path,
212
+ )
213
+ except Exception:
214
+ # Never leave a partially prepared overlay when marker application fails.
215
+ try:
216
+ output.unlink(missing_ok=True)
217
+ finally:
218
+ raise
219
+
220
+
221
+ def undo_marker(
222
+ path: str | Path,
223
+ manifest: RewriteManifest | str | Path,
224
+ ) -> None:
225
+ p = Path(path)
226
+ if isinstance(manifest, (str, Path)):
227
+ manifest = RewriteManifest.from_json(Path(manifest).read_text(encoding="utf-8"))
228
+ current = p.read_bytes()
229
+ if file_hash(current) != manifest.after_hash:
230
+ raise RuntimeError("source changed after marker application; refusing unsafe undo")
231
+ after_expected = bytes.fromhex(manifest.after_bytes_hex)
232
+ if current != after_expected:
233
+ raise RuntimeError("manifest after-image mismatch; refusing unsafe undo")
234
+ before = bytes.fromhex(manifest.before_bytes_hex)
235
+ if file_hash(before) != manifest.before_hash:
236
+ raise RuntimeError("manifest before-image hash mismatch")
237
+ p.write_bytes(before)
238
+
239
+
240
+ def audit_markers(paths: Iterable[str | Path]) -> dict[str, list[str]]:
241
+ """Return duplicate/malformed marker IDs without mutating source."""
242
+ by_id: dict[str, list[str]] = {}
243
+ malformed: list[str] = []
244
+ rx = re.compile(rb"\[\[CSEQ:([^\]]*)\]\]")
245
+ for path in paths:
246
+ p = Path(path)
247
+ data = p.read_bytes()
248
+ for m in rx.finditer(data):
249
+ marker_id = m.group(1).decode("ascii", errors="replace")
250
+ where = f"{p}:{m.start()}"
251
+ if not marker_id or not re.fullmatch(r"[A-Za-z0-9_-]+", marker_id):
252
+ malformed.append(where)
253
+ by_id.setdefault(marker_id, []).append(where)
254
+ duplicates = [f"{mid}:" + ",".join(locs) for mid, locs in sorted(by_id.items()) if mid and len(locs) > 1]
255
+ return {"duplicates": duplicates, "malformed": malformed}
256
+
257
+
258
+ def _relative_name(path: Path, project_root: str | Path | None) -> str:
259
+ if project_root is None:
260
+ return path.name
261
+ try:
262
+ return path.resolve().relative_to(Path(project_root).resolve()).as_posix()
263
+ except ValueError:
264
+ return path.name
265
+
266
+
267
+ def _fingerprint(rel: str, data: bytes, lit: CStringLiteral) -> str:
268
+ normalized = re.sub(rb"\[\[CSEQ(?::[^\]]+)?\]\]\s*", b"", lit.content)
269
+ left = data[max(0, lit.start - 80):lit.start]
270
+ right = data[lit.end:min(len(data), lit.end + 80)]
271
+ payload = b"\0".join([rel.encode("utf-8"), normalized, left, right])
272
+ return sha256(payload).hexdigest()
273
+
274
+
275
+ def _string_prefix_len(data: bytes, i: int) -> int:
276
+ # C11/C23-style prefixes commonly encountered in C source.
277
+ for prefix in (b'u8', b'u', b'U', b'L'):
278
+ if data.startswith(prefix + b'"', i):
279
+ return len(prefix)
280
+ return 0
281
+
282
+
283
+ def _skip_quoted(data: bytes, quote_pos: int, quote: int) -> int:
284
+ i = quote_pos + 1
285
+ n = len(data)
286
+ while i < n:
287
+ c = data[i]
288
+ if c == ord('\\'):
289
+ i += 2
290
+ continue
291
+ if c == quote:
292
+ return i + 1
293
+ i += 1
294
+ return n