awgit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- awgit/__init__.py +53 -0
- awgit/bodies.py +323 -0
- awgit/bridge.py +117 -0
- awgit/capture.py +441 -0
- awgit/cli.py +488 -0
- awgit/data_root.py +30 -0
- awgit/diff.py +82 -0
- awgit/hooks/chain.sh +38 -0
- awgit/hooks/post-commit.d/vcs-capture +38 -0
- awgit/hooks/pre-commit.d/vcs-lease-check +15 -0
- awgit/identity.py +80 -0
- awgit/leases.py +264 -0
- awgit/ledger.py +80 -0
- awgit/mcp.py +94 -0
- awgit/merge.py +331 -0
- awgit/nodeid.py +210 -0
- awgit/oplog.py +221 -0
- awgit/parser.py +186 -0
- awgit/schema.py +190 -0
- awgit/sync.py +180 -0
- awgit-0.1.0.dist-info/METADATA +117 -0
- awgit-0.1.0.dist-info/RECORD +26 -0
- awgit-0.1.0.dist-info/WHEEL +5 -0
- awgit-0.1.0.dist-info/entry_points.txt +2 -0
- awgit-0.1.0.dist-info/licenses/LICENSE +202 -0
- awgit-0.1.0.dist-info/top_level.txt +1 -0
awgit/oplog.py
ADDED
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""Durable append-only op-log for the semantic-VCS layer.
|
|
2
|
+
|
|
3
|
+
An ``EditOp`` is one commit's semantic annotation. The log is append-only
|
|
4
|
+
JSONL — each line one ``EditOp.to_dict()`` — under an OS-level exclusive lock
|
|
5
|
+
with an fsync on every append. It is NOT recomputable (ops carry summaries), so
|
|
6
|
+
``export`` is the durability story; a fresh machine clones + reindexes and
|
|
7
|
+
optionally imports a shipped export.
|
|
8
|
+
|
|
9
|
+
The post-commit hook runs as its own sync process, so ``FileLock`` below is a
|
|
10
|
+
plain blocking call reached only from sync code (PQ010 escape, see comment).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import threading
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from threading import RLock
|
|
20
|
+
from typing import Dict, Iterable, List, Optional
|
|
21
|
+
|
|
22
|
+
from awgit.data_root import vcs_data_root
|
|
23
|
+
from awgit.schema import EditOp
|
|
24
|
+
|
|
25
|
+
_HEADER = "# vcs-oplog v1"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _lock_path(data_root: Path) -> Path:
|
|
29
|
+
return data_root / "oplog.lock"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# Per-path process-local mutexes guarding the cross-process file locks.
|
|
33
|
+
# Windows msvcrt ``LK_LOCK`` retries ~10x then raises EDEADLK under dense
|
|
34
|
+
# INTRA-process thread contention (20 threads, one process, one lock byte) —
|
|
35
|
+
# cross-process locking works (separate processes block), but threads in ONE
|
|
36
|
+
# process trip the retry limit. The mutex serializes threads in-process; the
|
|
37
|
+
# file lock still serializes across processes (the real post-commit-hook case:
|
|
38
|
+
# concurrent captures from separate hook processes). Caught by the M6
|
|
39
|
+
# concurrent-put test; M1's threaded append test passed only by looser
|
|
40
|
+
# contention.
|
|
41
|
+
_PROCESS_LOCKS: Dict[str, "threading.Lock"] = {}
|
|
42
|
+
_PROCESS_LOCKS_GUARD = threading.Lock()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _process_lock(path: Path) -> "threading.Lock":
|
|
46
|
+
key = str(path.resolve())
|
|
47
|
+
with _PROCESS_LOCKS_GUARD:
|
|
48
|
+
return _PROCESS_LOCKS.setdefault(key, threading.Lock())
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class FileLock:
|
|
52
|
+
"""Cross-process exclusive lock (msvcrt on Windows, fcntl on POSIX),
|
|
53
|
+
serialized within the process by a per-path mutex.
|
|
54
|
+
|
|
55
|
+
blocking-ok: OS-level advisory lock; reached only from the sync CLI/hook
|
|
56
|
+
path, never from an event loop (PQ010).
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
def __init__(self, path: Path) -> None:
|
|
60
|
+
self._path = path
|
|
61
|
+
self._fh = None
|
|
62
|
+
self._plock = _process_lock(path)
|
|
63
|
+
|
|
64
|
+
def __enter__(self) -> "FileLock":
|
|
65
|
+
self._plock.acquire()
|
|
66
|
+
try:
|
|
67
|
+
self._path.parent.mkdir(parents=True, exist_ok=True)
|
|
68
|
+
self._fh = open(self._path, "a+b")
|
|
69
|
+
if self._fh.seek(0, os.SEEK_END) == 0:
|
|
70
|
+
self._fh.write(b"\0")
|
|
71
|
+
self._fh.flush()
|
|
72
|
+
self._fh.seek(0)
|
|
73
|
+
try:
|
|
74
|
+
import msvcrt
|
|
75
|
+
|
|
76
|
+
msvcrt.locking(self._fh.fileno(), msvcrt.LK_LOCK, 1)
|
|
77
|
+
except ImportError:
|
|
78
|
+
import fcntl
|
|
79
|
+
|
|
80
|
+
fcntl.flock(self._fh.fileno(), fcntl.LOCK_EX)
|
|
81
|
+
except BaseException:
|
|
82
|
+
self._plock.release()
|
|
83
|
+
raise
|
|
84
|
+
return self
|
|
85
|
+
|
|
86
|
+
def __exit__(self, *exc) -> bool:
|
|
87
|
+
try:
|
|
88
|
+
if self._fh is not None:
|
|
89
|
+
try:
|
|
90
|
+
import msvcrt
|
|
91
|
+
|
|
92
|
+
self._fh.seek(0)
|
|
93
|
+
msvcrt.locking(self._fh.fileno(), msvcrt.LK_UNLCK, 1)
|
|
94
|
+
except ImportError:
|
|
95
|
+
import fcntl
|
|
96
|
+
|
|
97
|
+
fcntl.flock(self._fh.fileno(), fcntl.LOCK_UN)
|
|
98
|
+
self._fh.close()
|
|
99
|
+
self._fh = None
|
|
100
|
+
finally:
|
|
101
|
+
self._plock.release()
|
|
102
|
+
return False
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class OpLog:
|
|
106
|
+
"""Append-only semantic op log backed by one JSONL file."""
|
|
107
|
+
|
|
108
|
+
def __init__(self, data_root: Optional[Path] = None) -> None:
|
|
109
|
+
self._data_root = data_root or vcs_data_root()
|
|
110
|
+
self.path = self._data_root / "oplog.jsonl"
|
|
111
|
+
self._lock = RLock()
|
|
112
|
+
self._ops: Dict[str, EditOp] = {}
|
|
113
|
+
self._load()
|
|
114
|
+
|
|
115
|
+
# ── persistence ────────────────────────────────────────────────────
|
|
116
|
+
|
|
117
|
+
def _load(self) -> None:
|
|
118
|
+
if not self.path.exists():
|
|
119
|
+
return
|
|
120
|
+
with FileLock(_lock_path(self._data_root)):
|
|
121
|
+
with open(self.path, encoding="utf-8") as f:
|
|
122
|
+
for line in f:
|
|
123
|
+
line = line.strip()
|
|
124
|
+
if not line or line.startswith("#"):
|
|
125
|
+
continue
|
|
126
|
+
op = EditOp.from_dict(json.loads(line))
|
|
127
|
+
self._ops[op.op_id] = op
|
|
128
|
+
|
|
129
|
+
def append(self, op: EditOp) -> None:
|
|
130
|
+
"""Append one op durably. Idempotent by op_id (callers may retry)."""
|
|
131
|
+
with self._lock:
|
|
132
|
+
if op.op_id in self._ops:
|
|
133
|
+
return
|
|
134
|
+
with FileLock(_lock_path(self._data_root)):
|
|
135
|
+
self._data_root.mkdir(parents=True, exist_ok=True)
|
|
136
|
+
fresh = not self.path.exists()
|
|
137
|
+
with open(self.path, "a", encoding="utf-8") as f:
|
|
138
|
+
if fresh:
|
|
139
|
+
f.write(_HEADER + "\n")
|
|
140
|
+
f.write(json.dumps(op.to_dict()) + "\n")
|
|
141
|
+
f.flush()
|
|
142
|
+
os.fsync(f.fileno())
|
|
143
|
+
self._ops[op.op_id] = op
|
|
144
|
+
|
|
145
|
+
# ── queries ────────────────────────────────────────────────────────
|
|
146
|
+
|
|
147
|
+
def all_ops(self) -> List[EditOp]:
|
|
148
|
+
return list(self._ops.values())
|
|
149
|
+
|
|
150
|
+
def ops_for_commit(self, git_sha: str) -> List[EditOp]:
|
|
151
|
+
return [op for op in self._ops.values() if op.git_sha == git_sha]
|
|
152
|
+
|
|
153
|
+
def ops_for_node(self, node_id: str) -> List[EditOp]:
|
|
154
|
+
return [
|
|
155
|
+
op for op in self._ops.values()
|
|
156
|
+
if any(nc.node_id == node_id for nc in op.node_changes)
|
|
157
|
+
]
|
|
158
|
+
|
|
159
|
+
def ops_since(self, ts: str) -> List[EditOp]:
|
|
160
|
+
return [op for op in self._ops.values() if op.ts >= ts]
|
|
161
|
+
|
|
162
|
+
def ops_by(self, actor: str) -> List[EditOp]:
|
|
163
|
+
return [op for op in self._ops.values() if op.actor == actor]
|
|
164
|
+
|
|
165
|
+
def sha_index(self) -> Dict[str, str]:
|
|
166
|
+
return {op.git_sha: op.op_id for op in self._ops.values()}
|
|
167
|
+
|
|
168
|
+
def has_commit(self, git_sha: str) -> bool:
|
|
169
|
+
return any(op.git_sha == git_sha for op in self._ops.values())
|
|
170
|
+
|
|
171
|
+
def get(self, op_id: str) -> Optional[EditOp]:
|
|
172
|
+
return self._ops.get(op_id)
|
|
173
|
+
|
|
174
|
+
def linearize(self, op_ids: Iterable[str]) -> List[EditOp]:
|
|
175
|
+
"""Order ops so causal parents precede their children (stable)."""
|
|
176
|
+
want = {oid for oid in op_ids if oid in self._ops}
|
|
177
|
+
if not want:
|
|
178
|
+
return []
|
|
179
|
+
indeg: Dict[str, int] = {oid: 0 for oid in want}
|
|
180
|
+
children: Dict[str, List[str]] = {oid: [] for oid in want}
|
|
181
|
+
for oid in want:
|
|
182
|
+
for p in self._ops[oid].parent_ops:
|
|
183
|
+
if p in want:
|
|
184
|
+
indeg[oid] += 1
|
|
185
|
+
children.setdefault(p, []).append(oid)
|
|
186
|
+
ready = sorted(oid for oid, d in indeg.items() if d == 0)
|
|
187
|
+
out: List[EditOp] = []
|
|
188
|
+
while ready:
|
|
189
|
+
oid = ready.pop(0)
|
|
190
|
+
out.append(self._ops[oid])
|
|
191
|
+
for c in children[oid]:
|
|
192
|
+
indeg[c] -= 1
|
|
193
|
+
if indeg[c] == 0:
|
|
194
|
+
ready.append(c)
|
|
195
|
+
done = {op.op_id for op in out}
|
|
196
|
+
for oid in sorted(want):
|
|
197
|
+
if oid not in done:
|
|
198
|
+
out.append(self._ops[oid])
|
|
199
|
+
return out
|
|
200
|
+
|
|
201
|
+
def export(self, dest_path: Path) -> int:
|
|
202
|
+
"""Write the full log to ``dest_path`` (for backup / shipping)."""
|
|
203
|
+
dest_path = Path(dest_path)
|
|
204
|
+
dest_path.parent.mkdir(parents=True, exist_ok=True)
|
|
205
|
+
with open(dest_path, "w", encoding="utf-8") as f:
|
|
206
|
+
f.write(_HEADER + "\n")
|
|
207
|
+
for op in self.all_ops():
|
|
208
|
+
f.write(json.dumps(op.to_dict()) + "\n")
|
|
209
|
+
return len(self._ops)
|
|
210
|
+
|
|
211
|
+
def import_log(self, src_path: Path) -> int:
|
|
212
|
+
"""Load ops from an exported log; returns count imported."""
|
|
213
|
+
count = 0
|
|
214
|
+
with open(src_path, encoding="utf-8") as f:
|
|
215
|
+
for line in f:
|
|
216
|
+
line = line.strip()
|
|
217
|
+
if not line or line.startswith("#"):
|
|
218
|
+
continue
|
|
219
|
+
self.append(EditOp.from_dict(json.loads(line)))
|
|
220
|
+
count += 1
|
|
221
|
+
return count
|
awgit/parser.py
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Minimal Python source parser for standalone awgit.
|
|
2
|
+
|
|
3
|
+
Replicates the chunk shape of ``lib.faculties.CodeGraph.parse_source_bytes``
|
|
4
|
+
that awgit's capture/diff/merge consume — a ``ParseResult.chunks`` list whose
|
|
5
|
+
members expose ``name``, ``chunk_type`` (a ``.value`` in
|
|
6
|
+
``function|class|method|module``), ``signature``, ``start_line`` (1-indexed),
|
|
7
|
+
``end_line`` (inclusive). Built on Python's ``ast`` — no AitherOS dependency.
|
|
8
|
+
|
|
9
|
+
Fidelity notes (each verified against the reference so node ids stay stable
|
|
10
|
+
across the monorepo and the standalone package):
|
|
11
|
+
- Method chunks are named ``Class.method``. The stable node id is keyed on
|
|
12
|
+
``(name, path, type)``, so an unqualified method name would mint a different
|
|
13
|
+
id than the reference parser and break cross-repo id stability.
|
|
14
|
+
- The MODULE chunk is emitted only when the file has a docstring or top-level
|
|
15
|
+
constant assignments (the reference skips empty module chunks, so an
|
|
16
|
+
unconditional one would mint a spurious module node on every file and turn
|
|
17
|
+
every capture into a module rewrite). Its range is ``1..1`` like the
|
|
18
|
+
reference.
|
|
19
|
+
- Function/method signatures replicate ``get_signature``: positional args with
|
|
20
|
+
annotations + return annotation, ``async def`` prefix; defaults, ``*args``/
|
|
21
|
+
``**kwargs`` and ``/``/``*`` separators are deliberately omitted (matching
|
|
22
|
+
the reference) so a signature change flags the same way.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import ast
|
|
28
|
+
import logging
|
|
29
|
+
from dataclasses import dataclass, field
|
|
30
|
+
from enum import Enum
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
from typing import List, Optional
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
_MODULE_CONST_LIMIT = 60
|
|
37
|
+
_MODULE_VALUE_CHARS = 120
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ChunkType(Enum):
|
|
41
|
+
MODULE = "module"
|
|
42
|
+
CLASS = "class"
|
|
43
|
+
METHOD = "method"
|
|
44
|
+
FUNCTION = "function"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class Chunk:
|
|
49
|
+
name: str
|
|
50
|
+
chunk_type: ChunkType
|
|
51
|
+
signature: str = ""
|
|
52
|
+
start_line: int = 0
|
|
53
|
+
end_line: int = 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class ParseResult:
|
|
58
|
+
chunks: List[Chunk] = field(default_factory=list)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _get_signature(node: ast.AST) -> str:
|
|
62
|
+
"""Full function signature — mirrors CodeGraph's ``get_signature``."""
|
|
63
|
+
args: List[str] = []
|
|
64
|
+
for arg in node.args.args:
|
|
65
|
+
arg_str = arg.arg
|
|
66
|
+
if arg.annotation:
|
|
67
|
+
try:
|
|
68
|
+
arg_str += f": {ast.unparse(arg.annotation)}"
|
|
69
|
+
except Exception as e: # malformed annotation: keep the bare arg
|
|
70
|
+
logger.debug("awgit parser: annotation unparse failed: %s", e)
|
|
71
|
+
args.append(arg_str)
|
|
72
|
+
returns = ""
|
|
73
|
+
if node.returns:
|
|
74
|
+
try:
|
|
75
|
+
returns = f" -> {ast.unparse(node.returns)}"
|
|
76
|
+
except Exception as e: # malformed return annotation: omit it
|
|
77
|
+
logger.debug("awgit parser: return annotation unparse failed: %s", e)
|
|
78
|
+
prefix = "async def" if isinstance(node, ast.AsyncFunctionDef) else "def"
|
|
79
|
+
return f"{prefix} {node.name}({', '.join(args)}){returns}"
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _module_chunk(tree: ast.Module, path: str) -> Optional[Chunk]:
|
|
83
|
+
"""One chunk at module scope — ONLY when there is something to index."""
|
|
84
|
+
docstring = ast.get_docstring(tree)
|
|
85
|
+
consts: List[str] = []
|
|
86
|
+
for node in ast.iter_child_nodes(tree):
|
|
87
|
+
targets: List[str] = []
|
|
88
|
+
value = None
|
|
89
|
+
if isinstance(node, ast.Assign):
|
|
90
|
+
targets = [t.id for t in node.targets if isinstance(t, ast.Name)]
|
|
91
|
+
value = node.value
|
|
92
|
+
elif isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name):
|
|
93
|
+
targets = [node.target.id]
|
|
94
|
+
value = node.value
|
|
95
|
+
if not targets or value is None:
|
|
96
|
+
continue
|
|
97
|
+
try:
|
|
98
|
+
rendered = ast.unparse(value)
|
|
99
|
+
except Exception as e: # un-unparseable const value: skip it
|
|
100
|
+
logger.debug("awgit parser: const unparse failed: %s", e)
|
|
101
|
+
continue
|
|
102
|
+
if len(rendered) > _MODULE_VALUE_CHARS:
|
|
103
|
+
rendered = rendered[:_MODULE_VALUE_CHARS] + "..."
|
|
104
|
+
for t in targets:
|
|
105
|
+
if t.startswith("__"):
|
|
106
|
+
continue
|
|
107
|
+
consts.append(f"{t} = {rendered}")
|
|
108
|
+
if len(consts) >= _MODULE_CONST_LIMIT:
|
|
109
|
+
break
|
|
110
|
+
if not docstring and not consts:
|
|
111
|
+
return None
|
|
112
|
+
name = Path(path).stem
|
|
113
|
+
return Chunk(
|
|
114
|
+
name=name,
|
|
115
|
+
chunk_type=ChunkType.MODULE,
|
|
116
|
+
signature=f"module {name}",
|
|
117
|
+
start_line=1,
|
|
118
|
+
end_line=1,
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _class_chunk(node: ast.ClassDef) -> Chunk:
|
|
123
|
+
bases: List[str] = []
|
|
124
|
+
for b in node.bases:
|
|
125
|
+
try:
|
|
126
|
+
bases.append(ast.unparse(b))
|
|
127
|
+
except Exception as e: # un-unparseable base: omit it from the header
|
|
128
|
+
logger.debug("awgit parser: class base unparse failed: %s", e)
|
|
129
|
+
signature = f"class {node.name}"
|
|
130
|
+
if bases:
|
|
131
|
+
signature += f"({', '.join(bases)})"
|
|
132
|
+
return Chunk(
|
|
133
|
+
name=node.name,
|
|
134
|
+
chunk_type=ChunkType.CLASS,
|
|
135
|
+
signature=signature,
|
|
136
|
+
start_line=node.lineno,
|
|
137
|
+
end_line=node.end_lineno or node.lineno,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _func_chunk(node: ast.AST, chunk_type: ChunkType) -> Chunk:
|
|
142
|
+
return Chunk(
|
|
143
|
+
name=node.name,
|
|
144
|
+
chunk_type=chunk_type,
|
|
145
|
+
signature=_get_signature(node),
|
|
146
|
+
start_line=node.lineno,
|
|
147
|
+
end_line=node.end_lineno or node.lineno,
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _method_chunk(class_name: str, node: ast.AST) -> Chunk:
|
|
152
|
+
return Chunk(
|
|
153
|
+
name=f"{class_name}.{node.name}",
|
|
154
|
+
chunk_type=ChunkType.METHOD,
|
|
155
|
+
signature=_get_signature(node),
|
|
156
|
+
start_line=node.lineno,
|
|
157
|
+
end_line=node.end_lineno or node.lineno,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def parse_source_bytes(content: bytes, path: str) -> ParseResult:
|
|
162
|
+
"""Parse Python source bytes into chunks (no disk, no AitherOS deps).
|
|
163
|
+
|
|
164
|
+
One chunk per top-level function, class, and class method, plus a module
|
|
165
|
+
chunk when the file carries a docstring or top-level constants. Mirrors the
|
|
166
|
+
reference extractor's traversal depth exactly (methods live one level below
|
|
167
|
+
classes; nested classes/functions are not indexed).
|
|
168
|
+
"""
|
|
169
|
+
source = content.decode("utf-8", errors="ignore")
|
|
170
|
+
try:
|
|
171
|
+
tree = ast.parse(source, filename=path)
|
|
172
|
+
except SyntaxError:
|
|
173
|
+
return ParseResult(chunks=[])
|
|
174
|
+
result = ParseResult()
|
|
175
|
+
module = _module_chunk(tree, path)
|
|
176
|
+
if module is not None:
|
|
177
|
+
result.chunks.append(module)
|
|
178
|
+
for node in tree.body:
|
|
179
|
+
if isinstance(node, ast.ClassDef):
|
|
180
|
+
result.chunks.append(_class_chunk(node))
|
|
181
|
+
for item in node.body:
|
|
182
|
+
if isinstance(item, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
183
|
+
result.chunks.append(_method_chunk(node.name, item))
|
|
184
|
+
elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
185
|
+
result.chunks.append(_func_chunk(node, ChunkType.FUNCTION))
|
|
186
|
+
return result
|
awgit/schema.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Edit-op schema — the contract every semantic-VCS component shares.
|
|
2
|
+
|
|
3
|
+
SCHEMA_VERSION gates ``from_dict``: a record written by a NEWER version is
|
|
4
|
+
refused loudly rather than silently misparsed (a future record is a hard
|
|
5
|
+
error, never a guess).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Dict, List, Optional
|
|
12
|
+
|
|
13
|
+
SCHEMA_VERSION = 1
|
|
14
|
+
|
|
15
|
+
CHANGE_TYPES = (
|
|
16
|
+
"body_rewrite",
|
|
17
|
+
"added",
|
|
18
|
+
"deleted",
|
|
19
|
+
"renamed",
|
|
20
|
+
"moved",
|
|
21
|
+
"signature_changed",
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class NodeChange:
|
|
27
|
+
"""One node (function/class/method) changed by an edit-op."""
|
|
28
|
+
|
|
29
|
+
node_id: str
|
|
30
|
+
change_type: str
|
|
31
|
+
old_body_sha: Optional[str] = None
|
|
32
|
+
new_body_sha: Optional[str] = None
|
|
33
|
+
symbol: str = ""
|
|
34
|
+
path: str = ""
|
|
35
|
+
renamed_from: Optional[str] = None
|
|
36
|
+
moved_from: Optional[str] = None
|
|
37
|
+
semantic_note: str = ""
|
|
38
|
+
|
|
39
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
40
|
+
return {
|
|
41
|
+
"node_id": self.node_id,
|
|
42
|
+
"change_type": self.change_type,
|
|
43
|
+
"old_body_sha": self.old_body_sha,
|
|
44
|
+
"new_body_sha": self.new_body_sha,
|
|
45
|
+
"symbol": self.symbol,
|
|
46
|
+
"path": self.path,
|
|
47
|
+
"renamed_from": self.renamed_from,
|
|
48
|
+
"moved_from": self.moved_from,
|
|
49
|
+
"semantic_note": self.semantic_note,
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
@classmethod
|
|
53
|
+
def from_dict(cls, d: Dict[str, Any]) -> "NodeChange":
|
|
54
|
+
return cls(
|
|
55
|
+
node_id=d["node_id"],
|
|
56
|
+
change_type=d["change_type"],
|
|
57
|
+
old_body_sha=d.get("old_body_sha"),
|
|
58
|
+
new_body_sha=d.get("new_body_sha"),
|
|
59
|
+
symbol=d.get("symbol", ""),
|
|
60
|
+
path=d.get("path", ""),
|
|
61
|
+
renamed_from=d.get("renamed_from"),
|
|
62
|
+
moved_from=d.get("moved_from"),
|
|
63
|
+
semantic_note=d.get("semantic_note", ""),
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass
|
|
68
|
+
class EditOp:
|
|
69
|
+
"""One commit's semantic annotation."""
|
|
70
|
+
|
|
71
|
+
op_id: str
|
|
72
|
+
parent_ops: List[str]
|
|
73
|
+
actor: str
|
|
74
|
+
ts: str
|
|
75
|
+
git_sha: str
|
|
76
|
+
git_parent_sha: str
|
|
77
|
+
file_paths: List[str]
|
|
78
|
+
node_changes: List[NodeChange]
|
|
79
|
+
summary: str = ""
|
|
80
|
+
leased: bool = False
|
|
81
|
+
schema_version: int = SCHEMA_VERSION
|
|
82
|
+
# Actor provenance (added additively — old ops parse with these defaults).
|
|
83
|
+
# `actor` is the CLAIMED attribution (explicit arg / AITHER_ACTOR / git
|
|
84
|
+
# author); `actor_verified` + `verified_actor` record the box's VERIFIED
|
|
85
|
+
# identity (the Aitherium GitHub OAuth app / GitHub App login via `gh`)
|
|
86
|
+
# INDEPENDENTLY, so a session claiming `AITHER_ACTOR=lyra` on a box verified
|
|
87
|
+
# as `wizzense` records both. The verified half is the authoritative
|
|
88
|
+
# attribution; self-asserted actor is forgeable and the op-log now says so
|
|
89
|
+
# explicitly.
|
|
90
|
+
actor_verified: bool = False
|
|
91
|
+
actor_source: str = "env"
|
|
92
|
+
verified_actor: str = ""
|
|
93
|
+
# Durable attribution handle — a stable id for this op that replays/exports
|
|
94
|
+
# can point at. Minted deterministically from (op_id, git_sha) — see
|
|
95
|
+
# ledger.py. The op-log only RECORDS; it never gates a commit.
|
|
96
|
+
ledger_ref: str = ""
|
|
97
|
+
|
|
98
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
99
|
+
return {
|
|
100
|
+
"op_id": self.op_id,
|
|
101
|
+
"parent_ops": list(self.parent_ops),
|
|
102
|
+
"actor": self.actor,
|
|
103
|
+
"ts": self.ts,
|
|
104
|
+
"git_sha": self.git_sha,
|
|
105
|
+
"git_parent_sha": self.git_parent_sha,
|
|
106
|
+
"file_paths": list(self.file_paths),
|
|
107
|
+
"node_changes": [nc.to_dict() for nc in self.node_changes],
|
|
108
|
+
"summary": self.summary,
|
|
109
|
+
"leased": self.leased,
|
|
110
|
+
"schema_version": self.schema_version,
|
|
111
|
+
"actor_verified": self.actor_verified,
|
|
112
|
+
"actor_source": self.actor_source,
|
|
113
|
+
"verified_actor": self.verified_actor,
|
|
114
|
+
"ledger_ref": self.ledger_ref,
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
@classmethod
|
|
118
|
+
def from_dict(cls, d: Dict[str, Any]) -> "EditOp":
|
|
119
|
+
sv = d.get("schema_version", 0)
|
|
120
|
+
if not isinstance(sv, int) or sv > SCHEMA_VERSION:
|
|
121
|
+
raise ValueError(
|
|
122
|
+
f"EditOp schema_version {sv!r} is newer than supported "
|
|
123
|
+
f"{SCHEMA_VERSION} — refusing to misparse. Rebuild the reader "
|
|
124
|
+
"or upgrade the exporter."
|
|
125
|
+
)
|
|
126
|
+
for key in ("op_id", "git_sha", "git_parent_sha"):
|
|
127
|
+
if key not in d:
|
|
128
|
+
raise ValueError(f"EditOp missing required field {key!r}")
|
|
129
|
+
return cls(
|
|
130
|
+
op_id=d["op_id"],
|
|
131
|
+
parent_ops=list(d.get("parent_ops", [])),
|
|
132
|
+
actor=d.get("actor", "unknown"),
|
|
133
|
+
ts=d.get("ts", ""),
|
|
134
|
+
git_sha=d["git_sha"],
|
|
135
|
+
git_parent_sha=d["git_parent_sha"],
|
|
136
|
+
file_paths=list(d.get("file_paths", [])),
|
|
137
|
+
node_changes=[NodeChange.from_dict(x) for x in d.get("node_changes", [])],
|
|
138
|
+
summary=d.get("summary", ""),
|
|
139
|
+
leased=bool(d.get("leased", False)),
|
|
140
|
+
schema_version=sv,
|
|
141
|
+
actor_verified=bool(d.get("actor_verified", False)),
|
|
142
|
+
actor_source=d.get("actor_source", "env"),
|
|
143
|
+
verified_actor=d.get("verified_actor", ""),
|
|
144
|
+
ledger_ref=d.get("ledger_ref", ""),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
@dataclass
|
|
149
|
+
class MergeConflict:
|
|
150
|
+
"""A collision the merge engine could not resolve and escalated."""
|
|
151
|
+
|
|
152
|
+
conflict_id: str
|
|
153
|
+
node_id: str
|
|
154
|
+
symbol: str
|
|
155
|
+
path: str = ""
|
|
156
|
+
base_body: Optional[str] = None
|
|
157
|
+
a_body: Optional[str] = None
|
|
158
|
+
b_body: Optional[str] = None
|
|
159
|
+
blast_radius: Dict[str, Any] = field(default_factory=dict)
|
|
160
|
+
suggested: Optional[str] = None
|
|
161
|
+
status: str = "escalated"
|
|
162
|
+
|
|
163
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
164
|
+
return {
|
|
165
|
+
"conflict_id": self.conflict_id,
|
|
166
|
+
"node_id": self.node_id,
|
|
167
|
+
"symbol": self.symbol,
|
|
168
|
+
"path": self.path,
|
|
169
|
+
"base_body": self.base_body,
|
|
170
|
+
"a_body": self.a_body,
|
|
171
|
+
"b_body": self.b_body,
|
|
172
|
+
"blast_radius": self.blast_radius,
|
|
173
|
+
"suggested": self.suggested,
|
|
174
|
+
"status": self.status,
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
@classmethod
|
|
178
|
+
def from_dict(cls, d: Dict[str, Any]) -> "MergeConflict":
|
|
179
|
+
return cls(
|
|
180
|
+
conflict_id=d["conflict_id"],
|
|
181
|
+
node_id=d["node_id"],
|
|
182
|
+
symbol=d.get("symbol", ""),
|
|
183
|
+
path=d.get("path", ""),
|
|
184
|
+
base_body=d.get("base_body"),
|
|
185
|
+
a_body=d.get("a_body"),
|
|
186
|
+
b_body=d.get("b_body"),
|
|
187
|
+
blast_radius=d.get("blast_radius", {}),
|
|
188
|
+
suggested=d.get("suggested"),
|
|
189
|
+
status=d.get("status", "escalated"),
|
|
190
|
+
)
|