claude-code-sessions 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- claude_code_sessions-0.9.0.dist-info/METADATA +372 -0
- claude_code_sessions-0.9.0.dist-info/RECORD +7 -0
- claude_code_sessions-0.9.0.dist-info/WHEEL +5 -0
- claude_code_sessions-0.9.0.dist-info/entry_points.txt +3 -0
- claude_code_sessions-0.9.0.dist-info/licenses/LICENSE +21 -0
- claude_code_sessions-0.9.0.dist-info/top_level.txt +1 -0
- claude_code_sessions.py +3443 -0
claude_code_sessions.py
ADDED
|
@@ -0,0 +1,3443 @@
|
|
|
1
|
+
"""claude-code-sessions: inspect and relocate Claude Code sessions on disk.
|
|
2
|
+
|
|
3
|
+
Unofficial. Fails closed: verifies the on-disk layout against evidence and
|
|
4
|
+
refuses to mutate anything it cannot positively verify.
|
|
5
|
+
|
|
6
|
+
Sections (in order):
|
|
7
|
+
1. Env, exceptions, constants 5. Transaction engine (move/undo/recover)
|
|
8
|
+
2. Helpers (hashing, atomic IO) 6. Commands (list/doctor/move/undo/recover)
|
|
9
|
+
3. Platform & store discovery, 7. CLI wiring
|
|
10
|
+
rows, encoding detection 8. Sync (cross-account)
|
|
11
|
+
4. Transcript location
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import dataclasses
|
|
16
|
+
import os
|
|
17
|
+
import sys
|
|
18
|
+
|
|
19
|
+
SCHEME_CURRENT = r"[^A-Za-z0-9]" # app >= ~2026-07-12: underscores also become '-'
|
|
20
|
+
SCHEME_LEGACY = r"[^A-Za-z0-9_]" # before: underscores survived
|
|
21
|
+
|
|
22
|
+
NONTERMINAL = ("journaled", "copying", "copied", "rewriting", "committed", "aborting",
|
|
23
|
+
"writing")
|
|
24
|
+
TERMINAL = ("completed", "rolled_back", "undone")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Refusal(Exception):
|
|
28
|
+
exit_code = 1
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class LayoutError(Exception):
|
|
32
|
+
exit_code = 2
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclasses.dataclass
|
|
36
|
+
class Env:
|
|
37
|
+
home: str
|
|
38
|
+
projects_root: str
|
|
39
|
+
store_candidates: list
|
|
40
|
+
ops_dir: str
|
|
41
|
+
moved_log: str
|
|
42
|
+
is_windows: bool
|
|
43
|
+
process_lister: object
|
|
44
|
+
now: object
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def default_env():
|
|
48
|
+
home = os.path.expanduser("~")
|
|
49
|
+
if sys.platform == "win32":
|
|
50
|
+
candidates = sorted(
|
|
51
|
+
__import__("glob").glob(os.path.join(os.environ.get("LOCALAPPDATA", ""),
|
|
52
|
+
"Packages", "Claude_*", "LocalCache", "Roaming", "Claude", "claude-code-sessions"))
|
|
53
|
+
) + [os.path.join(os.environ.get("APPDATA", ""), "Claude", "claude-code-sessions")]
|
|
54
|
+
elif sys.platform == "darwin":
|
|
55
|
+
candidates = [os.path.join(home, "Library", "Application Support", "Claude", "claude-code-sessions")]
|
|
56
|
+
else:
|
|
57
|
+
candidates = [os.path.join(home, ".config", "Claude", "claude-code-sessions")]
|
|
58
|
+
import time
|
|
59
|
+
return Env(
|
|
60
|
+
home=home,
|
|
61
|
+
projects_root=os.path.join(home, ".claude", "projects"),
|
|
62
|
+
store_candidates=candidates,
|
|
63
|
+
ops_dir=os.path.join(home, ".claude-code-journal", "ops"),
|
|
64
|
+
moved_log=os.path.join(home, ".claude-code-journal", "moved-log.jsonl"),
|
|
65
|
+
is_windows=(sys.platform == "win32"),
|
|
66
|
+
process_lister=_default_process_lister,
|
|
67
|
+
now=time.time,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# Fail-closed sentinel: returned (with pid -1) whenever the process list
|
|
72
|
+
# cannot be obtained. Contains "claude" so every guard's substring match
|
|
73
|
+
# treats it as a possibly-running desktop app, and no CLI marker so the
|
|
74
|
+
# narrowing never excuses it. "Couldn't look" is never "nothing there".
|
|
75
|
+
_PROC_UNAVAILABLE = ("(process listing unavailable - treating the claude "
|
|
76
|
+
"desktop app as possibly running)")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _parse_proc_lines(out):
|
|
80
|
+
"""(pid, text) tuples from 'pid|name|path' lines (one process per line).
|
|
81
|
+
|
|
82
|
+
text is the lowercased executable path when the process reports one,
|
|
83
|
+
else the lowercased image name. Malformed lines are skipped - this
|
|
84
|
+
parses our own PowerShell command's output, so anything unexpected is
|
|
85
|
+
noise, not data.
|
|
86
|
+
"""
|
|
87
|
+
result = []
|
|
88
|
+
for line in out.splitlines():
|
|
89
|
+
parts = line.strip().split("|", 2)
|
|
90
|
+
if len(parts) != 3:
|
|
91
|
+
continue
|
|
92
|
+
try:
|
|
93
|
+
pid = int(parts[0])
|
|
94
|
+
except ValueError:
|
|
95
|
+
continue
|
|
96
|
+
text = (parts[2].strip() or parts[1].strip()).lower()
|
|
97
|
+
if text:
|
|
98
|
+
result.append((pid, text))
|
|
99
|
+
return result
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _default_process_lister():
|
|
103
|
+
"""Running processes as (pid, text) tuples, text lowercased.
|
|
104
|
+
|
|
105
|
+
On Windows, text is the full executable path when CIM can supply it
|
|
106
|
+
(needed because BOTH the desktop app and the Claude Code CLI are now
|
|
107
|
+
image name claude.exe - measured 2026-08-02: the MSIX desktop at
|
|
108
|
+
...\\WindowsApps\\Claude_...\\app\\Claude.exe and the CLI, a native
|
|
109
|
+
binary since ~2.x, at ...\\AppData\\Roaming\\Claude\\claude-code\\...\\
|
|
110
|
+
claude.exe. The old docstring's claim that a node-hosted CLI was
|
|
111
|
+
invisible to tasklist is obsolete). If PowerShell/CIM yields nothing
|
|
112
|
+
usable, fall back to name-only tasklist output - callers treat an
|
|
113
|
+
unclassifiable claude-named entry as the desktop app (fail closed).
|
|
114
|
+
Total enumeration failure returns the _PROC_UNAVAILABLE sentinel, never
|
|
115
|
+
[] - "couldn't look" is never "nothing there". POSIX uses
|
|
116
|
+
`ps ... args=` unchanged.
|
|
117
|
+
"""
|
|
118
|
+
import subprocess
|
|
119
|
+
try:
|
|
120
|
+
if sys.platform == "win32":
|
|
121
|
+
try:
|
|
122
|
+
proc = subprocess.run(
|
|
123
|
+
["powershell", "-NoProfile", "-NonInteractive", "-Command",
|
|
124
|
+
"Get-CimInstance Win32_Process | ForEach-Object "
|
|
125
|
+
"{ '{0}|{1}|{2}' -f $_.ProcessId, $_.Name, $_.ExecutablePath }"],
|
|
126
|
+
capture_output=True, text=True, timeout=15)
|
|
127
|
+
if proc.returncode == 0:
|
|
128
|
+
parsed = _parse_proc_lines(proc.stdout)
|
|
129
|
+
if parsed:
|
|
130
|
+
return parsed
|
|
131
|
+
# empty or all-garbage CIM output falls through to tasklist
|
|
132
|
+
except (OSError, subprocess.SubprocessError):
|
|
133
|
+
pass
|
|
134
|
+
proc = subprocess.run(["tasklist", "/FO", "CSV"], capture_output=True,
|
|
135
|
+
text=True, timeout=15)
|
|
136
|
+
if proc.returncode != 0:
|
|
137
|
+
return [(-1, _PROC_UNAVAILABLE)]
|
|
138
|
+
out = proc.stdout
|
|
139
|
+
result = []
|
|
140
|
+
for line in out.splitlines()[1:]:
|
|
141
|
+
if not line.startswith('"'):
|
|
142
|
+
continue
|
|
143
|
+
fields = line.split('","')
|
|
144
|
+
if len(fields) < 2:
|
|
145
|
+
continue
|
|
146
|
+
name = fields[0].strip('"').lower()
|
|
147
|
+
try:
|
|
148
|
+
pid = int(fields[1].strip('"'))
|
|
149
|
+
except ValueError:
|
|
150
|
+
continue
|
|
151
|
+
result.append((pid, name))
|
|
152
|
+
return result if result else [(-1, _PROC_UNAVAILABLE)]
|
|
153
|
+
out = subprocess.run(["ps", "-A", "-o", "pid=,args="], capture_output=True,
|
|
154
|
+
text=True, timeout=15).stdout
|
|
155
|
+
result = []
|
|
156
|
+
for line in out.splitlines():
|
|
157
|
+
line = line.strip()
|
|
158
|
+
if not line:
|
|
159
|
+
continue
|
|
160
|
+
pid_s, _, rest = line.partition(" ")
|
|
161
|
+
try:
|
|
162
|
+
pid = int(pid_s)
|
|
163
|
+
except ValueError:
|
|
164
|
+
continue
|
|
165
|
+
result.append((pid, rest.strip().lower()))
|
|
166
|
+
return result if result else [(-1, _PROC_UNAVAILABLE)]
|
|
167
|
+
except Exception:
|
|
168
|
+
return [(-1, _PROC_UNAVAILABLE)]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# ---------------------------------------------------------------- 2. helpers
|
|
172
|
+
import base64
|
|
173
|
+
import hashlib
|
|
174
|
+
import json
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def sha256_file(path):
|
|
178
|
+
h = hashlib.sha256()
|
|
179
|
+
size = 0
|
|
180
|
+
with open(path, "rb") as fh:
|
|
181
|
+
for chunk in iter(lambda: fh.read(1 << 20), b""):
|
|
182
|
+
h.update(chunk)
|
|
183
|
+
size += len(chunk)
|
|
184
|
+
return h.hexdigest(), size
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def fsync_file(path):
|
|
188
|
+
# Windows FlushFileBuffers requires a write-capable handle; os.O_RDONLY fails with EBADF.
|
|
189
|
+
fd = os.open(path, os.O_RDWR)
|
|
190
|
+
try:
|
|
191
|
+
os.fsync(fd)
|
|
192
|
+
finally:
|
|
193
|
+
os.close(fd)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def atomic_write(path, data):
|
|
197
|
+
tmp = path + ".ct-tmp"
|
|
198
|
+
with open(tmp, "wb") as fh:
|
|
199
|
+
fh.write(data)
|
|
200
|
+
fh.flush()
|
|
201
|
+
os.fsync(fh.fileno())
|
|
202
|
+
os.replace(tmp, path)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def read_json(path):
|
|
206
|
+
try:
|
|
207
|
+
with open(path, encoding="utf-8") as fh:
|
|
208
|
+
return json.load(fh)
|
|
209
|
+
except (OSError, ValueError) as exc:
|
|
210
|
+
raise LayoutError("unreadable JSON at {0}: {1}".format(path, exc))
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def b64(data):
|
|
214
|
+
return base64.b64encode(data).decode("ascii")
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def unb64(s):
|
|
218
|
+
return base64.b64decode(s.encode("ascii"))
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# ------------------------------------------------------- 3. encoding detection
|
|
222
|
+
import re
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def encode(path, scheme):
|
|
226
|
+
return re.sub(scheme, "-", path)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def scheme_evidence(cwds, projects_root):
|
|
230
|
+
cur = leg = 0
|
|
231
|
+
for cwd in set(c for c in cwds if c):
|
|
232
|
+
a, b = encode(cwd, SCHEME_CURRENT), encode(cwd, SCHEME_LEGACY)
|
|
233
|
+
if a == b:
|
|
234
|
+
continue # agreeing paths carry no signal
|
|
235
|
+
cur += os.path.isdir(os.path.join(projects_root, a))
|
|
236
|
+
leg += os.path.isdir(os.path.join(projects_root, b))
|
|
237
|
+
return cur, leg
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def choose_scheme(evidence, target_path):
|
|
241
|
+
cur, leg = evidence
|
|
242
|
+
if cur > leg:
|
|
243
|
+
return SCHEME_CURRENT
|
|
244
|
+
if leg > cur:
|
|
245
|
+
return SCHEME_LEGACY
|
|
246
|
+
# tie (including 0-0): only safe when the choice cannot matter for this target
|
|
247
|
+
if encode(target_path, SCHEME_CURRENT) == encode(target_path, SCHEME_LEGACY):
|
|
248
|
+
return SCHEME_CURRENT
|
|
249
|
+
raise LayoutError(
|
|
250
|
+
"cannot determine the path-encoding scheme (evidence current={0} legacy={1}) "
|
|
251
|
+
"and the target '{2}' encodes differently under the two known schemes. "
|
|
252
|
+
"Refusing to guess.".format(cur, leg, target_path))
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
# --------------------------------------------------------- store discovery
|
|
256
|
+
@dataclasses.dataclass
|
|
257
|
+
class StoreDiscovery:
|
|
258
|
+
status: str # found | absent | error
|
|
259
|
+
roots: list
|
|
260
|
+
detail: str
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def discover_stores(env):
|
|
264
|
+
# FileNotFoundError while looking is PROOF of absence (a machine that never
|
|
265
|
+
# installed the desktop app has no %APPDATA%\Claude parent at all - that is
|
|
266
|
+
# the normal CLI-only case, not an error). Any OTHER OSError means "couldn't
|
|
267
|
+
# look", which is never "nothing there".
|
|
268
|
+
roots, errors, seen = [], [], set()
|
|
269
|
+
for cand in env.store_candidates:
|
|
270
|
+
try:
|
|
271
|
+
os.listdir(cand) # store exists and is enumerable
|
|
272
|
+
real = os.path.realpath(cand)
|
|
273
|
+
if real not in seen:
|
|
274
|
+
seen.add(real)
|
|
275
|
+
roots.append(real)
|
|
276
|
+
continue
|
|
277
|
+
except FileNotFoundError:
|
|
278
|
+
pass # candidate missing; prove the parent
|
|
279
|
+
except OSError as exc:
|
|
280
|
+
errors.append("{0}: {1}".format(cand, exc))
|
|
281
|
+
continue
|
|
282
|
+
parent = os.path.dirname(cand)
|
|
283
|
+
try:
|
|
284
|
+
os.listdir(parent)
|
|
285
|
+
except FileNotFoundError:
|
|
286
|
+
pass # parent absent too: proven absent
|
|
287
|
+
except OSError as exc:
|
|
288
|
+
errors.append("{0}: {1}".format(cand, exc))
|
|
289
|
+
if errors:
|
|
290
|
+
return StoreDiscovery("error", roots, "; ".join(errors))
|
|
291
|
+
if roots:
|
|
292
|
+
return StoreDiscovery("found", roots, "{0} root(s)".format(len(roots)))
|
|
293
|
+
return StoreDiscovery("absent", [], "no store under any known candidate")
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
# ----------------------------------------------------------- listing rows
|
|
297
|
+
import glob as _glob
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
@dataclasses.dataclass
|
|
301
|
+
class Row:
|
|
302
|
+
path: str
|
|
303
|
+
data: dict
|
|
304
|
+
|
|
305
|
+
@property
|
|
306
|
+
def local_id(self):
|
|
307
|
+
return self.data.get("sessionId") or os.path.splitext(os.path.basename(self.path))[0]
|
|
308
|
+
|
|
309
|
+
@property
|
|
310
|
+
def cli_session_id(self):
|
|
311
|
+
return self.data.get("cliSessionId") or ""
|
|
312
|
+
|
|
313
|
+
@property
|
|
314
|
+
def cwd(self):
|
|
315
|
+
return self.data.get("cwd") or ""
|
|
316
|
+
|
|
317
|
+
@property
|
|
318
|
+
def last_activity(self):
|
|
319
|
+
return self.data.get("lastActivityAt") or 0
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def load_rows(roots):
|
|
323
|
+
rows, errors = [], []
|
|
324
|
+
for root in roots:
|
|
325
|
+
for path in sorted(_glob.glob(os.path.join(root, "*", "*", "local_*.json"))):
|
|
326
|
+
try:
|
|
327
|
+
data = read_json(path)
|
|
328
|
+
except LayoutError as exc:
|
|
329
|
+
errors.append(str(exc))
|
|
330
|
+
continue
|
|
331
|
+
# I2: a row file whose top-level JSON is not an object (e.g. a
|
|
332
|
+
# bare list) must not become a Row - every Row property assumes
|
|
333
|
+
# dict.get() and would raise AttributeError, crashing
|
|
334
|
+
# doctor/list/move instead of reporting a clean, fail-closed
|
|
335
|
+
# error.
|
|
336
|
+
if not isinstance(data, dict):
|
|
337
|
+
errors.append("row is not a JSON object: {0}".format(path))
|
|
338
|
+
continue
|
|
339
|
+
rows.append(Row(path, data))
|
|
340
|
+
return rows, errors
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
# ------------------------------------------------------ 4. transcript location
|
|
344
|
+
def find_transcripts(projects_root, session_id):
|
|
345
|
+
hits = []
|
|
346
|
+
try:
|
|
347
|
+
for entry in sorted(os.listdir(projects_root)):
|
|
348
|
+
cand = os.path.join(projects_root, entry, session_id + ".jsonl")
|
|
349
|
+
if os.path.isfile(cand):
|
|
350
|
+
hits.append(cand)
|
|
351
|
+
except FileNotFoundError:
|
|
352
|
+
pass
|
|
353
|
+
return hits
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def iter_transcripts(projects_root):
|
|
357
|
+
out = []
|
|
358
|
+
try:
|
|
359
|
+
for entry in sorted(os.listdir(projects_root)):
|
|
360
|
+
folder = os.path.join(projects_root, entry)
|
|
361
|
+
if not os.path.isdir(folder):
|
|
362
|
+
continue
|
|
363
|
+
for name in sorted(os.listdir(folder)):
|
|
364
|
+
if name.endswith(".jsonl"):
|
|
365
|
+
out.append((entry, os.path.join(folder, name)))
|
|
366
|
+
except FileNotFoundError:
|
|
367
|
+
pass
|
|
368
|
+
return out
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _cwds_in(transcript_path):
|
|
372
|
+
vals = []
|
|
373
|
+
try:
|
|
374
|
+
with open(transcript_path, encoding="utf-8", errors="replace") as fh:
|
|
375
|
+
for line in fh:
|
|
376
|
+
try:
|
|
377
|
+
obj = json.loads(line)
|
|
378
|
+
except ValueError:
|
|
379
|
+
continue
|
|
380
|
+
if isinstance(obj, dict) and obj.get("cwd"):
|
|
381
|
+
vals.append(obj["cwd"])
|
|
382
|
+
except OSError:
|
|
383
|
+
pass
|
|
384
|
+
return vals
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def first_cwd(transcript_path):
|
|
388
|
+
vals = _cwds_in(transcript_path)
|
|
389
|
+
return vals[0] if vals else ""
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def last_cwd(transcript_path):
|
|
393
|
+
vals = _cwds_in(transcript_path)
|
|
394
|
+
return vals[-1] if vals else ""
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def sidecar_path(transcript_path):
|
|
398
|
+
return transcript_path[:-len(".jsonl")]
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
# ---------------------------------------------- 5. transaction engine: journal
|
|
402
|
+
import time
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
@dataclasses.dataclass
|
|
406
|
+
class Op:
|
|
407
|
+
op_dir: str
|
|
408
|
+
manifest: dict
|
|
409
|
+
now: object = time.time
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def manifest_path(op):
|
|
413
|
+
return os.path.join(op.op_dir, "manifest.json")
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def save_manifest(op):
|
|
417
|
+
atomic_write(manifest_path(op), json.dumps(op.manifest, indent=1).encode("utf-8"))
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def new_op(env, manifest):
|
|
421
|
+
op_id = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime(env.now())) + "-" + os.urandom(3).hex()
|
|
422
|
+
op_dir = os.path.join(env.ops_dir, op_id)
|
|
423
|
+
os.makedirs(op_dir)
|
|
424
|
+
manifest = dict(manifest)
|
|
425
|
+
manifest["op_id"] = op_id
|
|
426
|
+
manifest["status"] = "journaled"
|
|
427
|
+
manifest["history"] = [{"status": "journaled", "at": env.now()}]
|
|
428
|
+
op = Op(op_dir, manifest, env.now)
|
|
429
|
+
save_manifest(op)
|
|
430
|
+
return op
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def set_status(op, status):
|
|
434
|
+
op.manifest["status"] = status
|
|
435
|
+
op.manifest["history"].append({"status": status, "at": op.now()})
|
|
436
|
+
save_manifest(op)
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def list_ops(env):
|
|
440
|
+
out = []
|
|
441
|
+
if not os.path.isdir(env.ops_dir):
|
|
442
|
+
return out
|
|
443
|
+
ops = []
|
|
444
|
+
for name in os.listdir(env.ops_dir):
|
|
445
|
+
mp = os.path.join(env.ops_dir, name, "manifest.json")
|
|
446
|
+
if os.path.isfile(mp):
|
|
447
|
+
m = read_json(mp)
|
|
448
|
+
ops.append((m, Op(os.path.join(env.ops_dir, name), m)))
|
|
449
|
+
# Sort by creation time (history[0]["at"]) then op_id for stability
|
|
450
|
+
for m, op in sorted(ops, key=lambda x: (x[0].get("history", [{}])[0].get("at", 0), x[0]["op_id"])):
|
|
451
|
+
out.append(op)
|
|
452
|
+
return out
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def nonterminal_ops(env):
|
|
456
|
+
return [o for o in list_ops(env) if o.manifest.get("status") in NONTERMINAL]
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def rotate_ops(env):
|
|
460
|
+
import shutil
|
|
461
|
+
terminal = [o for o in list_ops(env) if o.manifest.get("status") in TERMINAL]
|
|
462
|
+
pruned = []
|
|
463
|
+
for op in terminal[:-10]:
|
|
464
|
+
try:
|
|
465
|
+
shutil.rmtree(op.op_dir)
|
|
466
|
+
pruned.append(op.manifest["op_id"])
|
|
467
|
+
except OSError:
|
|
468
|
+
pass
|
|
469
|
+
return pruned
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
LOCK_NAME = "lock"
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def _lock_path(env):
|
|
476
|
+
return os.path.join(env.ops_dir, LOCK_NAME)
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def acquire_lock(env, op_id):
|
|
480
|
+
os.makedirs(env.ops_dir, exist_ok=True)
|
|
481
|
+
try:
|
|
482
|
+
fd = os.open(_lock_path(env), os.O_CREAT | os.O_EXCL | os.O_WRONLY)
|
|
483
|
+
except FileExistsError:
|
|
484
|
+
holder = read_lock(env)
|
|
485
|
+
if holder:
|
|
486
|
+
holder_str = "pid {0}, op {1}".format(*holder)
|
|
487
|
+
else:
|
|
488
|
+
holder_str = "unknown holder"
|
|
489
|
+
raise Refusal("another claude-code-sessions operation holds the lock ({0}). "
|
|
490
|
+
"If it is dead, run: claude-code-sessions recover".format(holder_str))
|
|
491
|
+
with os.fdopen(fd, "w") as fh:
|
|
492
|
+
fh.write("{0} {1}".format(os.getpid(), op_id))
|
|
493
|
+
return _lock_path(env)
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def release_lock(env):
|
|
497
|
+
try:
|
|
498
|
+
os.unlink(_lock_path(env))
|
|
499
|
+
except FileNotFoundError:
|
|
500
|
+
pass
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def read_lock(env):
|
|
504
|
+
try:
|
|
505
|
+
with open(_lock_path(env)) as fh:
|
|
506
|
+
pid_s, _, op_id = fh.read().partition(" ")
|
|
507
|
+
return int(pid_s), op_id
|
|
508
|
+
except (OSError, ValueError):
|
|
509
|
+
return None
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def lock_is_stale(env):
|
|
513
|
+
info = read_lock(env)
|
|
514
|
+
if info is None:
|
|
515
|
+
return False
|
|
516
|
+
pid = info[0]
|
|
517
|
+
try:
|
|
518
|
+
os.kill(pid, 0)
|
|
519
|
+
return False
|
|
520
|
+
except OSError:
|
|
521
|
+
return True
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def append_moved_log(env, entry):
|
|
525
|
+
os.makedirs(os.path.dirname(env.moved_log), exist_ok=True)
|
|
526
|
+
with open(env.moved_log, "a", encoding="utf-8") as fh:
|
|
527
|
+
fh.write(json.dumps(entry) + "\n")
|
|
528
|
+
|
|
529
|
+
|
|
530
|
+
def moved_session_ids(env):
|
|
531
|
+
state = {}
|
|
532
|
+
try:
|
|
533
|
+
with open(env.moved_log, encoding="utf-8") as fh:
|
|
534
|
+
for line in fh:
|
|
535
|
+
try:
|
|
536
|
+
e = json.loads(line)
|
|
537
|
+
except ValueError:
|
|
538
|
+
continue
|
|
539
|
+
state[e.get("session_id")] = e.get("kind")
|
|
540
|
+
except FileNotFoundError:
|
|
541
|
+
return set()
|
|
542
|
+
except OSError as exc:
|
|
543
|
+
raise LayoutError("cannot read moved-log at {0}: {1}".format(env.moved_log, exc))
|
|
544
|
+
return {sid for sid, kind in state.items() if kind == "move"}
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
# ------------------------------------------- containment & sidecar inventory
|
|
548
|
+
import stat as _stat
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def ensure_contained(path, allowed_roots):
|
|
552
|
+
real = os.path.realpath(path)
|
|
553
|
+
for root in allowed_roots:
|
|
554
|
+
rreal = os.path.realpath(root)
|
|
555
|
+
if real == rreal or real.startswith(rreal + os.sep):
|
|
556
|
+
return real
|
|
557
|
+
raise LayoutError("path {0} resolves outside every recognized root".format(path))
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
def _is_reparse(path):
|
|
561
|
+
if os.path.islink(path):
|
|
562
|
+
return True
|
|
563
|
+
try:
|
|
564
|
+
st = os.lstat(path)
|
|
565
|
+
return bool(getattr(st, "st_file_attributes", 0) &
|
|
566
|
+
getattr(_stat, "FILE_ATTRIBUTE_REPARSE_POINT", 0))
|
|
567
|
+
except OSError:
|
|
568
|
+
return False
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def sidecar_inventory(sidecar_dir):
|
|
572
|
+
if not os.path.isdir(sidecar_dir):
|
|
573
|
+
return []
|
|
574
|
+
if _is_reparse(sidecar_dir):
|
|
575
|
+
raise Refusal("sidecar dir {0} is a symlink/junction; refusing to "
|
|
576
|
+
"traverse".format(sidecar_dir))
|
|
577
|
+
inv = []
|
|
578
|
+
for dirpath, dirnames, filenames in os.walk(sidecar_dir):
|
|
579
|
+
for name in dirnames + filenames:
|
|
580
|
+
full = os.path.join(dirpath, name)
|
|
581
|
+
if _is_reparse(full):
|
|
582
|
+
raise Refusal("symlink/junction inside sidecar tree at {0}; refusing "
|
|
583
|
+
"to traverse".format(full))
|
|
584
|
+
for name in filenames:
|
|
585
|
+
full = os.path.join(dirpath, name)
|
|
586
|
+
digest, size = sha256_file(full)
|
|
587
|
+
rel = os.path.relpath(full, sidecar_dir).replace(os.sep, "/")
|
|
588
|
+
inv.append({"rel": rel, "sha256": digest, "size": size})
|
|
589
|
+
inv.sort(key=lambda e: e["rel"])
|
|
590
|
+
return inv
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
# ------------------------------------------------------ move validation
|
|
594
|
+
@dataclasses.dataclass
|
|
595
|
+
class MoveFlags:
|
|
596
|
+
transcript_only: bool = False
|
|
597
|
+
row: list = ()
|
|
598
|
+
yes: bool = False
|
|
599
|
+
force: bool = False
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
# Our own console-script names. They contain "claude", so without this the
|
|
603
|
+
# process guard below would see this very tool and refuse to run.
|
|
604
|
+
OUR_COMMANDS = ("claude-code-sessions", "ccs")
|
|
605
|
+
|
|
606
|
+
|
|
607
|
+
def _is_cli_process(text):
|
|
608
|
+
"""True when TEXT (a lowercased lister entry) is the Claude Code CLI,
|
|
609
|
+
which must NOT count as the desktop app.
|
|
610
|
+
|
|
611
|
+
Recognised CLI locations (measured 2026-08-02):
|
|
612
|
+
...\\appdata\\roaming\\claude\\claude-code\\<ver>\\claude.exe (versioned binary;
|
|
613
|
+
also the backend the desktop spawns - harmless to exclude, because
|
|
614
|
+
it only exists while the desktop's own MSIX processes are running
|
|
615
|
+
and those still match)
|
|
616
|
+
...\\.local\\bin\\claude.exe / .../.local/bin/claude (the PATH shim)
|
|
617
|
+
any npm-style .../claude-code/... install
|
|
618
|
+
|
|
619
|
+
Everything else claude-named - the MSIX desktop, a non-MSIX desktop
|
|
620
|
+
install, or a bare image name the fallback lister could not resolve to
|
|
621
|
+
a path - stays a match: the guard fails closed on ambiguity.
|
|
622
|
+
Separators are normalised first so a forward-slash Windows path cannot
|
|
623
|
+
dodge the backslash patterns. The markers are precise path SEGMENTS,
|
|
624
|
+
not substrings: a bare 'claude-code' substring test would excuse a
|
|
625
|
+
desktop app installed under an unlucky parent directory (say, a user
|
|
626
|
+
account literally named claude-code), silently disabling the guard.
|
|
627
|
+
|
|
628
|
+
POSIX `ps -A -o args=` reports the full command line, not just argv0, so
|
|
629
|
+
a shimmed invocation carries trailing arguments (".../.local/bin/claude
|
|
630
|
+
--resume x") and never matches an ENDS-WITH check. The shim marker is
|
|
631
|
+
therefore also checked as a CONTAINS match when followed by a space -
|
|
632
|
+
additive, every ENDS-WITH marker above still applies unchanged. A bare
|
|
633
|
+
argv0 "claude" with no path is deliberately left unclassifiable: with no
|
|
634
|
+
path segment to test, there is nothing to safely exclude, so it stays a
|
|
635
|
+
match (fail-safe).
|
|
636
|
+
"""
|
|
637
|
+
text = text.replace("/", "\\")
|
|
638
|
+
return ("\\appdata\\roaming\\claude\\claude-code\\" in text # measured CLI home
|
|
639
|
+
or "\\@anthropic-ai\\claude-code\\" in text # npm install layout
|
|
640
|
+
or text.endswith("\\.local\\bin\\claude.exe")
|
|
641
|
+
or text.endswith("\\.local\\bin\\claude") # POSIX shim, post-normalise
|
|
642
|
+
or "\\.local\\bin\\claude.exe " in text # POSIX shim, argv w/ args
|
|
643
|
+
or "\\.local\\bin\\claude " in text)
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
def claude_running(env):
|
|
647
|
+
my_pids = {os.getpid(), os.getppid()}
|
|
648
|
+
try:
|
|
649
|
+
procs = env.process_lister()
|
|
650
|
+
except Exception:
|
|
651
|
+
procs = [(-1, _PROC_UNAVAILABLE)] # couldn't look != nothing there
|
|
652
|
+
out = []
|
|
653
|
+
for pid, text in procs:
|
|
654
|
+
if pid in my_pids:
|
|
655
|
+
continue # never self-refuse on our own process
|
|
656
|
+
if any(name in text for name in OUR_COMMANDS):
|
|
657
|
+
continue # nor on another instance of this tool
|
|
658
|
+
if _is_cli_process(text):
|
|
659
|
+
continue # the Claude Code CLI, not the desktop app
|
|
660
|
+
if "claude" in text:
|
|
661
|
+
out.append(text)
|
|
662
|
+
return out
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
MTIME_GUARD_SECONDS = 600
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def plan_move(env, session_id, target, flags):
|
|
669
|
+
target = os.path.normpath(os.path.abspath(target))
|
|
670
|
+
|
|
671
|
+
# 1. store discovery / platform posture
|
|
672
|
+
disc = discover_stores(env)
|
|
673
|
+
if disc.status == "error":
|
|
674
|
+
raise LayoutError("store discovery failed: {0}. 'Couldn't look' is never "
|
|
675
|
+
"'nothing there' - refusing to mutate.".format(disc.detail))
|
|
676
|
+
rows, row_errors = load_rows(disc.roots)
|
|
677
|
+
if row_errors:
|
|
678
|
+
raise LayoutError("unreadable listing rows (fail-closed): " + "; ".join(row_errors))
|
|
679
|
+
|
|
680
|
+
# 2. encoding + destination folder (computed before transcript lookup: a
|
|
681
|
+
# transcript that already exists exactly AT the computed destination is a
|
|
682
|
+
# destination collision, not an ambiguous source - see step 3)
|
|
683
|
+
moved = moved_session_ids(env)
|
|
684
|
+
if rows:
|
|
685
|
+
cwds = [r.cwd for r in sorted(rows, key=lambda r: r.last_activity)[-50:]]
|
|
686
|
+
else:
|
|
687
|
+
cwds = []
|
|
688
|
+
for folder, path in iter_transcripts(env.projects_root):
|
|
689
|
+
sid = os.path.splitext(os.path.basename(path))[0]
|
|
690
|
+
if sid in moved:
|
|
691
|
+
continue
|
|
692
|
+
c = first_cwd(path)
|
|
693
|
+
if not c:
|
|
694
|
+
continue
|
|
695
|
+
enc_c, enc_l = encode(c, SCHEME_CURRENT), encode(c, SCHEME_LEGACY)
|
|
696
|
+
if folder not in (enc_c, enc_l):
|
|
697
|
+
continue # worktree session: folder matches neither
|
|
698
|
+
cwds.append(c)
|
|
699
|
+
scheme = choose_scheme(scheme_evidence(cwds, env.projects_root), target)
|
|
700
|
+
dest_dir = os.path.join(env.projects_root, encode(target, scheme))
|
|
701
|
+
dest_transcript = os.path.join(dest_dir, session_id + ".jsonl")
|
|
702
|
+
|
|
703
|
+
# 3. transcript location, globally. A hit whose real path is exactly the
|
|
704
|
+
# computed destination transcript is not source-ambiguity - it is handled
|
|
705
|
+
# by the destination-exists check in step 4 - so it is excluded here.
|
|
706
|
+
hits = find_transcripts(env.projects_root, session_id)
|
|
707
|
+
dest_real = os.path.realpath(dest_transcript)
|
|
708
|
+
source_hits = [h for h in hits if os.path.realpath(h) != dest_real]
|
|
709
|
+
if not source_hits:
|
|
710
|
+
if hits:
|
|
711
|
+
raise Refusal("source and destination transcript are identical: "
|
|
712
|
+
"{0}".format(dest_transcript))
|
|
713
|
+
raise Refusal("No transcript found for {0}. Use 'claude-code-sessions list' to find "
|
|
714
|
+
"session ids.".format(session_id))
|
|
715
|
+
if len(source_hits) > 1:
|
|
716
|
+
raise Refusal("Ambiguous: transcript exists in several folders:\n " +
|
|
717
|
+
"\n ".join(source_hits))
|
|
718
|
+
source = source_hits[0]
|
|
719
|
+
|
|
720
|
+
# 4. destination checks
|
|
721
|
+
if not os.path.isdir(target):
|
|
722
|
+
raise Refusal("target must be an existing directory: {0}".format(target))
|
|
723
|
+
real_target = os.path.normcase(os.path.realpath(target))
|
|
724
|
+
for forbidden in (os.path.join(env.home, ".claude"), os.path.dirname(env.ops_dir)):
|
|
725
|
+
# normcase both sides: on a first run ~/.claude-code-journal does not exist
|
|
726
|
+
# yet, so realpath alone does not canonicalize case on Windows.
|
|
727
|
+
fr = os.path.normcase(os.path.realpath(forbidden))
|
|
728
|
+
if real_target == fr or real_target.startswith(fr + os.sep):
|
|
729
|
+
raise Refusal("refusing target inside {0}".format(forbidden))
|
|
730
|
+
if os.path.exists(dest_transcript) or os.path.exists(sidecar_path(dest_transcript)):
|
|
731
|
+
raise Refusal("destination already exists: {0}".format(dest_transcript))
|
|
732
|
+
if os.path.realpath(os.path.dirname(source)) == os.path.realpath(dest_dir):
|
|
733
|
+
raise Refusal("source and destination are the same folder")
|
|
734
|
+
if os.path.isdir(dest_dir):
|
|
735
|
+
for name in sorted(os.listdir(dest_dir)):
|
|
736
|
+
if not name.endswith(".jsonl"):
|
|
737
|
+
continue
|
|
738
|
+
other = os.path.join(dest_dir, name)
|
|
739
|
+
try:
|
|
740
|
+
with open(other, "rb"):
|
|
741
|
+
pass
|
|
742
|
+
except OSError as exc:
|
|
743
|
+
raise Refusal("cannot read {0} for the destination collision scan "
|
|
744
|
+
"(fail-closed): {1}".format(other, exc))
|
|
745
|
+
sid = name[:-len(".jsonl")]
|
|
746
|
+
if sid in moved:
|
|
747
|
+
continue
|
|
748
|
+
c = last_cwd(other)
|
|
749
|
+
if not c:
|
|
750
|
+
raise Refusal("destination collision: {0} has no recorded cwd; cannot "
|
|
751
|
+
"verify it belongs to this project - refusing to merge "
|
|
752
|
+
"(ambiguous, fail-closed).".format(other))
|
|
753
|
+
if os.path.normcase(os.path.normpath(c)) != os.path.normcase(os.path.normpath(target)):
|
|
754
|
+
raise Refusal("destination collision: {0} records cwd {1}, which is a "
|
|
755
|
+
"different real path than {2}. Two real paths can share "
|
|
756
|
+
"one encoded folder; refusing to merge projects."
|
|
757
|
+
.format(other, c, target))
|
|
758
|
+
import shutil as _shutil
|
|
759
|
+
t_hash, t_size = sha256_file(source)
|
|
760
|
+
side_src = sidecar_path(source)
|
|
761
|
+
inv = sidecar_inventory(side_src) if os.path.isdir(side_src) else []
|
|
762
|
+
need = t_size + sum(e["size"] for e in inv) + (1 << 20)
|
|
763
|
+
if _shutil.disk_usage(os.path.dirname(dest_dir)).free < need:
|
|
764
|
+
raise Refusal("not enough free space for a safe copy")
|
|
765
|
+
|
|
766
|
+
# 5. row set
|
|
767
|
+
my_rows = [r for r in rows if r.cli_session_id == session_id]
|
|
768
|
+
for local_id in (flags.row or ()):
|
|
769
|
+
lid = local_id if local_id.startswith("local_") else "local_" + local_id
|
|
770
|
+
# listing rows are per-account COPIES: the same local id can legitimately
|
|
771
|
+
# appear once per store (e.g. one desktop app, two org/account stores),
|
|
772
|
+
# so every matching row - not just the first found - must be adopted.
|
|
773
|
+
matches = [r for r in rows if r.local_id == lid]
|
|
774
|
+
if not matches:
|
|
775
|
+
raise Refusal("no listing row with sessionId " + lid)
|
|
776
|
+
if any(r.cli_session_id not in ("", session_id) for r in matches):
|
|
777
|
+
raise Refusal("row {0} is linked to a different live session; rows linked "
|
|
778
|
+
"to a different live session are never adoptable".format(lid))
|
|
779
|
+
if not flags.yes:
|
|
780
|
+
first = matches[0]
|
|
781
|
+
raise Refusal("adopting row {0} (title={1!r}, cwd={2!r}, "
|
|
782
|
+
"lastActivityAt={3!r}) requires confirmation: pass --yes"
|
|
783
|
+
.format(lid, first.data.get("title"), first.cwd,
|
|
784
|
+
first.data.get("lastActivityAt")))
|
|
785
|
+
for r in matches:
|
|
786
|
+
if r not in my_rows:
|
|
787
|
+
my_rows.append(r)
|
|
788
|
+
if not my_rows:
|
|
789
|
+
if disc.status == "found" and not flags.transcript_only:
|
|
790
|
+
raise Refusal("no listing row references this transcript; moving it would "
|
|
791
|
+
"orphan the desktop entry. If this session was created by the "
|
|
792
|
+
"CLI (not the desktop app), pass --transcript-only.")
|
|
793
|
+
if disc.status == "absent" and not flags.transcript_only:
|
|
794
|
+
raise Refusal("no desktop store found. If you don't use the desktop app, "
|
|
795
|
+
"pass --transcript-only. (On mac/Linux the store locations "
|
|
796
|
+
"are unverified - absence may mean we looked in the wrong "
|
|
797
|
+
"place.)")
|
|
798
|
+
mode = "desktop" if my_rows else "transcript_only"
|
|
799
|
+
if mode == "desktop":
|
|
800
|
+
_require_verified_platform(env, "mutate")
|
|
801
|
+
|
|
802
|
+
# 6. guards
|
|
803
|
+
running = claude_running(env)
|
|
804
|
+
if running:
|
|
805
|
+
raise Refusal("Claude appears to be running ({0}). Close the app, then retry."
|
|
806
|
+
.format(", ".join(sorted(set(running))[:3])))
|
|
807
|
+
age = env.now() - os.path.getmtime(source)
|
|
808
|
+
if age < MTIME_GUARD_SECONDS and not flags.force:
|
|
809
|
+
raise Refusal("transcript was written {0:.0f} seconds ago - this session may be "
|
|
810
|
+
"open (checked because a recent mtime lasts ~10 minutes). Close "
|
|
811
|
+
"the app; pass --force only if you are sure this is stale."
|
|
812
|
+
.format(age))
|
|
813
|
+
|
|
814
|
+
row_entries = []
|
|
815
|
+
for r in my_rows:
|
|
816
|
+
with open(r.path, "rb") as fh:
|
|
817
|
+
pre = fh.read()
|
|
818
|
+
post = dict(r.data)
|
|
819
|
+
post["cwd"] = target
|
|
820
|
+
post["originCwd"] = target
|
|
821
|
+
post["cliSessionId"] = session_id
|
|
822
|
+
row_entries.append({"path": r.path, "pre_b64": b64(pre),
|
|
823
|
+
"post_b64": b64(json.dumps(post, separators=(",", ":"))
|
|
824
|
+
.encode("utf-8")),
|
|
825
|
+
"rewritten": False})
|
|
826
|
+
return {
|
|
827
|
+
"op_type": "move", "session_id": session_id, "mode": mode,
|
|
828
|
+
"source_transcript": source, "dest_transcript": dest_transcript,
|
|
829
|
+
"transcript_sha256": t_hash, "transcript_size": t_size,
|
|
830
|
+
"sidecar_source": side_src if inv else None,
|
|
831
|
+
"sidecar_dest": sidecar_path(dest_transcript) if inv else None,
|
|
832
|
+
"sidecar_inventory": inv, "rows": row_entries, "target_cwd": target,
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
# ------------------------------------------------------ engine execution
|
|
837
|
+
_crash_hook = None
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
def _maybe_crash(point):
|
|
841
|
+
if _crash_hook is not None:
|
|
842
|
+
_crash_hook(point)
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def _engine_roots(env, manifest):
|
|
846
|
+
"""Allowed containment roots for listing-row paths.
|
|
847
|
+
|
|
848
|
+
Store roots come from `discover_stores`, never from the row path being
|
|
849
|
+
checked itself - deriving a row's "allowed root" from that same row's
|
|
850
|
+
path makes the containment check vacuous (it can never fail).
|
|
851
|
+
"""
|
|
852
|
+
roots = [os.path.dirname(env.ops_dir)]
|
|
853
|
+
if manifest.get("rows"):
|
|
854
|
+
disc = discover_stores(env)
|
|
855
|
+
if disc.status != "found":
|
|
856
|
+
raise LayoutError(
|
|
857
|
+
"cannot verify listing-row containment: store discovery status is "
|
|
858
|
+
"'{0}', not 'found'".format(disc.status))
|
|
859
|
+
roots.extend(disc.roots)
|
|
860
|
+
return roots
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
def _validate_sidecar_rel(rel):
|
|
864
|
+
if os.path.isabs(rel) or "\\" in rel or any(part == ".." for part in rel.split("/")):
|
|
865
|
+
raise LayoutError("unsafe sidecar rel path in manifest: {0!r}".format(rel))
|
|
866
|
+
|
|
867
|
+
|
|
868
|
+
def _delete_inventoried_files(root_dir, inventory):
|
|
869
|
+
"""Delete each inventoried file; return a list of (path, exc) for any
|
|
870
|
+
that could not be removed instead of swallowing the error - a caller
|
|
871
|
+
that silently ignores a failed delete here would let the file be
|
|
872
|
+
orphaned with no journal trail once the source is gone."""
|
|
873
|
+
failures = []
|
|
874
|
+
for e in inventory:
|
|
875
|
+
full = os.path.join(root_dir, *e["rel"].split("/"))
|
|
876
|
+
try:
|
|
877
|
+
os.unlink(full)
|
|
878
|
+
except OSError as exc:
|
|
879
|
+
failures.append((full, exc))
|
|
880
|
+
return failures
|
|
881
|
+
|
|
882
|
+
|
|
883
|
+
def _rmdirs_bottom_up(root_dir):
|
|
884
|
+
dirs = []
|
|
885
|
+
for dirpath, dirnames, filenames in os.walk(root_dir):
|
|
886
|
+
dirs.append(dirpath)
|
|
887
|
+
for d in sorted(dirs, key=len, reverse=True):
|
|
888
|
+
try:
|
|
889
|
+
os.rmdir(d)
|
|
890
|
+
except OSError:
|
|
891
|
+
pass
|
|
892
|
+
|
|
893
|
+
|
|
894
|
+
def _copy_file(src, dst):
|
|
895
|
+
os.makedirs(os.path.dirname(dst), exist_ok=True)
|
|
896
|
+
fd = os.open(dst, os.O_CREAT | os.O_EXCL | os.O_WRONLY) # exclusive create
|
|
897
|
+
with os.fdopen(fd, "wb") as out, open(src, "rb") as inp:
|
|
898
|
+
while True:
|
|
899
|
+
chunk = inp.read(1 << 20)
|
|
900
|
+
if not chunk:
|
|
901
|
+
break
|
|
902
|
+
out.write(chunk)
|
|
903
|
+
out.flush()
|
|
904
|
+
os.fsync(out.fileno())
|
|
905
|
+
|
|
906
|
+
|
|
907
|
+
def _dest_files(manifest):
|
|
908
|
+
files = [(manifest["dest_transcript"], manifest["transcript_sha256"],
|
|
909
|
+
manifest["transcript_size"])]
|
|
910
|
+
for e in manifest["sidecar_inventory"]:
|
|
911
|
+
files.append((os.path.join(manifest["sidecar_dest"], *e["rel"].split("/")),
|
|
912
|
+
e["sha256"], e["size"]))
|
|
913
|
+
return files
|
|
914
|
+
|
|
915
|
+
|
|
916
|
+
def _verify(path_hash_size_list):
|
|
917
|
+
for path, digest, size in path_hash_size_list:
|
|
918
|
+
got, gsize = sha256_file(path)
|
|
919
|
+
if got != digest or gsize != size:
|
|
920
|
+
return path
|
|
921
|
+
return None
|
|
922
|
+
|
|
923
|
+
|
|
924
|
+
def _row_state(row):
|
|
925
|
+
"""Classify a manifest row's CURRENT on-disk bytes against its journaled
|
|
926
|
+
pre/post images - a crash can land between os.replace and save_manifest,
|
|
927
|
+
so the "rewritten" flag alone can never be trusted; only the bytes can.
|
|
928
|
+
Returns "post" (needs no roll-forward, may need roll-back), "pre"
|
|
929
|
+
(untouched / already rolled back), or "drifted" (neither - some other
|
|
930
|
+
process wrote to it, or it is missing; never auto-resolved).
|
|
931
|
+
"""
|
|
932
|
+
pre = unb64(row["pre_b64"])
|
|
933
|
+
post = unb64(row["post_b64"])
|
|
934
|
+
try:
|
|
935
|
+
with open(row["path"], "rb") as fh:
|
|
936
|
+
current = fh.read()
|
|
937
|
+
except OSError:
|
|
938
|
+
current = None
|
|
939
|
+
if current == post:
|
|
940
|
+
return "post"
|
|
941
|
+
if current == pre:
|
|
942
|
+
return "pre"
|
|
943
|
+
return "drifted"
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
def _pre_abort_status(op):
|
|
947
|
+
"""The phase the op was in before it started (or resumed) aborting -
|
|
948
|
+
used to decide whether destination files are still tool-owned scratch
|
|
949
|
+
(I3). Trusting op.manifest["status"] directly breaks the moment a crash
|
|
950
|
+
interrupts an abort itself and recover re-enters _abort: by then status
|
|
951
|
+
already reads "aborting", which carries no information about the
|
|
952
|
+
original phase. History is durable and append-only, so walk it
|
|
953
|
+
backwards past every "aborting" entry to find the real one.
|
|
954
|
+
"""
|
|
955
|
+
for entry in reversed(op.manifest.get("history", [])):
|
|
956
|
+
if entry.get("status") != "aborting":
|
|
957
|
+
return entry.get("status")
|
|
958
|
+
return op.manifest["status"]
|
|
959
|
+
|
|
960
|
+
|
|
961
|
+
def _source_pre_verified(m):
|
|
962
|
+
"""C1(a): True iff the source transcript AND every inventoried sidecar
|
|
963
|
+
file are currently present and byte-identical to what was journaled as
|
|
964
|
+
the pre-state. Gates non-scratch (hash-gated) destination deletion
|
|
965
|
+
during abort - deleting a hash-verified destination copy is only safe
|
|
966
|
+
when the source being kept instead is itself provably intact. Without
|
|
967
|
+
this, a source that vanished or drifted in a crash-adjacent window
|
|
968
|
+
would leave the destination - possibly the only remaining copy -
|
|
969
|
+
deleted anyway, because the existing hash-gate only ever checked the
|
|
970
|
+
DEST against its own journaled hash and said nothing about the
|
|
971
|
+
source's current state.
|
|
972
|
+
"""
|
|
973
|
+
if not os.path.isfile(m["source_transcript"]):
|
|
974
|
+
return False
|
|
975
|
+
got, gsize = sha256_file(m["source_transcript"])
|
|
976
|
+
if got != m["transcript_sha256"] or gsize != m["transcript_size"]:
|
|
977
|
+
return False
|
|
978
|
+
if m.get("sidecar_source"):
|
|
979
|
+
for e in m["sidecar_inventory"]:
|
|
980
|
+
p = os.path.join(m["sidecar_source"], *e["rel"].split("/"))
|
|
981
|
+
if not os.path.isfile(p):
|
|
982
|
+
return False
|
|
983
|
+
got, gsize = sha256_file(p)
|
|
984
|
+
if got != e["sha256"] or gsize != e["size"]:
|
|
985
|
+
return False
|
|
986
|
+
return True
|
|
987
|
+
|
|
988
|
+
|
|
989
|
+
def _abort(env, op, delete_dest=True, trigger=None):
|
|
990
|
+
prior_status = _pre_abort_status(op)
|
|
991
|
+
m = op.manifest
|
|
992
|
+
# C1(b): once this op has committed to a keep-both resolution - either
|
|
993
|
+
# the phase-6 decision a caller persisted to the manifest BEFORE ever
|
|
994
|
+
# calling _abort, or one _abort itself reaches below - every future
|
|
995
|
+
# invocation for this op must keep honoring it, including a
|
|
996
|
+
# crash-resumed one via recover's "back" (which always calls _abort
|
|
997
|
+
# with its own default delete_dest=True). Without this, the earlier
|
|
998
|
+
# decision is invisible to a later call and "back" can silently
|
|
999
|
+
# complete a hash-gated delete the first call deliberately declined.
|
|
1000
|
+
if m.get("abort_keep_dest"):
|
|
1001
|
+
delete_dest = False
|
|
1002
|
+
if trigger and not m.get("abort_reason"):
|
|
1003
|
+
m["abort_reason"] = trigger
|
|
1004
|
+
set_status(op, "aborting")
|
|
1005
|
+
_maybe_crash("after-aborting")
|
|
1006
|
+
|
|
1007
|
+
# Classify everything FIRST, as pure reads - no row is rewritten and no
|
|
1008
|
+
# destination file is deleted until we know the WHOLE rollback can
|
|
1009
|
+
# complete cleanly. Interleaving classification with mutation meant a
|
|
1010
|
+
# drifted row (or dest file) discovered partway through left some rows
|
|
1011
|
+
# already reverted and/or some dest files already deleted before the
|
|
1012
|
+
# Refusal - making a "nothing was deleted" claim false.
|
|
1013
|
+
row_restores = []
|
|
1014
|
+
drifted_rows = []
|
|
1015
|
+
for r in m["rows"]:
|
|
1016
|
+
state = _row_state(r)
|
|
1017
|
+
if state == "post":
|
|
1018
|
+
row_restores.append((r, unb64(r["pre_b64"])))
|
|
1019
|
+
elif state == "drifted":
|
|
1020
|
+
drifted_rows.append(r["path"])
|
|
1021
|
+
|
|
1022
|
+
scratch = prior_status in ("journaled", "copying")
|
|
1023
|
+
|
|
1024
|
+
# C1(a): a hash-gated (non-scratch) destination deletion only ever
|
|
1025
|
+
# checked the DEST against its own journaled hash; it said nothing
|
|
1026
|
+
# about whether the SOURCE we are keeping instead is actually still
|
|
1027
|
+
# there. Verify it before any such delete is allowed to happen.
|
|
1028
|
+
source_unverifiable = delete_dest and not scratch and not _source_pre_verified(m)
|
|
1029
|
+
|
|
1030
|
+
do_delete = delete_dest and not source_unverifiable
|
|
1031
|
+
dest_deletes = []
|
|
1032
|
+
drifted_dest = []
|
|
1033
|
+
if do_delete:
|
|
1034
|
+
for path, digest, size in _dest_files(m):
|
|
1035
|
+
if not os.path.isfile(path):
|
|
1036
|
+
continue
|
|
1037
|
+
if scratch:
|
|
1038
|
+
dest_deletes.append(path)
|
|
1039
|
+
continue
|
|
1040
|
+
got, gsize = sha256_file(path)
|
|
1041
|
+
if got == digest and gsize == size:
|
|
1042
|
+
dest_deletes.append(path)
|
|
1043
|
+
else:
|
|
1044
|
+
drifted_dest.append(path)
|
|
1045
|
+
|
|
1046
|
+
problems = drifted_rows + drifted_dest
|
|
1047
|
+
if problems:
|
|
1048
|
+
m["drifted_rows"] = drifted_rows
|
|
1049
|
+
save_manifest(op)
|
|
1050
|
+
raise Refusal("rollback could not verify every file ({0}); nothing "
|
|
1051
|
+
"was changed. Use 'claude-code-sessions recover' to "
|
|
1052
|
+
"resolve.".format(", ".join(problems)))
|
|
1053
|
+
|
|
1054
|
+
for r, pre_bytes in row_restores:
|
|
1055
|
+
atomic_write(r["path"], pre_bytes)
|
|
1056
|
+
for r in m["rows"]:
|
|
1057
|
+
r["rewritten"] = False
|
|
1058
|
+
m["drifted_rows"] = []
|
|
1059
|
+
save_manifest(op)
|
|
1060
|
+
|
|
1061
|
+
if source_unverifiable:
|
|
1062
|
+
# C1(a): rows are restored as usual above, but the destination is
|
|
1063
|
+
# never touched - the source we would be relying on to justify
|
|
1064
|
+
# deleting a hash-verified dest copy could not itself be verified,
|
|
1065
|
+
# so both copies are kept. Persist that decision (mirrors C1(b))
|
|
1066
|
+
# so a later resumed "back" never re-attempts the same unsafe
|
|
1067
|
+
# hash-gated delete.
|
|
1068
|
+
m["abort_keep_dest"] = True
|
|
1069
|
+
if not m.get("abort_reason"):
|
|
1070
|
+
m["abort_reason"] = "source changed at last instant"
|
|
1071
|
+
save_manifest(op)
|
|
1072
|
+
raise Refusal(
|
|
1073
|
+
"rollback could not verify the source ({0}) against its journaled "
|
|
1074
|
+
"pre-state; the destination copy at {1} is being kept, not deleted "
|
|
1075
|
+
"- nothing was lost, both copies remain. Run 'claude-code-sessions "
|
|
1076
|
+
"recover' to resolve.".format(m["source_transcript"], m["dest_transcript"]))
|
|
1077
|
+
|
|
1078
|
+
if do_delete:
|
|
1079
|
+
for path in dest_deletes:
|
|
1080
|
+
os.unlink(path)
|
|
1081
|
+
# I7: never rmtree - only the journaled files are ours to delete; any
|
|
1082
|
+
# leftover (non-inventoried) file makes its directory fail to rmdir
|
|
1083
|
+
# and survives, exactly like the source-side rule in execute_op.
|
|
1084
|
+
if m.get("sidecar_dest") and os.path.isdir(m["sidecar_dest"]):
|
|
1085
|
+
_rmdirs_bottom_up(m["sidecar_dest"])
|
|
1086
|
+
set_status(op, "rolled_back")
|
|
1087
|
+
|
|
1088
|
+
|
|
1089
|
+
def _validate_manifest_paths(env, m):
|
|
1090
|
+
"""Structural + containment validation for every path a manifest could
|
|
1091
|
+
direct a write or delete to. Must run before ANY mutation - a
|
|
1092
|
+
tampered/foreign manifest (or one whose target has since moved behind a
|
|
1093
|
+
symlink/junction) must be rejected before a single file is touched.
|
|
1094
|
+
Shared by execute_op (fresh runs) and recover_op (resumed runs, I4) so a
|
|
1095
|
+
resumed op gets exactly the same up-front check a fresh one does.
|
|
1096
|
+
"""
|
|
1097
|
+
# C2(a): a tampered/foreign manifest's rel paths must be structurally
|
|
1098
|
+
# safe before they are ever joined onto a filesystem path.
|
|
1099
|
+
for e in m.get("sidecar_inventory", []):
|
|
1100
|
+
_validate_sidecar_rel(e["rel"])
|
|
1101
|
+
|
|
1102
|
+
# C2(b): containment on the actual files (not just their dirnames), and
|
|
1103
|
+
# on every sidecar path we are about to touch - all before any mutation.
|
|
1104
|
+
ensure_contained(m["source_transcript"], [env.projects_root])
|
|
1105
|
+
ensure_contained(m["dest_transcript"], [env.projects_root])
|
|
1106
|
+
if m.get("sidecar_source"):
|
|
1107
|
+
ensure_contained(m["sidecar_source"], [env.projects_root])
|
|
1108
|
+
if m.get("sidecar_dest"):
|
|
1109
|
+
ensure_contained(m["sidecar_dest"], [env.projects_root])
|
|
1110
|
+
for e in m.get("sidecar_inventory", []):
|
|
1111
|
+
ensure_contained(os.path.join(m["sidecar_source"], *e["rel"].split("/")),
|
|
1112
|
+
[env.projects_root])
|
|
1113
|
+
ensure_contained(os.path.join(m["sidecar_dest"], *e["rel"].split("/")),
|
|
1114
|
+
[env.projects_root])
|
|
1115
|
+
|
|
1116
|
+
roots = _engine_roots(env, m)
|
|
1117
|
+
for r in m["rows"]:
|
|
1118
|
+
ensure_contained(r["path"], roots)
|
|
1119
|
+
|
|
1120
|
+
|
|
1121
|
+
def execute_op(env, op):
|
|
1122
|
+
"""Drive a freshly-journaled op through copy -> verify -> commit -> delete-last.
|
|
1123
|
+
|
|
1124
|
+
Only accepts ops whose status is 'journaled': this function always runs
|
|
1125
|
+
a full transaction from the top and is not itself resumption-aware.
|
|
1126
|
+
Resuming an op interrupted mid-flight is `recover`'s job (Task 11) - it
|
|
1127
|
+
inspects each phase individually rather than re-entering here. Callers
|
|
1128
|
+
hold the lock.
|
|
1129
|
+
"""
|
|
1130
|
+
m = op.manifest
|
|
1131
|
+
if m.get("status") != "journaled":
|
|
1132
|
+
raise LayoutError("execute_op only runs ops from 'journaled'; use recover "
|
|
1133
|
+
"for interrupted ops")
|
|
1134
|
+
|
|
1135
|
+
_validate_manifest_paths(env, m)
|
|
1136
|
+
|
|
1137
|
+
_maybe_crash("after-journaled")
|
|
1138
|
+
|
|
1139
|
+
set_status(op, "copying")
|
|
1140
|
+
_maybe_crash("after-copying")
|
|
1141
|
+
try:
|
|
1142
|
+
_copy_file(m["source_transcript"], m["dest_transcript"])
|
|
1143
|
+
for e in m["sidecar_inventory"]:
|
|
1144
|
+
_copy_file(os.path.join(m["sidecar_source"], *e["rel"].split("/")),
|
|
1145
|
+
os.path.join(m["sidecar_dest"], *e["rel"].split("/")))
|
|
1146
|
+
except OSError:
|
|
1147
|
+
_abort(env, op, delete_dest=True, trigger="copy failed")
|
|
1148
|
+
return "rolled_back"
|
|
1149
|
+
|
|
1150
|
+
bad = _verify(_dest_files(m))
|
|
1151
|
+
if bad is not None:
|
|
1152
|
+
_abort(env, op, trigger="destination verification failed")
|
|
1153
|
+
return "rolled_back"
|
|
1154
|
+
for path, _, _ in _dest_files(m):
|
|
1155
|
+
fsync_file(path)
|
|
1156
|
+
set_status(op, "copied")
|
|
1157
|
+
_maybe_crash("after-copied")
|
|
1158
|
+
|
|
1159
|
+
set_status(op, "rewriting")
|
|
1160
|
+
_maybe_crash("after-rewriting")
|
|
1161
|
+
try:
|
|
1162
|
+
rows = m["rows"]
|
|
1163
|
+
for i, r in enumerate(rows):
|
|
1164
|
+
# A row that changed between planning and rewriting (some other
|
|
1165
|
+
# process touched it) must never be blindly overwritten - re-read
|
|
1166
|
+
# its CURRENT bytes right before the write and compare against
|
|
1167
|
+
# the journaled pre-image. _abort independently re-derives each
|
|
1168
|
+
# row's state from its current bytes (never from the "rewritten"
|
|
1169
|
+
# flag), so it will correctly leave this drifted row untouched
|
|
1170
|
+
# and, per its existing fail-closed contract, refuse to complete
|
|
1171
|
+
# automatically if it cannot verify every row - `recover` is the
|
|
1172
|
+
# path out, exactly like any other drifted-row abort.
|
|
1173
|
+
with open(r["path"], "rb") as fh:
|
|
1174
|
+
current = fh.read()
|
|
1175
|
+
if current != unb64(r["pre_b64"]):
|
|
1176
|
+
# M3: the following return is unreachable in practice - this
|
|
1177
|
+
# is the FIRST time execute_op ever touches this row within a
|
|
1178
|
+
# fresh run (only journaled ops reach execute_op), so a
|
|
1179
|
+
# mismatch here can only mean "drifted" (never "post"), and
|
|
1180
|
+
# _abort always raises Refusal for a drifted row rather than
|
|
1181
|
+
# returning. Kept as a call, not inlined, so the abort still
|
|
1182
|
+
# happens if that invariant is ever wrong.
|
|
1183
|
+
_abort(env, op, trigger="row changed before rewrite")
|
|
1184
|
+
atomic_write(r["path"], unb64(r["post_b64"]))
|
|
1185
|
+
r["rewritten"] = True
|
|
1186
|
+
save_manifest(op)
|
|
1187
|
+
if i < len(rows) - 1:
|
|
1188
|
+
_maybe_crash("mid-rewriting")
|
|
1189
|
+
except OSError:
|
|
1190
|
+
_abort(env, op, trigger="row changed before rewrite")
|
|
1191
|
+
return "rolled_back"
|
|
1192
|
+
|
|
1193
|
+
set_status(op, "committed")
|
|
1194
|
+
_maybe_crash("after-committed")
|
|
1195
|
+
|
|
1196
|
+
# last-instant revalidation: BOTH sides + process guard (spec phase 6)
|
|
1197
|
+
src_ok = os.path.isfile(m["source_transcript"])
|
|
1198
|
+
if src_ok:
|
|
1199
|
+
got, gsize = sha256_file(m["source_transcript"])
|
|
1200
|
+
if got != m["transcript_sha256"] or gsize != m["transcript_size"]:
|
|
1201
|
+
src_ok = False
|
|
1202
|
+
if src_ok and m.get("sidecar_source"):
|
|
1203
|
+
for e in m["sidecar_inventory"]:
|
|
1204
|
+
p = os.path.join(m["sidecar_source"], *e["rel"].split("/"))
|
|
1205
|
+
if not os.path.isfile(p):
|
|
1206
|
+
src_ok = False
|
|
1207
|
+
break
|
|
1208
|
+
got, gsize = sha256_file(p)
|
|
1209
|
+
if got != e["sha256"] or gsize != e["size"]:
|
|
1210
|
+
src_ok = False
|
|
1211
|
+
break
|
|
1212
|
+
dest_ok = _verify(_dest_files(m)) is None
|
|
1213
|
+
running = claude_running(env)
|
|
1214
|
+
if not src_ok or not dest_ok or running:
|
|
1215
|
+
# I3/C1(b): persist the keep-both decision BEFORE _abort is even
|
|
1216
|
+
# called - a crash inside _abort itself (e.g. right after it enters
|
|
1217
|
+
# 'aborting') must not lose the fact that this rollback was always
|
|
1218
|
+
# meant to keep both copies. Once this is on the manifest, _abort
|
|
1219
|
+
# forces delete_dest=False on any future call for this op,
|
|
1220
|
+
# including a crash-resumed 'back' via recover.
|
|
1221
|
+
if not src_ok:
|
|
1222
|
+
reason = "source changed at last instant"
|
|
1223
|
+
elif not dest_ok:
|
|
1224
|
+
reason = "destination verification failed"
|
|
1225
|
+
else:
|
|
1226
|
+
reason = "process guard"
|
|
1227
|
+
m["abort_keep_dest"] = True
|
|
1228
|
+
m["abort_reason"] = reason
|
|
1229
|
+
save_manifest(op)
|
|
1230
|
+
_abort(env, op, delete_dest=False) # phase-6 abort keeps BOTH copies (spec)
|
|
1231
|
+
return "rolled_back"
|
|
1232
|
+
|
|
1233
|
+
# C1: never destroy a source-sidecar file that was never journaled - a
|
|
1234
|
+
# file that is the only copy of its data must not die with the source.
|
|
1235
|
+
if m.get("sidecar_source") and os.path.isdir(m["sidecar_source"]):
|
|
1236
|
+
inv_rels = {e["rel"] for e in m["sidecar_inventory"]}
|
|
1237
|
+
extra = []
|
|
1238
|
+
for dirpath, dirnames, filenames in os.walk(m["sidecar_source"]):
|
|
1239
|
+
for name in filenames:
|
|
1240
|
+
full = os.path.join(dirpath, name)
|
|
1241
|
+
rel = os.path.relpath(full, m["sidecar_source"]).replace(os.sep, "/")
|
|
1242
|
+
if rel not in inv_rels:
|
|
1243
|
+
extra.append(full)
|
|
1244
|
+
if extra:
|
|
1245
|
+
# I3/C1(b): same pre-persisted keep-both decision as above - a
|
|
1246
|
+
# newly-appeared source sidecar file is itself a form of
|
|
1247
|
+
# "source changed" since planning.
|
|
1248
|
+
m["abort_keep_dest"] = True
|
|
1249
|
+
m["abort_reason"] = "source changed at last instant"
|
|
1250
|
+
save_manifest(op)
|
|
1251
|
+
_abort(env, op, delete_dest=False)
|
|
1252
|
+
return "rolled_back"
|
|
1253
|
+
|
|
1254
|
+
# I8: sidecar files first, then now-empty dirs, transcript LAST. A
|
|
1255
|
+
# cleanup failure (e.g. a locked file) leaves the op at 'committed'
|
|
1256
|
+
# (non-terminal, no new journal state) for `recover` to finish instead
|
|
1257
|
+
# of crashing after the move has already been fully committed. A failed
|
|
1258
|
+
# sidecar delete must NOT be swallowed and must NOT let the transcript
|
|
1259
|
+
# get deleted anyway - that would orphan the sidecar file with no
|
|
1260
|
+
# journal trail. Leaving the transcript in place keeps the source
|
|
1261
|
+
# coherent for recover's classification.
|
|
1262
|
+
if m.get("sidecar_source") and os.path.isdir(m["sidecar_source"]):
|
|
1263
|
+
failures = _delete_inventoried_files(m["sidecar_source"], m["sidecar_inventory"])
|
|
1264
|
+
if failures:
|
|
1265
|
+
print("warning: move committed, but the old copy could not be fully "
|
|
1266
|
+
"removed ({0}). Run 'claude-code-sessions recover' to finish deleting "
|
|
1267
|
+
"it.".format(", ".join(p for p, _ in failures)))
|
|
1268
|
+
return "committed"
|
|
1269
|
+
_rmdirs_bottom_up(m["sidecar_source"])
|
|
1270
|
+
|
|
1271
|
+
try:
|
|
1272
|
+
os.unlink(m["source_transcript"])
|
|
1273
|
+
except OSError as exc:
|
|
1274
|
+
print("warning: move committed, but the old copy could not be fully "
|
|
1275
|
+
"removed ({0}). Run 'claude-code-sessions recover' to finish deleting "
|
|
1276
|
+
"it.".format(exc))
|
|
1277
|
+
return "committed"
|
|
1278
|
+
|
|
1279
|
+
set_status(op, "completed")
|
|
1280
|
+
return "completed"
|
|
1281
|
+
|
|
1282
|
+
|
|
1283
|
+
def run_move(env, manifest):
|
|
1284
|
+
lock_owner_op = "pending"
|
|
1285
|
+
acquire_lock(env, lock_owner_op)
|
|
1286
|
+
try:
|
|
1287
|
+
op = new_op(env, manifest)
|
|
1288
|
+
# we already hold the lock (no O_EXCL needed) - just record the real op_id
|
|
1289
|
+
with open(_lock_path(env), "w") as fh:
|
|
1290
|
+
fh.write("{0} {1}".format(os.getpid(), op.manifest["op_id"]))
|
|
1291
|
+
final = execute_op(env, op)
|
|
1292
|
+
if final == "completed":
|
|
1293
|
+
append_moved_log(env, {"kind": "move", "session_id": manifest["session_id"],
|
|
1294
|
+
"from": manifest["source_transcript"],
|
|
1295
|
+
"to": manifest["dest_transcript"],
|
|
1296
|
+
"at": env.now()})
|
|
1297
|
+
rotate_ops(env)
|
|
1298
|
+
return final
|
|
1299
|
+
finally:
|
|
1300
|
+
release_lock(env)
|
|
1301
|
+
|
|
1302
|
+
|
|
1303
|
+
# ------------------------------------------------------ undo
|
|
1304
|
+
def _op_sort_key(manifest):
|
|
1305
|
+
"""(creation time, op_id) - the same compound key list_ops sorts by.
|
|
1306
|
+
op_id alone is not a safe "is this newer" comparison: its trailing
|
|
1307
|
+
os.urandom(3).hex() carries no chronological meaning, only the
|
|
1308
|
+
strftime-derived prefix does, and two ops created in the same wall-clock
|
|
1309
|
+
second collapse to comparing that random suffix.
|
|
1310
|
+
"""
|
|
1311
|
+
return (manifest.get("history", [{}])[0].get("at", 0), manifest.get("op_id", ""))
|
|
1312
|
+
|
|
1313
|
+
|
|
1314
|
+
def plan_undo(env, prior_op):
|
|
1315
|
+
"""Build a reversal manifest for a completed move: source/dest swapped,
|
|
1316
|
+
row pre/post images swapped, same hashes. Every precondition here checks
|
|
1317
|
+
the CURRENT on-disk state against what the move journaled as its
|
|
1318
|
+
post-state - any drift (the app resumed the moved session, edited a row,
|
|
1319
|
+
etc.) means undoing would silently discard that activity, so it refuses
|
|
1320
|
+
instead of guessing. This is undo, not recover: growth at the
|
|
1321
|
+
destination is never accepted here the way `classify_op` accepts it for
|
|
1322
|
+
a crash-interrupted move.
|
|
1323
|
+
"""
|
|
1324
|
+
pm = prior_op.manifest
|
|
1325
|
+
if pm.get("op_type") == "undo":
|
|
1326
|
+
raise Refusal("op {0} is itself an undo; to redo, run move again"
|
|
1327
|
+
.format(pm.get("op_id")))
|
|
1328
|
+
if pm.get("status") != "completed":
|
|
1329
|
+
raise Refusal("op {0} is '{1}', not 'completed'; only completed ops can be "
|
|
1330
|
+
"undone".format(pm.get("op_id"), pm.get("status")))
|
|
1331
|
+
pm_key = _op_sort_key(pm)
|
|
1332
|
+
for other in list_ops(env):
|
|
1333
|
+
if _op_sort_key(other.manifest) > pm_key and \
|
|
1334
|
+
other.manifest.get("session_id") == pm["session_id"] and \
|
|
1335
|
+
other.manifest.get("status") not in ("rolled_back", "undone"):
|
|
1336
|
+
raise Refusal("a newer op touches this session; undo newest-first")
|
|
1337
|
+
|
|
1338
|
+
# C1: the undo's OWN destination is the original move's source path.
|
|
1339
|
+
# _abort's scratch rule treats a journaled/copying-phase destination as
|
|
1340
|
+
# tool-owned and deletes it unconditionally on rollback (no hash check);
|
|
1341
|
+
# a foreign file the user manually put back at that path - they restored
|
|
1342
|
+
# and resumed the session there by hand - would otherwise be destroyed
|
|
1343
|
+
# the moment the copy's O_EXCL create fails. Mirror plan_move's own
|
|
1344
|
+
# destination-exists check here, before any op is even journaled.
|
|
1345
|
+
if os.path.exists(pm["source_transcript"]) or \
|
|
1346
|
+
os.path.exists(sidecar_path(pm["source_transcript"])):
|
|
1347
|
+
raise Refusal("undo target already exists: {0}; refusing to overwrite it."
|
|
1348
|
+
.format(pm["source_transcript"]))
|
|
1349
|
+
|
|
1350
|
+
if not os.path.isfile(pm["dest_transcript"]):
|
|
1351
|
+
raise Refusal("the moved transcript is missing at {0}; cannot undo."
|
|
1352
|
+
.format(pm["dest_transcript"]))
|
|
1353
|
+
got, gsize = sha256_file(pm["dest_transcript"])
|
|
1354
|
+
if got != pm["transcript_sha256"] or gsize != pm["transcript_size"]:
|
|
1355
|
+
raise Refusal("the moved transcript has changed since the move (resumed or "
|
|
1356
|
+
"edited). Undoing would overwrite that activity; refusing.")
|
|
1357
|
+
for e in pm["sidecar_inventory"]:
|
|
1358
|
+
p = os.path.join(pm["sidecar_dest"], *e["rel"].split("/"))
|
|
1359
|
+
if not os.path.isfile(p):
|
|
1360
|
+
raise Refusal("sidecar file {0} has changed since the move; refusing."
|
|
1361
|
+
.format(e["rel"]))
|
|
1362
|
+
got_s, gsize_s = sha256_file(p) # M3: size AND hash, not hash alone
|
|
1363
|
+
if got_s != e["sha256"] or gsize_s != e["size"]:
|
|
1364
|
+
raise Refusal("sidecar file {0} has changed since the move; refusing."
|
|
1365
|
+
.format(e["rel"]))
|
|
1366
|
+
# I3: the reverse of the loop above - journaled-subset-of-present is not
|
|
1367
|
+
# enough; a file that appeared in the moved sidecar AFTER the move (never
|
|
1368
|
+
# journaled, so nothing above would ever notice it) must also block undo.
|
|
1369
|
+
# That file is post-move activity and must survive; refusing at plan
|
|
1370
|
+
# time (before any op is journaled) avoids ever landing in a stuck
|
|
1371
|
+
# two-folder state over it.
|
|
1372
|
+
if pm.get("sidecar_dest") and os.path.isdir(pm["sidecar_dest"]):
|
|
1373
|
+
inv_rels = {e["rel"] for e in pm["sidecar_inventory"]}
|
|
1374
|
+
extra = []
|
|
1375
|
+
for dirpath, dirnames, filenames in os.walk(pm["sidecar_dest"]):
|
|
1376
|
+
for name in filenames:
|
|
1377
|
+
full = os.path.join(dirpath, name)
|
|
1378
|
+
rel = os.path.relpath(full, pm["sidecar_dest"]).replace(os.sep, "/")
|
|
1379
|
+
if rel not in inv_rels:
|
|
1380
|
+
extra.append(full)
|
|
1381
|
+
if extra:
|
|
1382
|
+
raise Refusal("untracked file(s) appeared in the moved sidecar since the "
|
|
1383
|
+
"move ({0}); refusing - resolve manually."
|
|
1384
|
+
.format(", ".join(extra)))
|
|
1385
|
+
|
|
1386
|
+
rows = []
|
|
1387
|
+
for r in pm["rows"]:
|
|
1388
|
+
with open(r["path"], "rb") as fh:
|
|
1389
|
+
cur = fh.read()
|
|
1390
|
+
if cur != unb64(r["post_b64"]):
|
|
1391
|
+
raise Refusal("listing row {0} has changed since the move (the app may "
|
|
1392
|
+
"have updated it); refusing.".format(r["path"]))
|
|
1393
|
+
rows.append({"path": r["path"], "pre_b64": r["post_b64"],
|
|
1394
|
+
"post_b64": r["pre_b64"], "rewritten": False})
|
|
1395
|
+
|
|
1396
|
+
# M4: target_cwd describes where THIS manifest is taking the session -
|
|
1397
|
+
# for undo that is the original pre-move location, not the move's own
|
|
1398
|
+
# target_cwd (which described where the FORWARD move went).
|
|
1399
|
+
if pm["rows"]:
|
|
1400
|
+
first_pre = json.loads(unb64(pm["rows"][0]["pre_b64"]).decode("utf-8"))
|
|
1401
|
+
target_cwd = first_pre.get("cwd", "")
|
|
1402
|
+
else:
|
|
1403
|
+
target_cwd = ""
|
|
1404
|
+
|
|
1405
|
+
return {
|
|
1406
|
+
"op_type": "undo", "undo_of": pm["op_id"], "session_id": pm["session_id"],
|
|
1407
|
+
"mode": pm["mode"],
|
|
1408
|
+
"source_transcript": pm["dest_transcript"],
|
|
1409
|
+
"dest_transcript": pm["source_transcript"],
|
|
1410
|
+
"transcript_sha256": pm["transcript_sha256"],
|
|
1411
|
+
"transcript_size": pm["transcript_size"],
|
|
1412
|
+
"sidecar_source": pm.get("sidecar_dest"),
|
|
1413
|
+
"sidecar_dest": pm.get("sidecar_source"),
|
|
1414
|
+
"sidecar_inventory": pm["sidecar_inventory"],
|
|
1415
|
+
"rows": rows, "target_cwd": target_cwd,
|
|
1416
|
+
}
|
|
1417
|
+
|
|
1418
|
+
|
|
1419
|
+
def run_undo(env, prior_op):
|
|
1420
|
+
"""Lock -> claude_running check -> plan_undo -> new_op -> execute_op.
|
|
1421
|
+
The engine itself needs no changes for undo: an undo op is just a move
|
|
1422
|
+
manifest pointing the other way. But plan_move's process guard does not
|
|
1423
|
+
run for undo, so run_undo checks claude_running itself before executing
|
|
1424
|
+
(execute_op's own guard only fires at its last-instant revalidation,
|
|
1425
|
+
deep into the transaction) - and, like run_move/recover_op, it takes the
|
|
1426
|
+
single-instance lock FIRST, before doing any of its own checks, so two
|
|
1427
|
+
concurrent claude-code-sessions invocations can never race each other here.
|
|
1428
|
+
On 'completed' the prior op is marked 'undone' and a moved-log entry
|
|
1429
|
+
cancels its 'move' entry; any other outcome (e.g. 'committed' if final
|
|
1430
|
+
cleanup could not fully finish, or 'rolled_back') leaves the prior op's
|
|
1431
|
+
status untouched for the user to retry or recover.
|
|
1432
|
+
"""
|
|
1433
|
+
acquire_lock(env, "undo-" + prior_op.manifest["op_id"])
|
|
1434
|
+
try:
|
|
1435
|
+
if claude_running(env):
|
|
1436
|
+
raise Refusal("Claude appears to be running; close the app before undoing.")
|
|
1437
|
+
manifest = plan_undo(env, prior_op)
|
|
1438
|
+
op = new_op(env, manifest)
|
|
1439
|
+
final = execute_op(env, op)
|
|
1440
|
+
if final == "completed":
|
|
1441
|
+
set_status(prior_op, "undone")
|
|
1442
|
+
append_moved_log(env, {"kind": "undo",
|
|
1443
|
+
"session_id": manifest["session_id"],
|
|
1444
|
+
"at": env.now()})
|
|
1445
|
+
rotate_ops(env)
|
|
1446
|
+
return final
|
|
1447
|
+
finally:
|
|
1448
|
+
release_lock(env)
|
|
1449
|
+
|
|
1450
|
+
|
|
1451
|
+
# ------------------------------------------------------ recovery
|
|
1452
|
+
def is_prefix_of(journaled_hash, journaled_size, path):
|
|
1453
|
+
h = hashlib.sha256()
|
|
1454
|
+
remaining = journaled_size
|
|
1455
|
+
try:
|
|
1456
|
+
with open(path, "rb") as fh:
|
|
1457
|
+
while remaining > 0:
|
|
1458
|
+
chunk = fh.read(min(1 << 20, remaining))
|
|
1459
|
+
if not chunk:
|
|
1460
|
+
return False
|
|
1461
|
+
h.update(chunk)
|
|
1462
|
+
remaining -= len(chunk)
|
|
1463
|
+
except OSError:
|
|
1464
|
+
return False
|
|
1465
|
+
return h.hexdigest() == journaled_hash
|
|
1466
|
+
|
|
1467
|
+
|
|
1468
|
+
def classify_op(env, op):
|
|
1469
|
+
"""Classify a non-terminal op's source/destination/row state and the
|
|
1470
|
+
safe recovery resolutions. Rules per spec Recovery classification, plus
|
|
1471
|
+
the adversarial-review fixes noted inline (I3, I7, C2).
|
|
1472
|
+
"""
|
|
1473
|
+
m = op.manifest
|
|
1474
|
+
status = m["status"]
|
|
1475
|
+
|
|
1476
|
+
if m.get("op_type") == "sync":
|
|
1477
|
+
return classify_sync_op(env, op)
|
|
1478
|
+
|
|
1479
|
+
def _src_state():
|
|
1480
|
+
transcript_present = os.path.isfile(m["source_transcript"])
|
|
1481
|
+
if not transcript_present:
|
|
1482
|
+
if status == "committed":
|
|
1483
|
+
# I8 idempotency: a crash (or a prior partial recover) can
|
|
1484
|
+
# have already deleted the source transcript before this
|
|
1485
|
+
# classification runs. That is evidence deletion already
|
|
1486
|
+
# progressed, not drift - tolerate it here and let the
|
|
1487
|
+
# sidecar/dest checks below decide the rest.
|
|
1488
|
+
pass
|
|
1489
|
+
else:
|
|
1490
|
+
return "missing"
|
|
1491
|
+
else:
|
|
1492
|
+
got, gsize = sha256_file(m["source_transcript"])
|
|
1493
|
+
if got != m["transcript_sha256"] or gsize != m["transcript_size"]:
|
|
1494
|
+
return "drifted"
|
|
1495
|
+
if status == "committed" and m.get("sidecar_source"):
|
|
1496
|
+
for e in m["sidecar_inventory"]:
|
|
1497
|
+
p = os.path.join(m["sidecar_source"], *e["rel"].split("/"))
|
|
1498
|
+
if not os.path.isfile(p):
|
|
1499
|
+
continue # I8: already deleted - tolerated
|
|
1500
|
+
got, gsize = sha256_file(p)
|
|
1501
|
+
if got != e["sha256"] or gsize != e["size"]:
|
|
1502
|
+
return "drifted"
|
|
1503
|
+
return "pre"
|
|
1504
|
+
|
|
1505
|
+
def _dest_sidecar_ok(e):
|
|
1506
|
+
p = os.path.join(m["sidecar_dest"], *e["rel"].split("/"))
|
|
1507
|
+
if not os.path.isfile(p):
|
|
1508
|
+
return False
|
|
1509
|
+
got, gsize = sha256_file(p)
|
|
1510
|
+
return got == e["sha256"] and gsize == e["size"] # minor: size, not just hash
|
|
1511
|
+
|
|
1512
|
+
def _dest_state():
|
|
1513
|
+
if not os.path.exists(m["dest_transcript"]):
|
|
1514
|
+
return "absent"
|
|
1515
|
+
got, gsize = sha256_file(m["dest_transcript"])
|
|
1516
|
+
sidecars_ok = all(_dest_sidecar_ok(e) for e in m["sidecar_inventory"]) \
|
|
1517
|
+
if m.get("sidecar_dest") else True
|
|
1518
|
+
if got == m["transcript_sha256"] and gsize == m["transcript_size"] and sidecars_ok:
|
|
1519
|
+
return "intact"
|
|
1520
|
+
if gsize > m["transcript_size"] and sidecars_ok and \
|
|
1521
|
+
is_prefix_of(m["transcript_sha256"], m["transcript_size"], m["dest_transcript"]):
|
|
1522
|
+
return "grown"
|
|
1523
|
+
return "drifted"
|
|
1524
|
+
|
|
1525
|
+
src, dest = _src_state(), _dest_state()
|
|
1526
|
+
# C1: rows are inspected here too so the CLI can warn about a drifted
|
|
1527
|
+
# row before the user even picks a direction; the actual write-vs-refuse
|
|
1528
|
+
# decision for a "copied"/"rewriting" forward happens in
|
|
1529
|
+
# _forward_rewrite_and_commit, not here.
|
|
1530
|
+
drifted_rows = [r["path"] for r in m["rows"] if _row_state(r) == "drifted"]
|
|
1531
|
+
|
|
1532
|
+
if status in ("journaled", "copying"):
|
|
1533
|
+
dest_label = "partial-scratch" if dest != "absent" else "absent"
|
|
1534
|
+
if src != "pre":
|
|
1535
|
+
# I7: the scratch-deletion rule assumes the source is still
|
|
1536
|
+
# pristine. If it has vanished or drifted, the partial
|
|
1537
|
+
# destination might be the only remaining copy of the data -
|
|
1538
|
+
# never delete it, and never resume the copy automatically
|
|
1539
|
+
# either (it would read from an unverified source).
|
|
1540
|
+
return {"status": status, "source": src, "dest": dest_label, "resolutions": [],
|
|
1541
|
+
"drifted_rows": drifted_rows,
|
|
1542
|
+
"note": "source is {0}; the partial destination may be the only "
|
|
1543
|
+
"remaining copy - refusing to delete it or resume "
|
|
1544
|
+
"automatically".format(src)}
|
|
1545
|
+
return {"status": status, "source": src, "dest": dest_label,
|
|
1546
|
+
"resolutions": ["back", "forward"], "drifted_rows": drifted_rows,
|
|
1547
|
+
"note": "destination is tool-owned scratch at this phase"}
|
|
1548
|
+
if status == "aborting":
|
|
1549
|
+
if drifted_rows:
|
|
1550
|
+
# I3: a drifted row makes the rollback itself impossible to
|
|
1551
|
+
# complete automatically - offering "back" forever (which will
|
|
1552
|
+
# only raise the same Refusal every time) is a dead end.
|
|
1553
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": [],
|
|
1554
|
+
"drifted_rows": drifted_rows,
|
|
1555
|
+
"note": "row(s) changed unexpectedly during rollback ({0}); "
|
|
1556
|
+
"automatic recovery cannot proceed - resolve manually"
|
|
1557
|
+
.format(", ".join(drifted_rows))}
|
|
1558
|
+
# C1(c): mirror the journaled/copying branch's source gate. A
|
|
1559
|
+
# hash-gated (non-scratch) "back" deletes the destination only after
|
|
1560
|
+
# verifying the source is still "pre" (C1a) - if it is not, and this
|
|
1561
|
+
# op has not already committed to keeping the destination
|
|
1562
|
+
# (abort_keep_dest), automatic recovery cannot know in advance
|
|
1563
|
+
# whether "back" will complete or refuse, and offering it forever
|
|
1564
|
+
# would be the same dead end as a drifted row.
|
|
1565
|
+
if src != "pre" and not m.get("abort_keep_dest"):
|
|
1566
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": [],
|
|
1567
|
+
"drifted_rows": drifted_rows,
|
|
1568
|
+
"note": "source is {0} and the destination was never confirmed "
|
|
1569
|
+
"kept; refusing to resolve automatically - resolve "
|
|
1570
|
+
"manually".format(src)}
|
|
1571
|
+
note = "completing an interrupted rollback"
|
|
1572
|
+
if dest == "drifted":
|
|
1573
|
+
# I4: a drifted destination is never deleted either way - the
|
|
1574
|
+
# hash-gate in _abort only ever deletes a dest file that still
|
|
1575
|
+
# matches its journaled hash, so "back" remains safe to offer;
|
|
1576
|
+
# it will just keep (and report) the drifted file rather than
|
|
1577
|
+
# touch it.
|
|
1578
|
+
note += "; destination has drifted since it was journaled and will " \
|
|
1579
|
+
"be kept, not deleted"
|
|
1580
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": ["back"],
|
|
1581
|
+
"drifted_rows": drifted_rows, "note": note}
|
|
1582
|
+
if status in ("copied", "rewriting"):
|
|
1583
|
+
if dest == "grown":
|
|
1584
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": ["forward"],
|
|
1585
|
+
"drifted_rows": drifted_rows,
|
|
1586
|
+
"note": "destination has post-crash growth; it will never be deleted"}
|
|
1587
|
+
if dest != "intact":
|
|
1588
|
+
# C2: only offer "forward" when finishing can actually succeed.
|
|
1589
|
+
# A genuinely drifted (or vanished) destination can never pass
|
|
1590
|
+
# the finish gate, so offering "forward" here would let the
|
|
1591
|
+
# rows get flipped to point at a bad copy before discovering
|
|
1592
|
+
# that. Neither direction is safe automatically - both copies
|
|
1593
|
+
# are kept and the row is never touched.
|
|
1594
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": [],
|
|
1595
|
+
"drifted_rows": drifted_rows,
|
|
1596
|
+
"note": "destination no longer contains a verifiable copy; both "
|
|
1597
|
+
"copies are kept - resolve manually"}
|
|
1598
|
+
return {"status": status, "source": src, "dest": dest,
|
|
1599
|
+
"resolutions": ["back", "forward"], "drifted_rows": drifted_rows, "note": ""}
|
|
1600
|
+
if status == "committed":
|
|
1601
|
+
if src != "pre":
|
|
1602
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": [],
|
|
1603
|
+
"drifted_rows": drifted_rows,
|
|
1604
|
+
"note": "source changed after commit; resolve manually - both copies kept"}
|
|
1605
|
+
if dest in ("intact", "grown"):
|
|
1606
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": ["forward"],
|
|
1607
|
+
"drifted_rows": drifted_rows,
|
|
1608
|
+
"note": "finishing means deleting the stale source duplicate"}
|
|
1609
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": [],
|
|
1610
|
+
"drifted_rows": drifted_rows,
|
|
1611
|
+
"note": "destination no longer contains the copy; keeping the source"}
|
|
1612
|
+
return {"status": status, "source": src, "dest": dest, "resolutions": [],
|
|
1613
|
+
"drifted_rows": drifted_rows, "note": "terminal"}
|
|
1614
|
+
|
|
1615
|
+
|
|
1616
|
+
def _forward_rewrite_and_commit(env, op, c):
|
|
1617
|
+
"""Finish rolling a 'copied'/'rewriting' op forward: gate first (C2),
|
|
1618
|
+
classify every row's CURRENT bytes before writing any of them (C1), and
|
|
1619
|
+
only then mutate - either everything proceeds together or nothing does.
|
|
1620
|
+
`c` is the classify_op result already computed by the caller before any
|
|
1621
|
+
mutation, so its source/dest snapshot is still valid here.
|
|
1622
|
+
"""
|
|
1623
|
+
m = op.manifest
|
|
1624
|
+
if c["source"] != "pre" or c["dest"] not in ("intact", "grown"):
|
|
1625
|
+
raise Refusal("cannot roll op {0} forward: {1}".format(m["op_id"], c["note"]))
|
|
1626
|
+
|
|
1627
|
+
row_writes = []
|
|
1628
|
+
drifted = []
|
|
1629
|
+
for r in m["rows"]:
|
|
1630
|
+
state = _row_state(r)
|
|
1631
|
+
if state == "pre":
|
|
1632
|
+
row_writes.append(r)
|
|
1633
|
+
elif state == "drifted":
|
|
1634
|
+
drifted.append(r["path"])
|
|
1635
|
+
# state == "post": already applied, nothing to do
|
|
1636
|
+
if drifted:
|
|
1637
|
+
raise Refusal("cannot roll op {0} forward: row(s) changed unexpectedly since "
|
|
1638
|
+
"the journaled images ({1}); nothing was written. Resolve "
|
|
1639
|
+
"manually, then retry recover.".format(m["op_id"], ", ".join(drifted)))
|
|
1640
|
+
|
|
1641
|
+
for r in row_writes:
|
|
1642
|
+
atomic_write(r["path"], unb64(r["post_b64"]))
|
|
1643
|
+
r["rewritten"] = True
|
|
1644
|
+
save_manifest(op)
|
|
1645
|
+
set_status(op, "committed")
|
|
1646
|
+
return _finish_committed(env, op)
|
|
1647
|
+
|
|
1648
|
+
|
|
1649
|
+
def recover_op(env, op, direction):
|
|
1650
|
+
# I5: recover mutates a journaled op exactly like run_move does, so it
|
|
1651
|
+
# needs the same single-instance lock - two concurrent recover attempts
|
|
1652
|
+
# (or a recover racing a live move) must not interleave.
|
|
1653
|
+
acquire_lock(env, "recover-" + op.manifest["op_id"])
|
|
1654
|
+
try:
|
|
1655
|
+
# I4: validate containment before ANY mutation, including the
|
|
1656
|
+
# scratch-dest unlink below - not just on execute_op's fresh runs.
|
|
1657
|
+
# A sync manifest carries no source_transcript/sidecar_inventory/
|
|
1658
|
+
# row["path"] for this validator to inspect - execute_sync_op does
|
|
1659
|
+
# its own containment check inline, per-row, instead.
|
|
1660
|
+
if op.manifest.get("op_type") != "sync":
|
|
1661
|
+
_validate_manifest_paths(env, op.manifest)
|
|
1662
|
+
|
|
1663
|
+
c = classify_op(env, op)
|
|
1664
|
+
if direction not in c["resolutions"]:
|
|
1665
|
+
raise Refusal("'{0}' is not a safe resolution for op {1} ({2}); options: {3}"
|
|
1666
|
+
.format(direction, op.manifest["op_id"], c["note"],
|
|
1667
|
+
c["resolutions"] or "none - manual intervention"))
|
|
1668
|
+
m = op.manifest
|
|
1669
|
+
if m.get("op_type") == "sync":
|
|
1670
|
+
# direction is already guaranteed to be a member of c["resolutions"]
|
|
1671
|
+
# by the check above, and classify_sync_op only ever offers
|
|
1672
|
+
# "back", "forward" or both (never anything else) for a
|
|
1673
|
+
# non-terminal sync op - so no further validation is needed here.
|
|
1674
|
+
if direction == "back":
|
|
1675
|
+
# Always available for a non-terminal sync, because drift is
|
|
1676
|
+
# only one of the ways forward can be permanently blocked (an
|
|
1677
|
+
# I/O or layout failure leaves the pending row absent, which
|
|
1678
|
+
# looks like no drift at all). Unlike undo_sync's
|
|
1679
|
+
# all-or-nothing, back must
|
|
1680
|
+
# always terminate - refusing over a written row that also
|
|
1681
|
+
# drifted or turned unreadable would recreate the exact
|
|
1682
|
+
# dead end this resolution exists to close (recover back
|
|
1683
|
+
# refuses, recover forward refuses, undo refuses:
|
|
1684
|
+
# permanently stuck). So a row this op cannot verify is
|
|
1685
|
+
# SKIPPED, never deleted, and the rest are removed; the
|
|
1686
|
+
# blocking pending row itself is never even considered,
|
|
1687
|
+
# because this op never wrote it.
|
|
1688
|
+
drifted, unreadable, removable = _sync_delete_targets(env, m)
|
|
1689
|
+
_sync_unlink_all(removable)
|
|
1690
|
+
skipped = drifted + unreadable
|
|
1691
|
+
# Same reporting mechanism _abort already uses for move -
|
|
1692
|
+
# cmd_recover's existing _print_abort_reason picks this up for
|
|
1693
|
+
# free once status is non-"completed".
|
|
1694
|
+
if skipped:
|
|
1695
|
+
op.manifest["abort_reason"] = (
|
|
1696
|
+
"back removed {0} row(s) it could verify; left {1} "
|
|
1697
|
+
"untouched because {2}".format(
|
|
1698
|
+
len(removable), len(skipped),
|
|
1699
|
+
_drift_clause(drifted, unreadable)))
|
|
1700
|
+
elif not removable:
|
|
1701
|
+
# Say so out loud rather than printing a bare
|
|
1702
|
+
# "rolled_back". This is reachable and it is a trap: a
|
|
1703
|
+
# hard kill (not an exception - those journal on the way
|
|
1704
|
+
# out) during a batched run can leave rows on disk that
|
|
1705
|
+
# the manifest never marked written, and 'back' only
|
|
1706
|
+
# removes rows it can see it wrote. Name the forward
|
|
1707
|
+
# route, because it is the one that cleans them up.
|
|
1708
|
+
op.manifest["abort_reason"] = (
|
|
1709
|
+
"back removed nothing - this op's manifest records no "
|
|
1710
|
+
"row as written. If rows did land in the destination, "
|
|
1711
|
+
"'recover --resolve {0} --forward --apply' will "
|
|
1712
|
+
"recognise and record them, after which 'undo --id {0} "
|
|
1713
|
+
"--apply' removes them exactly."
|
|
1714
|
+
.format(m["op_id"]))
|
|
1715
|
+
set_status(op, "rolled_back")
|
|
1716
|
+
rotate_ops(env)
|
|
1717
|
+
return "rolled_back"
|
|
1718
|
+
# forward: re-enter execute_sync_op to finish the remaining writes.
|
|
1719
|
+
set_status(op, "journaled")
|
|
1720
|
+
final = execute_sync_op(env, op)
|
|
1721
|
+
if final == "completed":
|
|
1722
|
+
rotate_ops(env)
|
|
1723
|
+
return final
|
|
1724
|
+
if direction == "back":
|
|
1725
|
+
_abort(env, op)
|
|
1726
|
+
return "rolled_back"
|
|
1727
|
+
# forward
|
|
1728
|
+
if m["status"] in ("journaled", "copying"):
|
|
1729
|
+
for path, _, _ in _dest_files(m): # scratch rule: clear partials
|
|
1730
|
+
if os.path.exists(path):
|
|
1731
|
+
os.unlink(path)
|
|
1732
|
+
set_status(op, "journaled") # minor: history entry, not a raw write
|
|
1733
|
+
final = execute_op(env, op)
|
|
1734
|
+
elif m["status"] in ("copied", "rewriting"):
|
|
1735
|
+
final = _forward_rewrite_and_commit(env, op, c)
|
|
1736
|
+
else: # committed
|
|
1737
|
+
final = _finish_committed(env, op)
|
|
1738
|
+
if final == "completed":
|
|
1739
|
+
# C2: a recovered op can be either direction of the engine, not
|
|
1740
|
+
# just a forward move - the moved-log entry (and any follow-on
|
|
1741
|
+
# bookkeeping) must be derived from what THIS op actually is,
|
|
1742
|
+
# not hardcoded to "move".
|
|
1743
|
+
if m.get("op_type") == "undo":
|
|
1744
|
+
append_moved_log(env, {"kind": "undo", "session_id": m["session_id"],
|
|
1745
|
+
"at": env.now()})
|
|
1746
|
+
_mark_undo_of_undone(env, m)
|
|
1747
|
+
else:
|
|
1748
|
+
append_moved_log(env, {"kind": "move", "session_id": m["session_id"],
|
|
1749
|
+
"from": m["source_transcript"], "to": m["dest_transcript"],
|
|
1750
|
+
"at": env.now()})
|
|
1751
|
+
rotate_ops(env)
|
|
1752
|
+
return final
|
|
1753
|
+
finally:
|
|
1754
|
+
release_lock(env)
|
|
1755
|
+
|
|
1756
|
+
|
|
1757
|
+
def _mark_undo_of_undone(env, undo_manifest):
|
|
1758
|
+
"""After a recovered undo op reaches 'completed', mark the ORIGINAL op
|
|
1759
|
+
it reversed as 'undone' - mirroring what run_undo does on its own
|
|
1760
|
+
successful path. Without this, a crash between an undo op's 'committed'
|
|
1761
|
+
status and run_undo's own set_status(prior_op, "undone") call would
|
|
1762
|
+
leave the original move op stuck reading 'completed' forever, even
|
|
1763
|
+
though its transcript has actually moved back home. Silently no-ops if
|
|
1764
|
+
the referenced op can no longer be found (e.g. already rotated away by
|
|
1765
|
+
a much later cleanup) rather than failing an otherwise-successful
|
|
1766
|
+
recovery over bookkeeping for an op that is long gone either way.
|
|
1767
|
+
"""
|
|
1768
|
+
prior_id = undo_manifest.get("undo_of")
|
|
1769
|
+
if not prior_id:
|
|
1770
|
+
return
|
|
1771
|
+
for other in list_ops(env):
|
|
1772
|
+
if other.manifest.get("op_id") == prior_id:
|
|
1773
|
+
set_status(other, "undone")
|
|
1774
|
+
return
|
|
1775
|
+
|
|
1776
|
+
|
|
1777
|
+
def _finish_committed(env, op):
|
|
1778
|
+
"""Finish a 'committed' op by deleting the now-redundant source copy.
|
|
1779
|
+
|
|
1780
|
+
Uses the same delete helpers and failure contract as execute_op's final
|
|
1781
|
+
commit step (never rmtree): the same un-inventoried-file and
|
|
1782
|
+
claude-running guards (I6), then inventoried sidecar files, then their
|
|
1783
|
+
now-empty directories, then the transcript LAST. A file already missing
|
|
1784
|
+
(I8: deletion already progressed - see classify_op) is simply skipped
|
|
1785
|
+
rather than treated as a failure. Any real delete failure, or either
|
|
1786
|
+
guard tripping, raises a Refusal naming the paths and leaves the op at
|
|
1787
|
+
'committed' (non-terminal) for a later recover to retry - it is never
|
|
1788
|
+
silently swallowed.
|
|
1789
|
+
"""
|
|
1790
|
+
m = op.manifest
|
|
1791
|
+
c = classify_op(env, op)
|
|
1792
|
+
if c["source"] != "pre" or c["dest"] not in ("intact", "grown"):
|
|
1793
|
+
raise Refusal("cannot finish op {0}: {1}".format(m["op_id"], c["note"]))
|
|
1794
|
+
|
|
1795
|
+
# I6(a): a live Claude process could be actively appending to the
|
|
1796
|
+
# source right now; deleting it out from under a running process is
|
|
1797
|
+
# exactly the hazard the pre-move guard exists to prevent.
|
|
1798
|
+
running = claude_running(env)
|
|
1799
|
+
if running:
|
|
1800
|
+
raise Refusal("cannot finish op {0}: Claude appears to be running ({1}). "
|
|
1801
|
+
"Close it, then retry recover.".format(m["op_id"], ", ".join(sorted(set(running))[:3])))
|
|
1802
|
+
|
|
1803
|
+
if m.get("sidecar_source") and os.path.isdir(m["sidecar_source"]):
|
|
1804
|
+
# I6(b): same C1 guard as execute_op's own commit step - a file
|
|
1805
|
+
# that appeared in the source sidecar after commit was never
|
|
1806
|
+
# journaled and must never be destroyed; both copies are kept.
|
|
1807
|
+
inv_rels = {e["rel"] for e in m["sidecar_inventory"]}
|
|
1808
|
+
extra = []
|
|
1809
|
+
for dirpath, dirnames, filenames in os.walk(m["sidecar_source"]):
|
|
1810
|
+
for name in filenames:
|
|
1811
|
+
full = os.path.join(dirpath, name)
|
|
1812
|
+
rel = os.path.relpath(full, m["sidecar_source"]).replace(os.sep, "/")
|
|
1813
|
+
if rel not in inv_rels:
|
|
1814
|
+
extra.append(full)
|
|
1815
|
+
if extra:
|
|
1816
|
+
raise Refusal("cannot finish op {0}: untracked file(s) appeared in the "
|
|
1817
|
+
"source sidecar since commit ({1}); both copies are kept - "
|
|
1818
|
+
"resolve manually".format(m["op_id"], ", ".join(extra)))
|
|
1819
|
+
|
|
1820
|
+
remaining = [e for e in m["sidecar_inventory"]
|
|
1821
|
+
if os.path.isfile(os.path.join(m["sidecar_source"], *e["rel"].split("/")))]
|
|
1822
|
+
failures = _delete_inventoried_files(m["sidecar_source"], remaining)
|
|
1823
|
+
if failures:
|
|
1824
|
+
raise Refusal("cannot finish op {0}: could not remove {1}".format(
|
|
1825
|
+
m["op_id"], ", ".join(p for p, _ in failures)))
|
|
1826
|
+
_rmdirs_bottom_up(m["sidecar_source"])
|
|
1827
|
+
|
|
1828
|
+
if os.path.isfile(m["source_transcript"]):
|
|
1829
|
+
try:
|
|
1830
|
+
os.unlink(m["source_transcript"])
|
|
1831
|
+
except OSError as exc:
|
|
1832
|
+
raise Refusal("cannot finish op {0}: could not remove source transcript "
|
|
1833
|
+
"{1}: {2}".format(m["op_id"], m["source_transcript"], exc))
|
|
1834
|
+
|
|
1835
|
+
set_status(op, "completed")
|
|
1836
|
+
return "completed"
|
|
1837
|
+
|
|
1838
|
+
|
|
1839
|
+
def clear_stale_lock(env):
|
|
1840
|
+
if lock_is_stale(env):
|
|
1841
|
+
release_lock(env)
|
|
1842
|
+
return True
|
|
1843
|
+
return False
|
|
1844
|
+
|
|
1845
|
+
|
|
1846
|
+
# ------------------------------------------------- 6. commands: list, doctor
|
|
1847
|
+
_UUID_RE = re.compile(r"\b([0-9a-f]{8})-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b", re.IGNORECASE)
|
|
1848
|
+
|
|
1849
|
+
|
|
1850
|
+
def redact(env, text, keep_ids=False):
|
|
1851
|
+
for h in {env.home, os.path.realpath(env.home)}:
|
|
1852
|
+
text = text.replace(h, "~")
|
|
1853
|
+
if not keep_ids:
|
|
1854
|
+
text = _UUID_RE.sub(lambda m: m.group(1) + "…", text)
|
|
1855
|
+
return text
|
|
1856
|
+
|
|
1857
|
+
|
|
1858
|
+
def gather_list(env, query="", project=None):
|
|
1859
|
+
disc = discover_stores(env)
|
|
1860
|
+
rows, _ = load_rows(disc.roots)
|
|
1861
|
+
items, seen = [], {}
|
|
1862
|
+
for r in rows:
|
|
1863
|
+
if not r.cli_session_id:
|
|
1864
|
+
continue
|
|
1865
|
+
if r.cli_session_id not in seen or seen[r.cli_session_id]["last_activity"] < r.last_activity:
|
|
1866
|
+
seen[r.cli_session_id] = {"session_id": r.cli_session_id, "title": r.data.get("title") or "",
|
|
1867
|
+
"cwd": r.cwd, "last_activity": r.last_activity, "listed": True}
|
|
1868
|
+
for folder, path in iter_transcripts(env.projects_root):
|
|
1869
|
+
sid = os.path.splitext(os.path.basename(path))[0]
|
|
1870
|
+
if sid in seen or not path.endswith(".jsonl"):
|
|
1871
|
+
continue
|
|
1872
|
+
seen[sid] = {"session_id": sid, "title": "", "cwd": first_cwd(path),
|
|
1873
|
+
"last_activity": int(os.path.getmtime(path) * 1000),
|
|
1874
|
+
"listed": False}
|
|
1875
|
+
items = list(seen.values())
|
|
1876
|
+
if query:
|
|
1877
|
+
q = query.lower()
|
|
1878
|
+
items = [i for i in items
|
|
1879
|
+
if q in (i["title"] + " " + i["cwd"] + " " + i["session_id"]).lower()]
|
|
1880
|
+
if project:
|
|
1881
|
+
p = os.path.normpath(os.path.abspath(project)).lower() + os.sep
|
|
1882
|
+
items = [i for i in items
|
|
1883
|
+
if os.path.normpath(i["cwd"]).lower() == p.rstrip(os.sep)
|
|
1884
|
+
or os.path.normpath(i["cwd"]).lower().startswith(p)]
|
|
1885
|
+
items.sort(key=lambda i: i["last_activity"], reverse=True)
|
|
1886
|
+
return items
|
|
1887
|
+
|
|
1888
|
+
|
|
1889
|
+
def cmd_list(env, ns):
|
|
1890
|
+
items = gather_list(env, query=getattr(ns, "query", "") or "",
|
|
1891
|
+
project=getattr(ns, "project", None))
|
|
1892
|
+
if ns.json:
|
|
1893
|
+
print(json.dumps(items, indent=1))
|
|
1894
|
+
return 0
|
|
1895
|
+
for i in items:
|
|
1896
|
+
line = "{0} {1:40.40} {2}".format(i["session_id"], i["title"] or "(no title)", i["cwd"])
|
|
1897
|
+
if ns.verbose:
|
|
1898
|
+
print(line)
|
|
1899
|
+
elif getattr(ns, "full", False):
|
|
1900
|
+
print(redact(env, line, keep_ids=True))
|
|
1901
|
+
else:
|
|
1902
|
+
print(redact(env, line))
|
|
1903
|
+
if not items:
|
|
1904
|
+
print("no sessions found")
|
|
1905
|
+
return 0
|
|
1906
|
+
|
|
1907
|
+
|
|
1908
|
+
RETENTION_HINT_DAYS = 30
|
|
1909
|
+
|
|
1910
|
+
|
|
1911
|
+
def gather_doctor(env):
|
|
1912
|
+
disc = discover_stores(env)
|
|
1913
|
+
rows, row_errors = load_rows(disc.roots)
|
|
1914
|
+
transcripts = iter_transcripts(env.projects_root)
|
|
1915
|
+
tids = {os.path.splitext(os.path.basename(p))[0] for _, p in transcripts}
|
|
1916
|
+
blank = [r.local_id for r in rows if not r.cli_session_id]
|
|
1917
|
+
dead = [{"local_id": r.local_id,
|
|
1918
|
+
"age_days": round((env.now() * 1000 - r.last_activity) / 86_400_000)}
|
|
1919
|
+
for r in rows if r.cli_session_id and r.cli_session_id not in tids]
|
|
1920
|
+
listed = {r.cli_session_id for r in rows if r.cli_session_id}
|
|
1921
|
+
unlisted = sorted(tids - listed)
|
|
1922
|
+
cwds = [r.cwd for r in rows if r.cwd]
|
|
1923
|
+
cur, leg = scheme_evidence(cwds, env.projects_root)
|
|
1924
|
+
# Recent-50 evidence: the SAME population plan_move itself consults to
|
|
1925
|
+
# choose a scheme for a live move (the 50 most-recently-active rows).
|
|
1926
|
+
# Ruling fix (Task 13's "cur > 0 and leg > 0" over ALL rows was wrong):
|
|
1927
|
+
# a machine that lived through the 2026-07 encoding change legitimately
|
|
1928
|
+
# has folders under both schemes forever after - that history alone is
|
|
1929
|
+
# not an "unknown layout", it is exactly what legacy_folders already
|
|
1930
|
+
# reports (a exit-1 finding). Only a genuine TIE in the evidence that
|
|
1931
|
+
# would actually decide a future move - both counts equal and nonzero
|
|
1932
|
+
# among the most-recent 50 rows - means the layout cannot be
|
|
1933
|
+
# determined and is worth exit 2.
|
|
1934
|
+
recent_rows = sorted(rows, key=lambda r: r.last_activity)[-50:]
|
|
1935
|
+
recent_cwds = [r.cwd for r in recent_rows]
|
|
1936
|
+
cur_recent, leg_recent = scheme_evidence(recent_cwds, env.projects_root)
|
|
1937
|
+
legacy_folders = []
|
|
1938
|
+
legacy_folders_set = set()
|
|
1939
|
+
for cwd in set(cwds):
|
|
1940
|
+
a, b = encode(cwd, SCHEME_CURRENT), encode(cwd, SCHEME_LEGACY)
|
|
1941
|
+
if a != b and os.path.isdir(os.path.join(env.projects_root, a)) \
|
|
1942
|
+
and os.path.isdir(os.path.join(env.projects_root, b)):
|
|
1943
|
+
if b not in legacy_folders_set:
|
|
1944
|
+
n = len([x for x in os.listdir(os.path.join(env.projects_root, b))
|
|
1945
|
+
if x.endswith(".jsonl")])
|
|
1946
|
+
legacy_folders.append({"folder": b, "transcripts": n})
|
|
1947
|
+
legacy_folders_set.add(b)
|
|
1948
|
+
unknown_layout = []
|
|
1949
|
+
if cur_recent == leg_recent > 0:
|
|
1950
|
+
unknown_layout = ["encoding-scheme evidence is tied/undecidable (recent 50: "
|
|
1951
|
+
"current={0} legacy={1})".format(cur_recent, leg_recent)]
|
|
1952
|
+
nt = [o.manifest["op_id"] for o in nonterminal_ops(env)]
|
|
1953
|
+
report = {
|
|
1954
|
+
"stores": {"status": disc.status, "roots": disc.roots, "detail": disc.detail},
|
|
1955
|
+
"row_count": len(rows), "row_errors": row_errors, "blank_rows": sorted(blank),
|
|
1956
|
+
"dead_rows": dead, "unlisted_transcripts": unlisted,
|
|
1957
|
+
"encoding": {"current": cur, "legacy": leg},
|
|
1958
|
+
"encoding_recent": {"current": cur_recent, "legacy": leg_recent},
|
|
1959
|
+
"legacy_folders": legacy_folders, "nonterminal_ops": nt,
|
|
1960
|
+
"stale_lock": lock_is_stale(env),
|
|
1961
|
+
"unknown_layout": unknown_layout,
|
|
1962
|
+
}
|
|
1963
|
+
if disc.status == "error" or row_errors or unknown_layout:
|
|
1964
|
+
report["exit_code"] = 2
|
|
1965
|
+
elif blank or dead or nt or report["stale_lock"] or legacy_folders:
|
|
1966
|
+
report["exit_code"] = 1
|
|
1967
|
+
else:
|
|
1968
|
+
report["exit_code"] = 0
|
|
1969
|
+
return report
|
|
1970
|
+
|
|
1971
|
+
|
|
1972
|
+
def cmd_doctor(env, ns):
|
|
1973
|
+
rep = gather_doctor(env)
|
|
1974
|
+
if ns.json:
|
|
1975
|
+
print(json.dumps(rep, indent=1))
|
|
1976
|
+
return rep["exit_code"]
|
|
1977
|
+
def say(line):
|
|
1978
|
+
print(line if ns.verbose else redact(env, line))
|
|
1979
|
+
say("[observed] store: {0} ({1})".format(rep["stores"]["status"],
|
|
1980
|
+
rep["stores"]["detail"]))
|
|
1981
|
+
for r in rep["stores"]["roots"]:
|
|
1982
|
+
say("[observed] root: " + r)
|
|
1983
|
+
say("[observed] listing rows: {0}".format(rep["row_count"]))
|
|
1984
|
+
for e in rep["row_errors"]:
|
|
1985
|
+
say("[observed] UNREADABLE ROW (mutations blocked): " + e)
|
|
1986
|
+
for lid in rep["blank_rows"]:
|
|
1987
|
+
say("[observed] row {0} has a blank cliSessionId".format(lid))
|
|
1988
|
+
say("[hypothesis] the app blanks the link when a transcript goes missing")
|
|
1989
|
+
for d in rep["dead_rows"]:
|
|
1990
|
+
say("[observed] row {0}: transcript missing (last activity {1}d ago)"
|
|
1991
|
+
.format(d["local_id"], d["age_days"]))
|
|
1992
|
+
if d["age_days"] >= RETENTION_HINT_DAYS:
|
|
1993
|
+
say("[hypothesis] age is consistent with the ~30-day retention default")
|
|
1994
|
+
for sid in rep["unlisted_transcripts"]:
|
|
1995
|
+
say("[observed] transcript {0} has no listing row".format(sid))
|
|
1996
|
+
say("[hypothesis] normal for CLI-created sessions; also what an interrupted "
|
|
1997
|
+
"external move leaves behind")
|
|
1998
|
+
say("[observed] encoding evidence (recent 50): current={0} legacy={1}"
|
|
1999
|
+
.format(rep["encoding_recent"]["current"], rep["encoding_recent"]["legacy"]))
|
|
2000
|
+
say("[observed] encoding evidence (all rows): current={0} legacy={1}"
|
|
2001
|
+
.format(rep["encoding"]["current"], rep["encoding"]["legacy"]))
|
|
2002
|
+
for msg in rep.get("unknown_layout", []):
|
|
2003
|
+
say("[observed] " + msg)
|
|
2004
|
+
for lf in rep["legacy_folders"]:
|
|
2005
|
+
say("[observed] legacy-encoded folder {0} ({1} transcripts) is shadowed"
|
|
2006
|
+
.format(lf["folder"], lf["transcripts"]))
|
|
2007
|
+
for oid in rep["nonterminal_ops"]:
|
|
2008
|
+
say("[observed] unresolved operation {0} - run: claude-code-sessions recover".format(oid))
|
|
2009
|
+
if rep["stale_lock"]:
|
|
2010
|
+
say("[observed] stale lock - run: claude-code-sessions recover")
|
|
2011
|
+
if rep["exit_code"] == 2:
|
|
2012
|
+
say("[observed] unrecognized or unreadable state - please open an issue including the output above (paths and ids are redacted by default)")
|
|
2013
|
+
return rep["exit_code"]
|
|
2014
|
+
|
|
2015
|
+
|
|
2016
|
+
# --------------------------------------------------------------- 7. CLI wiring
|
|
2017
|
+
import argparse
|
|
2018
|
+
|
|
2019
|
+
|
|
2020
|
+
def build_parser():
|
|
2021
|
+
p = argparse.ArgumentParser(prog="claude-code-sessions",
|
|
2022
|
+
description="Inspect and relocate Claude Code sessions on disk. Unofficial; "
|
|
2023
|
+
"fails closed. Close the Claude app before any mutation.")
|
|
2024
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
2025
|
+
|
|
2026
|
+
def common(sp):
|
|
2027
|
+
sp.add_argument("--verbose", action="store_true",
|
|
2028
|
+
help="full paths and ids (default output is redacted)")
|
|
2029
|
+
|
|
2030
|
+
sp = sub.add_parser("list", help="inventory sessions")
|
|
2031
|
+
sp.add_argument("query", nargs="?", default="")
|
|
2032
|
+
sp.add_argument("--project")
|
|
2033
|
+
sp.add_argument("--json", action="store_true")
|
|
2034
|
+
sp.add_argument("--full", action="store_true")
|
|
2035
|
+
common(sp)
|
|
2036
|
+
|
|
2037
|
+
sp = sub.add_parser("doctor", help="read-only health report")
|
|
2038
|
+
sp.add_argument("--json", action="store_true")
|
|
2039
|
+
common(sp)
|
|
2040
|
+
|
|
2041
|
+
sp = sub.add_parser("move", help="relocate a session to another project folder")
|
|
2042
|
+
sp.add_argument("session_id")
|
|
2043
|
+
sp.add_argument("target")
|
|
2044
|
+
sp.add_argument("--apply", action="store_true")
|
|
2045
|
+
sp.add_argument("--transcript-only", action="store_true", dest="transcript_only")
|
|
2046
|
+
sp.add_argument("--row", action="append", default=[])
|
|
2047
|
+
sp.add_argument("--yes", action="store_true")
|
|
2048
|
+
sp.add_argument("--force", action="store_true")
|
|
2049
|
+
common(sp)
|
|
2050
|
+
|
|
2051
|
+
sp = sub.add_parser("undo", help="reverse the most recent operation")
|
|
2052
|
+
sp.add_argument("--list", action="store_true", dest="show")
|
|
2053
|
+
sp.add_argument("--id", dest="op_id")
|
|
2054
|
+
sp.add_argument("--apply", action="store_true")
|
|
2055
|
+
common(sp)
|
|
2056
|
+
|
|
2057
|
+
sp = sub.add_parser("recover", help="resolve interrupted operations")
|
|
2058
|
+
sp.add_argument("--resolve", dest="op_id")
|
|
2059
|
+
direction = sp.add_mutually_exclusive_group() # M2: --forward/--back are exclusive
|
|
2060
|
+
direction.add_argument("--forward", action="store_true")
|
|
2061
|
+
direction.add_argument("--back", action="store_true")
|
|
2062
|
+
sp.add_argument("--apply", action="store_true")
|
|
2063
|
+
common(sp)
|
|
2064
|
+
|
|
2065
|
+
sp = sub.add_parser("sync", help="copy session listing rows to your other account")
|
|
2066
|
+
sp.add_argument("--to", default="", metavar="SUBSTRING",
|
|
2067
|
+
help="destination account id, org id, store path, or email "
|
|
2068
|
+
"(required if more than one exists)")
|
|
2069
|
+
sp.add_argument("--only", default="", metavar="SUBSTRING",
|
|
2070
|
+
help="only sessions whose title contains this")
|
|
2071
|
+
sp.add_argument("--include-deleted", action="append", default=[],
|
|
2072
|
+
dest="include_deleted", metavar="TITLE_OR_ID",
|
|
2073
|
+
help="also copy this session even though the destination "
|
|
2074
|
+
"deleted it (names one session; not a blanket switch)")
|
|
2075
|
+
sp.add_argument("--verbatim", action="store_true",
|
|
2076
|
+
help="copy rows unchanged instead of stripping connector config")
|
|
2077
|
+
sp.add_argument("--apply", action="store_true")
|
|
2078
|
+
sp.add_argument("--json", action="store_true")
|
|
2079
|
+
common(sp)
|
|
2080
|
+
return p
|
|
2081
|
+
|
|
2082
|
+
|
|
2083
|
+
def _flags_from(ns):
|
|
2084
|
+
return MoveFlags(transcript_only=ns.transcript_only,
|
|
2085
|
+
row=ns.row, yes=ns.yes, force=ns.force)
|
|
2086
|
+
|
|
2087
|
+
|
|
2088
|
+
def _print_abort_reason(env, ns, op):
|
|
2089
|
+
"""I3: a rollback that completes silently (no exception - e.g. a
|
|
2090
|
+
phase-6 keep-both abort) gives the user no clue anything unusual
|
|
2091
|
+
happened beyond the bare word "rolled_back". If the op's manifest
|
|
2092
|
+
carries an abort_reason (set by _abort/execute_op), surface it - and,
|
|
2093
|
+
when the destination copy was deliberately kept, name it too.
|
|
2094
|
+
"""
|
|
2095
|
+
reason = op.manifest.get("abort_reason")
|
|
2096
|
+
if not reason:
|
|
2097
|
+
return
|
|
2098
|
+
line = "reason: " + reason
|
|
2099
|
+
if op.manifest.get("abort_keep_dest"):
|
|
2100
|
+
line += "; both copies were kept (destination retained at {0})".format(
|
|
2101
|
+
op.manifest.get("dest_transcript", ""))
|
|
2102
|
+
print(line if ns.verbose else redact(env, line))
|
|
2103
|
+
|
|
2104
|
+
|
|
2105
|
+
def _print_new_op_reason(env, ns, before_ids):
|
|
2106
|
+
"""run_move/run_undo return only a plain status string, not the Op they
|
|
2107
|
+
created - so to print its abort reason (I3) after a non-completed
|
|
2108
|
+
result, find the op that appeared since `before_ids` was snapshotted.
|
|
2109
|
+
Safe because callers hold the single-instance lock for the duration of
|
|
2110
|
+
the call that created it, so at most one new op can have appeared."""
|
|
2111
|
+
for op in list_ops(env):
|
|
2112
|
+
if op.manifest["op_id"] not in before_ids:
|
|
2113
|
+
_print_abort_reason(env, ns, op)
|
|
2114
|
+
|
|
2115
|
+
|
|
2116
|
+
def cmd_move(env, ns):
|
|
2117
|
+
manifest = plan_move(env, ns.session_id, ns.target, _flags_from(ns))
|
|
2118
|
+
summary = ("mode={0}\nsource={1}\ndest={2}\nrows={3}"
|
|
2119
|
+
.format(manifest["mode"], manifest["source_transcript"],
|
|
2120
|
+
manifest["dest_transcript"], len(manifest["rows"])))
|
|
2121
|
+
print(summary if ns.verbose else redact(env, summary))
|
|
2122
|
+
if not ns.apply:
|
|
2123
|
+
print("dry run - pass --apply to execute")
|
|
2124
|
+
return 0
|
|
2125
|
+
before_ids = {o.manifest["op_id"] for o in list_ops(env)}
|
|
2126
|
+
final = run_move(env, manifest)
|
|
2127
|
+
print("result: " + final)
|
|
2128
|
+
if final != "completed":
|
|
2129
|
+
_print_new_op_reason(env, ns, before_ids)
|
|
2130
|
+
return 0 if final == "completed" else 1
|
|
2131
|
+
|
|
2132
|
+
|
|
2133
|
+
def cmd_undo(env, ns):
|
|
2134
|
+
ops = list_ops(env)
|
|
2135
|
+
if ns.show:
|
|
2136
|
+
for o in ops:
|
|
2137
|
+
line = "{0} {1:12} {2}".format(o.manifest["op_id"],
|
|
2138
|
+
o.manifest["status"],
|
|
2139
|
+
o.manifest.get("session_id", ""))
|
|
2140
|
+
print(line if ns.verbose else redact(env, line))
|
|
2141
|
+
return 0
|
|
2142
|
+
# delta: only a completed op whose op_type is "move" or "sync" (or
|
|
2143
|
+
# missing, which in practice never happens - every manifest sets
|
|
2144
|
+
# op_type) is eligible as "the operation to undo". A completed *undo*
|
|
2145
|
+
# op is itself terminal from cmd_undo's point of view - plan_undo
|
|
2146
|
+
# always refuses an undo-of-undo ("to redo, run move again") - so
|
|
2147
|
+
# selecting one here would only ever produce that refusal instead of
|
|
2148
|
+
# reaching an older, still-undoable completed move/sync underneath it.
|
|
2149
|
+
candidates = [o for o in ops if o.manifest.get("status") == "completed"
|
|
2150
|
+
and o.manifest.get("op_type", "move") in ("move", "sync")]
|
|
2151
|
+
if ns.op_id:
|
|
2152
|
+
candidates = [o for o in candidates if o.manifest["op_id"] == ns.op_id]
|
|
2153
|
+
if not candidates:
|
|
2154
|
+
raise Refusal("no completed operation to undo" +
|
|
2155
|
+
(" with id " + ns.op_id if ns.op_id else ""))
|
|
2156
|
+
prior = candidates[-1]
|
|
2157
|
+
if not ns.apply:
|
|
2158
|
+
if prior.manifest.get("op_type") == "sync":
|
|
2159
|
+
# A sync manifest has no session_id - the move-shaped preview
|
|
2160
|
+
# below would print "session None". Name what undo would
|
|
2161
|
+
# actually remove instead: how many rows landed, and where.
|
|
2162
|
+
n_written = sum(1 for r in prior.manifest.get("rows", []) if r.get("written"))
|
|
2163
|
+
dest = prior.manifest.get("dest_email") or prior.manifest.get("dest_account", "")
|
|
2164
|
+
line = ("would undo {0} (sync: {1} row(s) written to {2}); pass --apply "
|
|
2165
|
+
"to execute".format(prior.manifest["op_id"], n_written, dest))
|
|
2166
|
+
else:
|
|
2167
|
+
line = ("would undo {0} (session {1}); pass --apply to execute"
|
|
2168
|
+
.format(prior.manifest["op_id"], prior.manifest.get("session_id")))
|
|
2169
|
+
print(line if ns.verbose else redact(env, line)) # M1: redact the preview too
|
|
2170
|
+
return 0
|
|
2171
|
+
before_ids = {o.manifest["op_id"] for o in list_ops(env)}
|
|
2172
|
+
if prior.manifest.get("op_type") == "sync":
|
|
2173
|
+
final = undo_sync(env, prior)
|
|
2174
|
+
else:
|
|
2175
|
+
final = run_undo(env, prior)
|
|
2176
|
+
print("result: " + final)
|
|
2177
|
+
# undo_sync's own terminal status is "undone" (it mutates the completed
|
|
2178
|
+
# sync op in place rather than journaling a fresh reversal op the way
|
|
2179
|
+
# run_undo/execute_op do) - "completed" remains the success value for
|
|
2180
|
+
# every move/undo op the engine drives.
|
|
2181
|
+
success = final == "completed" or final == "undone"
|
|
2182
|
+
if not success:
|
|
2183
|
+
_print_new_op_reason(env, ns, before_ids)
|
|
2184
|
+
return 0 if success else 1
|
|
2185
|
+
|
|
2186
|
+
|
|
2187
|
+
def cmd_recover(env, ns):
|
|
2188
|
+
if clear_stale_lock(env):
|
|
2189
|
+
print("cleared a stale lock")
|
|
2190
|
+
pending = nonterminal_ops(env)
|
|
2191
|
+
if not ns.op_id:
|
|
2192
|
+
for op in pending:
|
|
2193
|
+
c = classify_op(env, op)
|
|
2194
|
+
line = "{0} {1:10} source={2} dest={3} options={4} {5}".format(
|
|
2195
|
+
op.manifest["op_id"], c["status"], c["source"], c["dest"],
|
|
2196
|
+
",".join(c["resolutions"]) or "manual", c["note"])
|
|
2197
|
+
print(line if ns.verbose else redact(env, line))
|
|
2198
|
+
return 1 if pending else 0
|
|
2199
|
+
matches = [o for o in pending if o.manifest["op_id"] == ns.op_id]
|
|
2200
|
+
if not matches:
|
|
2201
|
+
raise Refusal("no unresolved op with id " + ns.op_id)
|
|
2202
|
+
direction = "forward" if ns.forward else ("back" if ns.back else None)
|
|
2203
|
+
if direction is None:
|
|
2204
|
+
raise Refusal("--resolve needs --forward or --back")
|
|
2205
|
+
if not ns.apply:
|
|
2206
|
+
print("would resolve {0} {1}; pass --apply to execute".format(ns.op_id, direction))
|
|
2207
|
+
return 0
|
|
2208
|
+
final = recover_op(env, matches[0], direction)
|
|
2209
|
+
print("result: " + final)
|
|
2210
|
+
if final != "completed":
|
|
2211
|
+
_print_abort_reason(env, ns, matches[0])
|
|
2212
|
+
return 0
|
|
2213
|
+
|
|
2214
|
+
|
|
2215
|
+
def main(argv=None):
|
|
2216
|
+
ns = build_parser().parse_args(argv)
|
|
2217
|
+
env = default_env()
|
|
2218
|
+
handlers = {"list": cmd_list, "doctor": cmd_doctor, "move": cmd_move,
|
|
2219
|
+
"undo": cmd_undo, "recover": cmd_recover, "sync": cmd_sync}
|
|
2220
|
+
try:
|
|
2221
|
+
return handlers[ns.cmd](env, ns)
|
|
2222
|
+
except (Refusal, LayoutError) as exc:
|
|
2223
|
+
label = "refused" if isinstance(exc, Refusal) else "unsafe"
|
|
2224
|
+
msg = str(exc) if getattr(ns, "verbose", False) else redact(env, str(exc))
|
|
2225
|
+
print("{0}: {1}".format(label, msg), file=sys.stderr)
|
|
2226
|
+
return exc.exit_code
|
|
2227
|
+
|
|
2228
|
+
|
|
2229
|
+
# ----------------------------------------------------------------- 8. sync
|
|
2230
|
+
@dataclasses.dataclass
|
|
2231
|
+
class Account:
|
|
2232
|
+
account_uuid: str
|
|
2233
|
+
org_uuid: str
|
|
2234
|
+
email: str
|
|
2235
|
+
path: str
|
|
2236
|
+
# How live_account() decided this was the signed-in account:
|
|
2237
|
+
# "oauth" - ~/.claude.json's oauthAccount named it outright.
|
|
2238
|
+
# "config" - only config.json's lastKnownAccountUuid named it.
|
|
2239
|
+
# "" - not a live-account determination at all (every dormant
|
|
2240
|
+
# candidate resolve_sync_endpoints builds).
|
|
2241
|
+
# Since RULING 4 (2026-08-02) provenance buys no guard exemption - E4
|
|
2242
|
+
# measured oauth stale across a real switch - it is kept for messages
|
|
2243
|
+
# and diagnostics only; every mutation route takes _guard_mutation.
|
|
2244
|
+
resolved_from: str = ""
|
|
2245
|
+
|
|
2246
|
+
|
|
2247
|
+
def _listdir_or_refuse(path, what):
|
|
2248
|
+
"""os.listdir, but a failure is a LayoutError rather than a raw OSError.
|
|
2249
|
+
|
|
2250
|
+
discover_stores proves only that the store ROOT is enumerable; every
|
|
2251
|
+
listdir deeper than that (account dirs, org dirs, the two store folders
|
|
2252
|
+
sync reads) can still hit a PermissionError, and main() catches only
|
|
2253
|
+
Refusal/LayoutError - so a bare OSError escapes as an unredacted
|
|
2254
|
+
traceback carrying full paths and account uuids. Same fail-closed rule
|
|
2255
|
+
the rest of the module uses: "couldn't look" is never "nothing there"."""
|
|
2256
|
+
try:
|
|
2257
|
+
return os.listdir(path)
|
|
2258
|
+
except OSError as exc:
|
|
2259
|
+
raise LayoutError("could not read {0} at {1}: {2}. 'Couldn't look' is never "
|
|
2260
|
+
"'nothing there' - refusing.".format(what, path, exc))
|
|
2261
|
+
|
|
2262
|
+
|
|
2263
|
+
def _account_dirs(env):
|
|
2264
|
+
"""Every <accountUuid>/<organizationUuid> pair present on disk."""
|
|
2265
|
+
disc = discover_stores(env)
|
|
2266
|
+
if disc.status == "error":
|
|
2267
|
+
raise LayoutError("store discovery failed: {0}. 'Couldn't look' is never "
|
|
2268
|
+
"'nothing there' - refusing.".format(disc.detail))
|
|
2269
|
+
out = []
|
|
2270
|
+
for root in disc.roots:
|
|
2271
|
+
for acct in sorted(_listdir_or_refuse(root, "the store root")):
|
|
2272
|
+
ap = os.path.join(root, acct)
|
|
2273
|
+
if not os.path.isdir(ap):
|
|
2274
|
+
continue
|
|
2275
|
+
for org in sorted(_listdir_or_refuse(ap, "an account directory")):
|
|
2276
|
+
op = os.path.join(ap, org)
|
|
2277
|
+
if os.path.isdir(op):
|
|
2278
|
+
out.append((acct, org, op))
|
|
2279
|
+
return out
|
|
2280
|
+
|
|
2281
|
+
|
|
2282
|
+
def _identity_disagreement(env):
|
|
2283
|
+
"""(oauth_uuid, config_uuid) when the two identity files name different
|
|
2284
|
+
accounts; None otherwise.
|
|
2285
|
+
|
|
2286
|
+
Measured 2026-08-02 (E4 verification): across a real desktop account
|
|
2287
|
+
switch, ~/.claude.json's oauthAccount stayed STALE while config.json's
|
|
2288
|
+
lastKnownAccountUuid tracked the switch - the inverse of the trust
|
|
2289
|
+
ordering this module shipped with. The whole-branch review had already
|
|
2290
|
+
built the opposite case (config stale, oauth fresh) synthetically. So
|
|
2291
|
+
either file can be the stale one; a disagreement between them means the
|
|
2292
|
+
live account is genuinely unknowable from files, and callers must fail
|
|
2293
|
+
closed rather than pick a side. The likely mechanism (unverified): the
|
|
2294
|
+
CLI owns ~/.claude.json, the desktop owns config.json, so each kind of
|
|
2295
|
+
sign-in freshens only its own file.
|
|
2296
|
+
|
|
2297
|
+
An unreadable or malformed file is NO SIGNAL, deliberately - oauth-only
|
|
2298
|
+
and config-only are legitimate states, not failures, and after RULING 4
|
|
2299
|
+
the safety of every mutation rests on the universal process guard
|
|
2300
|
+
(_guard_mutation), not on this comparison. Both values must be
|
|
2301
|
+
non-empty strings; anything else is treated as absent so a garbage
|
|
2302
|
+
value can never traceback later inside a refusal message's [:8] slice.
|
|
2303
|
+
"""
|
|
2304
|
+
try:
|
|
2305
|
+
with open(os.path.join(env.home, ".claude.json"), encoding="utf-8") as fh:
|
|
2306
|
+
oauth = ((json.load(fh) or {}).get("oauthAccount") or {}).get("accountUuid")
|
|
2307
|
+
except (OSError, ValueError, AttributeError, TypeError):
|
|
2308
|
+
oauth = None
|
|
2309
|
+
if not isinstance(oauth, str) or not oauth:
|
|
2310
|
+
return None
|
|
2311
|
+
for cand in env.store_candidates:
|
|
2312
|
+
cfg = os.path.join(os.path.dirname(cand), "config.json")
|
|
2313
|
+
try:
|
|
2314
|
+
with open(cfg, encoding="utf-8") as fh:
|
|
2315
|
+
last = (json.load(fh) or {}).get("lastKnownAccountUuid")
|
|
2316
|
+
except (OSError, ValueError, AttributeError, TypeError):
|
|
2317
|
+
continue
|
|
2318
|
+
if isinstance(last, str) and last and last != oauth:
|
|
2319
|
+
return (oauth, last)
|
|
2320
|
+
return None
|
|
2321
|
+
|
|
2322
|
+
|
|
2323
|
+
def live_account(env):
|
|
2324
|
+
"""The signed-in account, named outright rather than guessed.
|
|
2325
|
+
|
|
2326
|
+
~/.claude.json's oauthAccount carries accountUuid, organizationUuid AND
|
|
2327
|
+
emailAddress, which resolves the whole store path and gives the user a
|
|
2328
|
+
destination they can recognise. The exact (account, org) pair is
|
|
2329
|
+
preferred; if organizationUuid names a dir that doesn't exist on disk
|
|
2330
|
+
(config known, dir not yet created), fall back to matching the account
|
|
2331
|
+
alone. config.json's lastKnownAccountUuid is the last-resort fallback
|
|
2332
|
+
but names only the account half - if more than one org dir sits under
|
|
2333
|
+
that account there is no evidence which is live, so this refuses to
|
|
2334
|
+
guess and returns None rather than picking one.
|
|
2335
|
+
|
|
2336
|
+
The returned Account records WHICH path answered, in `resolved_from` -
|
|
2337
|
+
kept for messages and diagnostics. It no longer gates anything: RULING 4
|
|
2338
|
+
(2026-08-02) put the running-app check on every mutation route after E4
|
|
2339
|
+
measured oauthAccount stale across a real switch (see _guard_mutation
|
|
2340
|
+
and _identity_disagreement).
|
|
2341
|
+
"""
|
|
2342
|
+
if _identity_disagreement(env):
|
|
2343
|
+
return None # fail closed - see _identity_disagreement
|
|
2344
|
+
try:
|
|
2345
|
+
with open(os.path.join(env.home, ".claude.json"), encoding="utf-8") as fh:
|
|
2346
|
+
oa = (json.load(fh) or {}).get("oauthAccount") or {}
|
|
2347
|
+
except (OSError, ValueError, AttributeError, TypeError):
|
|
2348
|
+
oa = {}
|
|
2349
|
+
if not isinstance(oa, dict):
|
|
2350
|
+
oa = {}
|
|
2351
|
+
dirs = _account_dirs(env)
|
|
2352
|
+
acct_uuid = oa.get("accountUuid")
|
|
2353
|
+
if acct_uuid:
|
|
2354
|
+
org_uuid = oa.get("organizationUuid")
|
|
2355
|
+
exact = [(a, o, p) for a, o, p in dirs if a == acct_uuid and o == org_uuid]
|
|
2356
|
+
if exact:
|
|
2357
|
+
a, o, p = exact[0]
|
|
2358
|
+
return Account(a, o, oa.get("emailAddress") or "", p, "oauth")
|
|
2359
|
+
for a, o, p in dirs:
|
|
2360
|
+
if a == acct_uuid:
|
|
2361
|
+
return Account(a, o, oa.get("emailAddress") or "", p, "oauth")
|
|
2362
|
+
for cand in env.store_candidates:
|
|
2363
|
+
cfg = os.path.join(os.path.dirname(cand), "config.json")
|
|
2364
|
+
try:
|
|
2365
|
+
with open(cfg, encoding="utf-8") as fh:
|
|
2366
|
+
last = (json.load(fh) or {}).get("lastKnownAccountUuid")
|
|
2367
|
+
except (OSError, ValueError, AttributeError, TypeError):
|
|
2368
|
+
continue
|
|
2369
|
+
if last:
|
|
2370
|
+
matches = [(a, o, p) for a, o, p in dirs if a == last]
|
|
2371
|
+
if len(matches) == 1:
|
|
2372
|
+
a, o, p = matches[0]
|
|
2373
|
+
return Account(a, o, "", p, "config")
|
|
2374
|
+
if len(matches) > 1:
|
|
2375
|
+
return None # ambiguous org under this account - fail closed
|
|
2376
|
+
# zero matches: this candidate's config names an account with no
|
|
2377
|
+
# store dir on disk yet - try the next store candidate's config
|
|
2378
|
+
return None
|
|
2379
|
+
|
|
2380
|
+
|
|
2381
|
+
def _require_verified_platform(env, what):
|
|
2382
|
+
"""Refuse desktop-store mutations on a platform whose layout is unverified.
|
|
2383
|
+
|
|
2384
|
+
There is deliberately NO override flag. The store layout is confirmed only
|
|
2385
|
+
on Windows; macOS reportedly has two candidate layouts - the ordinary
|
|
2386
|
+
Application Support path and a sandboxed ~/Library/Containers/... one - and
|
|
2387
|
+
neither has been confirmed here. An override would let a user waive a risk
|
|
2388
|
+
they have no way to evaluate, which inverts how every other refusal in this
|
|
2389
|
+
module works: we fail closed on what we cannot verify rather than asking the
|
|
2390
|
+
user to certify it for us.
|
|
2391
|
+
|
|
2392
|
+
Unaffected: read-only commands, and --transcript-only mutations - that
|
|
2393
|
+
layout IS verified cross-platform.
|
|
2394
|
+
"""
|
|
2395
|
+
if env.is_windows:
|
|
2396
|
+
return
|
|
2397
|
+
raise Refusal(
|
|
2398
|
+
"desktop-store mutations are Windows-only for now - this platform's store "
|
|
2399
|
+
"layout has never been verified, so refusing to {0} it. Read-only commands "
|
|
2400
|
+
"(list, doctor) work here, and a session with no desktop listing row (a "
|
|
2401
|
+
"CLI-created one) is still movable via --transcript-only, because THAT "
|
|
2402
|
+
"layout is verified. On macOS and want the desktop store supported? "
|
|
2403
|
+
"'claude-code-sessions doctor --verbose' output in an issue is exactly what "
|
|
2404
|
+
"is needed - it is read-only and mutates nothing.".format(what))
|
|
2405
|
+
|
|
2406
|
+
|
|
2407
|
+
def _guard_mutation(env, what):
|
|
2408
|
+
"""Refuse to WHAT another account's store while the Claude desktop app
|
|
2409
|
+
is running. Applies to every mutation route, whatever named the live
|
|
2410
|
+
account.
|
|
2411
|
+
|
|
2412
|
+
RULING 4 (2026-08-02). The E4 verification measured ~/.claude.json's
|
|
2413
|
+
oauthAccount STALE across a real desktop account switch while
|
|
2414
|
+
config.json's lastKnownAccountUuid tracked it - the inverse of the
|
|
2415
|
+
ordering this module shipped trusting, and the whole-branch review had
|
|
2416
|
+
already built the opposite case synthetically. Either identity file can
|
|
2417
|
+
be the stale one, so no file evidence is allowed to certify "the
|
|
2418
|
+
destination is dormant" while the app runs; the oauth exemption this
|
|
2419
|
+
function used to carry is gone. claude_running is narrowed to the
|
|
2420
|
+
desktop app's own processes, so a Claude Code CLI session never trips
|
|
2421
|
+
this.
|
|
2422
|
+
"""
|
|
2423
|
+
_require_verified_platform(env, what)
|
|
2424
|
+
running = claude_running(env)
|
|
2425
|
+
if not running:
|
|
2426
|
+
return
|
|
2427
|
+
dis = _identity_disagreement(env)
|
|
2428
|
+
extra = ""
|
|
2429
|
+
if dis:
|
|
2430
|
+
extra = (
|
|
2431
|
+
"\nAlso: ~/.claude.json ({0}) and config.json ({1}) disagree about the "
|
|
2432
|
+
"signed-in account. Re-authenticate the CLI (run 'claude', then /login) "
|
|
2433
|
+
"as the account you use, or switch the desktop app to it, so the two "
|
|
2434
|
+
"agree.".format(dis[0][:8], dis[1][:8]))
|
|
2435
|
+
if running[0] == _PROC_UNAVAILABLE:
|
|
2436
|
+
# "Couldn't look" is never "nothing there" (Task 2), but it is also
|
|
2437
|
+
# never "the app IS running" - that wording would be a lie here, and
|
|
2438
|
+
# "close the desktop app" is misleading advice when what actually
|
|
2439
|
+
# failed is reading the process list. Say what is really true.
|
|
2440
|
+
raise Refusal(
|
|
2441
|
+
"the running-process list could not be read, so whether the Claude "
|
|
2442
|
+
"desktop app is running cannot be confirmed; refusing to {0} another "
|
|
2443
|
+
"account's store while that is unavailable - re-run once the process "
|
|
2444
|
+
"list can be read.{1}".format(what, extra))
|
|
2445
|
+
raise Refusal(
|
|
2446
|
+
"the Claude desktop app appears to be running ({0}); refusing to {1} "
|
|
2447
|
+
"another account's store while it is. No identity-file evidence can make "
|
|
2448
|
+
"'the destination is dormant' certain enough to mutate under a running "
|
|
2449
|
+
"app - close the desktop app and re-run.{2}".format(running[0], what, extra))
|
|
2450
|
+
|
|
2451
|
+
|
|
2452
|
+
def _refuse_dest_possibly_live(env, live, dest_path, what, live_match_message):
|
|
2453
|
+
"""Shared by execute_sync_op and _sync_delete_targets: refuse when
|
|
2454
|
+
dest_path might be the live account's store, by either of two
|
|
2455
|
+
independent tests. Factored into one place so the two sites cannot
|
|
2456
|
+
drift apart on this - the whole reason this helper exists.
|
|
2457
|
+
|
|
2458
|
+
1. live_account() resolved a live account outright, and dest_path IS
|
|
2459
|
+
that account's store (realpath/normcase both sides, matching
|
|
2460
|
+
ensure_contained - see the callers' own comments on junctions).
|
|
2461
|
+
Unchanged from before Task 1: callers pass their own
|
|
2462
|
+
`live_match_message` (a zero-arg callable, evaluated only on an
|
|
2463
|
+
actual match - so it may safely assume `live` is not None) because
|
|
2464
|
+
the two sites' wording differs (sync's write-side voice vs undo's
|
|
2465
|
+
delete-side voice) and existing tests pin that wording.
|
|
2466
|
+
|
|
2467
|
+
2. live_account() returned None *because the identity files disagree*
|
|
2468
|
+
(_identity_disagreement). Task 1 made None mean this too, not only
|
|
2469
|
+
"no evidence at all" - and a disagreement is not "safe to proceed":
|
|
2470
|
+
either of the two disagreeing accounts could be the one genuinely
|
|
2471
|
+
live. So if dest_path resolves under EITHER named account's store on
|
|
2472
|
+
disk, refuse, naming both 8-char id prefixes and the fix - mirroring
|
|
2473
|
+
_guard_mutation's own disagreement note. dest_path under some THIRD
|
|
2474
|
+
account's store (named by neither uuid) is not covered by this
|
|
2475
|
+
disagreement at all and proceeds, same as today.
|
|
2476
|
+
|
|
2477
|
+
No disagreement and live is None (genuinely no evidence, e.g. no
|
|
2478
|
+
identity file resolves anything) falls through both checks and
|
|
2479
|
+
proceeds - unchanged from before Task 1.
|
|
2480
|
+
"""
|
|
2481
|
+
real_dest = os.path.normcase(os.path.realpath(dest_path))
|
|
2482
|
+
if live is not None:
|
|
2483
|
+
if real_dest == os.path.normcase(os.path.realpath(live.path)):
|
|
2484
|
+
raise Refusal(live_match_message())
|
|
2485
|
+
return
|
|
2486
|
+
dis = _identity_disagreement(env)
|
|
2487
|
+
if dis is None:
|
|
2488
|
+
return
|
|
2489
|
+
oauth_uuid, config_uuid = dis
|
|
2490
|
+
named_dirs = _account_dirs(env)
|
|
2491
|
+
for acct_uuid in (oauth_uuid, config_uuid):
|
|
2492
|
+
for a, o, p in named_dirs:
|
|
2493
|
+
if a == acct_uuid and real_dest == os.path.normcase(os.path.realpath(p)):
|
|
2494
|
+
raise Refusal(
|
|
2495
|
+
"~/.claude.json ({0}) and config.json ({1}) disagree about the "
|
|
2496
|
+
"signed-in account, so which one is actually live is unknowable "
|
|
2497
|
+
"from files alone; the destination matches the store of one of "
|
|
2498
|
+
"those two possibly-live accounts, so refusing to {2}. "
|
|
2499
|
+
"Re-authenticate the CLI (run 'claude', then /login) as the "
|
|
2500
|
+
"account you use, or switch the desktop app to it, so the two "
|
|
2501
|
+
"agree, then re-run.".format(oauth_uuid[:8], config_uuid[:8], what))
|
|
2502
|
+
|
|
2503
|
+
|
|
2504
|
+
def _candidate_line(account_uuid, org_uuid, path):
|
|
2505
|
+
"""One line of a "which store did you mean" listing. The store path is
|
|
2506
|
+
part of it because the 8-char id prefixes alone are not always
|
|
2507
|
+
distinguishing: two store roots (Windows' MSIX path and the classic
|
|
2508
|
+
%APPDATA% path) can hold the same account, and telling the user to "be
|
|
2509
|
+
more specific" while showing two identical lines is a wall, not a
|
|
2510
|
+
refusal. Redacted like everything else by main()'s redact()."""
|
|
2511
|
+
return " {0}/{1} {2}".format(account_uuid[:8], org_uuid[:8], path)
|
|
2512
|
+
|
|
2513
|
+
|
|
2514
|
+
_AGENT_MODE_DIR = "local-agent-mode-sessions"
|
|
2515
|
+
|
|
2516
|
+
|
|
2517
|
+
def dormant_account_email(env, account_uuid):
|
|
2518
|
+
"""Best-effort email for an account that is NOT signed in, or "".
|
|
2519
|
+
|
|
2520
|
+
`oauthAccount` in ~/.claude.json names only the live account, so for a
|
|
2521
|
+
long time this returned nothing and the dry run printed "(email unknown)"
|
|
2522
|
+
for the destination - eight hex characters to identify the account you are
|
|
2523
|
+
about to write into, which is a poor safety surface for the one command
|
|
2524
|
+
that touches a second account.
|
|
2525
|
+
|
|
2526
|
+
It is recoverable. The desktop app runs local agent mode inside a
|
|
2527
|
+
per-account sandbox and drops a Claude Code config in it, at
|
|
2528
|
+
`<AGENT_MODE_DIR>/<accountUuid>/<orgUuid>/**/.claude/.claude.json`, whose
|
|
2529
|
+
own `oauthAccount` names THAT account. Observed on Windows, August 2026.
|
|
2530
|
+
|
|
2531
|
+
Deliberately best-effort, and it must stay that way: the directory only
|
|
2532
|
+
exists for an account that has used local agent mode (of the two accounts
|
|
2533
|
+
it was found on, one had 109 such files and the other none), it is a
|
|
2534
|
+
nested implementation detail of a feature we do not otherwise touch, and
|
|
2535
|
+
it can move. Any failure means "unknown", never an error - this only ever
|
|
2536
|
+
improves a label.
|
|
2537
|
+
|
|
2538
|
+
The account uuid inside the file must match the one asked for. Reading a
|
|
2539
|
+
config and trusting its email without that check would let an unrelated
|
|
2540
|
+
sandbox mislabel an account, which is worse than no label at all.
|
|
2541
|
+
"""
|
|
2542
|
+
for root in getattr(env, "store_candidates", ()) or ():
|
|
2543
|
+
base = os.path.join(os.path.dirname(root), _AGENT_MODE_DIR, account_uuid)
|
|
2544
|
+
for dirpath, dirnames, filenames in os.walk(base):
|
|
2545
|
+
if ".claude.json" not in filenames:
|
|
2546
|
+
continue
|
|
2547
|
+
if os.path.basename(dirpath) != ".claude":
|
|
2548
|
+
continue
|
|
2549
|
+
try:
|
|
2550
|
+
oa = (read_json(os.path.join(dirpath, ".claude.json"))
|
|
2551
|
+
or {}).get("oauthAccount") or {}
|
|
2552
|
+
except (LayoutError, OSError, ValueError, AttributeError):
|
|
2553
|
+
continue
|
|
2554
|
+
if oa.get("accountUuid") == account_uuid and oa.get("emailAddress"):
|
|
2555
|
+
return oa["emailAddress"]
|
|
2556
|
+
return ""
|
|
2557
|
+
|
|
2558
|
+
|
|
2559
|
+
def resolve_sync_endpoints(env, to=None):
|
|
2560
|
+
"""(source, destination). Source is the signed-in account; destination is
|
|
2561
|
+
the other store. Refuses rather than guessing - row-freshness is NEVER
|
|
2562
|
+
used to choose, because sync's whole safety model is 'we only ever write
|
|
2563
|
+
the dormant store', and a wrong guess writes the live one."""
|
|
2564
|
+
dirs = _account_dirs(env)
|
|
2565
|
+
source = live_account(env)
|
|
2566
|
+
if source is None:
|
|
2567
|
+
listing = "\n".join(_candidate_line(a, o, p) for a, o, p in dirs)
|
|
2568
|
+
dis = _identity_disagreement(env)
|
|
2569
|
+
if dis:
|
|
2570
|
+
raise Refusal(
|
|
2571
|
+
"cannot identify the signed-in account: ~/.claude.json's oauthAccount "
|
|
2572
|
+
"({0}) and config.json's lastKnownAccountUuid ({1}) disagree, and either "
|
|
2573
|
+
"can be the stale one - refusing to guess which store is live.\n"
|
|
2574
|
+
"--to cannot override this: it names the destination, and without knowing\n"
|
|
2575
|
+
"which account is live we cannot verify the one you named is not it.\n"
|
|
2576
|
+
"Fix: re-authenticate the CLI (run 'claude', then /login) as the account\n"
|
|
2577
|
+
"you are using, or switch the desktop app to that account, so the two\n"
|
|
2578
|
+
"files agree.\n"
|
|
2579
|
+
"Stores found:\n".format(dis[0][:8], dis[1][:8]) + listing)
|
|
2580
|
+
raise Refusal(
|
|
2581
|
+
"cannot identify the signed-in account from ~/.claude.json or config.json.\n"
|
|
2582
|
+
"Refusing to guess which store is live - naming the wrong one would write the\n"
|
|
2583
|
+
"account the app is actively using.\n"
|
|
2584
|
+
"--to cannot override this: it names the destination, and without knowing\n"
|
|
2585
|
+
"which account is live we cannot verify the one you named is not it.\n"
|
|
2586
|
+
"Fix: sign in to the Claude desktop app (which writes config.json) or\n"
|
|
2587
|
+
"authenticate the CLI (which writes ~/.claude.json) so one of them names\n"
|
|
2588
|
+
"the account.\n"
|
|
2589
|
+
"Stores found:\n" + listing)
|
|
2590
|
+
others = [Account(a, o, dormant_account_email(env, a), p)
|
|
2591
|
+
for a, o, p in dirs if a != source.account_uuid]
|
|
2592
|
+
if not others:
|
|
2593
|
+
raise Refusal("no other account store on this machine - nothing to sync into")
|
|
2594
|
+
if to:
|
|
2595
|
+
# The PATH is part of the match string, not just the ids and email.
|
|
2596
|
+
# default_env legitimately yields two store roots on Windows (the MSIX
|
|
2597
|
+
# package path and the classic %APPDATA%\Claude path), and a machine
|
|
2598
|
+
# that migrated between installers can hold the SAME account uuids
|
|
2599
|
+
# under both - in which case account_uuid/org_uuid/email are identical
|
|
2600
|
+
# for both candidates and no --to value could ever tell them apart.
|
|
2601
|
+
# The path is the only thing that differs, so it has to be matchable
|
|
2602
|
+
# (and, below, printed) or sync is simply unusable on such a machine.
|
|
2603
|
+
matched = [c for c in others if to.lower() in
|
|
2604
|
+
(c.account_uuid + " " + c.org_uuid + " " + c.email + " " +
|
|
2605
|
+
c.path).lower()]
|
|
2606
|
+
if not matched:
|
|
2607
|
+
raise Refusal("--to {0!r} matched no other account store".format(to))
|
|
2608
|
+
if len(matched) > 1:
|
|
2609
|
+
listing = "\n".join(_candidate_line(c.account_uuid, c.org_uuid, c.path)
|
|
2610
|
+
for c in matched)
|
|
2611
|
+
raise Refusal("--to {0!r} matched {1} accounts; be more specific (a longer "
|
|
2612
|
+
"id, or part of the store path):\n{2}"
|
|
2613
|
+
.format(to, len(matched), listing))
|
|
2614
|
+
return source, matched[0]
|
|
2615
|
+
if len(others) > 1:
|
|
2616
|
+
listing = "\n".join(_candidate_line(c.account_uuid, c.org_uuid, c.path)
|
|
2617
|
+
for c in others)
|
|
2618
|
+
raise Refusal("more than one other account store; name one with --to:\n" + listing)
|
|
2619
|
+
return source, others[0]
|
|
2620
|
+
|
|
2621
|
+
|
|
2622
|
+
@dataclasses.dataclass
|
|
2623
|
+
class SyncFlags:
|
|
2624
|
+
to: str = ""
|
|
2625
|
+
only: str = ""
|
|
2626
|
+
include_deleted: tuple = ()
|
|
2627
|
+
verbatim: bool = False
|
|
2628
|
+
|
|
2629
|
+
|
|
2630
|
+
def _destination_tombstones(dest):
|
|
2631
|
+
"""Ids the DESTINATION account has deleted. Only the destination's history
|
|
2632
|
+
matters - tombstones are per-account, so the source's deletions say
|
|
2633
|
+
nothing about what this account should see.
|
|
2634
|
+
|
|
2635
|
+
Returns raw ids, deliberately not "session ids": a tombstone is filed
|
|
2636
|
+
under a row's cliSessionId OR its local id, and callers must test both
|
|
2637
|
+
(see _tombstone_ids)."""
|
|
2638
|
+
out = set()
|
|
2639
|
+
for name in _listdir_or_refuse(dest.path, "the destination store"):
|
|
2640
|
+
if name.startswith("deleted_"):
|
|
2641
|
+
out.add(name[len("deleted_"):])
|
|
2642
|
+
return out
|
|
2643
|
+
|
|
2644
|
+
|
|
2645
|
+
def _tombstone_ids(e):
|
|
2646
|
+
"""Both ids a tombstone for this row could be filed under.
|
|
2647
|
+
|
|
2648
|
+
The spec said `deleted_<cliSessionId>`, and that was what E4 measured. It
|
|
2649
|
+
is not the whole truth. On this machine's own live store the session titled
|
|
2650
|
+
'E4 tombstone test' carries TWO tombstones: one named for its cliSessionId
|
|
2651
|
+
(bc7333f9...) and one named for its filename stem, i.e. its local id
|
|
2652
|
+
(747a0b6e...). So the app files deletions in both id spaces, and a skip
|
|
2653
|
+
that checks only the session id can miss a real deletion and resurrect a
|
|
2654
|
+
session the account's user deliberately removed - the first row of this
|
|
2655
|
+
design's own risk table.
|
|
2656
|
+
|
|
2657
|
+
Checking both is safe in the direction that matters. A false positive
|
|
2658
|
+
means declining to copy one row, which the report names and
|
|
2659
|
+
--include-deleted overrides; a false negative resurrects a deletion
|
|
2660
|
+
silently.
|
|
2661
|
+
"""
|
|
2662
|
+
return [i for i in (e.get("session_id"), e.get("local_id")) if i]
|
|
2663
|
+
|
|
2664
|
+
|
|
2665
|
+
def _resolve_tombstone_overrides(entries, tombs, named):
|
|
2666
|
+
"""Map each --include-deleted term to exactly ONE tombstoned source row.
|
|
2667
|
+
|
|
2668
|
+
The flag's contract is that it names a single session and "never applies
|
|
2669
|
+
blanket to a whole run" - but the match used to be a bare title substring
|
|
2670
|
+
tested per row, so one term silently resurrected every tombstoned session
|
|
2671
|
+
whose title happened to contain it (the reviewer got three from one
|
|
2672
|
+
term). Resolve each term against the tombstoned rows up front instead: a
|
|
2673
|
+
full id - either of the two a tombstone can be filed under, see
|
|
2674
|
+
_tombstone_ids - matches exactly; anything else is a title substring and
|
|
2675
|
+
must single one out. More than one match is a refusal that names the
|
|
2676
|
+
candidates - resurrecting a deliberately deleted session is the first row
|
|
2677
|
+
of this design's own risk table and must never happen by accident.
|
|
2678
|
+
|
|
2679
|
+
A term that matches nothing is deliberately NOT an error: the destination
|
|
2680
|
+
may simply hold no tombstone for that session, in which case the row is
|
|
2681
|
+
copied by the ordinary rules and the report's "resurrected" section
|
|
2682
|
+
correctly stays empty. Nothing is claimed that did not happen.
|
|
2683
|
+
|
|
2684
|
+
Returns the set of row filenames whose tombstone skip is overridden.
|
|
2685
|
+
"""
|
|
2686
|
+
out = set()
|
|
2687
|
+
candidates = [e for e in entries
|
|
2688
|
+
if any(i in tombs for i in _tombstone_ids(e))]
|
|
2689
|
+
for term in (named or ()):
|
|
2690
|
+
t = term.lower()
|
|
2691
|
+
matched = [e for e in candidates
|
|
2692
|
+
if t in [i.lower() for i in _tombstone_ids(e)]]
|
|
2693
|
+
if not matched:
|
|
2694
|
+
matched = [e for e in candidates if t in e["title"].lower()]
|
|
2695
|
+
if len(matched) > 1:
|
|
2696
|
+
listing = "\n".join(" {0} (session {1})".format(e["title"], e["session_id"])
|
|
2697
|
+
for e in matched)
|
|
2698
|
+
raise Refusal(
|
|
2699
|
+
"--include-deleted {0!r} matched {1} sessions the destination account "
|
|
2700
|
+
"deleted. It names ONE session; it is not a blanket override. Re-run "
|
|
2701
|
+
"naming a full session id, or a title substring unique to one of:\n{2}"
|
|
2702
|
+
.format(term, len(matched), listing))
|
|
2703
|
+
out.update(e["name"] for e in matched)
|
|
2704
|
+
return out
|
|
2705
|
+
|
|
2706
|
+
|
|
2707
|
+
def select_sync_rows(env, source, dest, flags):
|
|
2708
|
+
"""Which source rows are eligible to copy, and why the rest were skipped.
|
|
2709
|
+
|
|
2710
|
+
A row qualifies only if: it is a local_*.json row (not a tombstone or
|
|
2711
|
+
other sidecar), it is absent from the destination by filename, its
|
|
2712
|
+
transcript still exists somewhere under ~/.claude/projects (a row with no
|
|
2713
|
+
transcript is a dead pointer), and the destination holds no tombstone for
|
|
2714
|
+
its cliSessionId - unless --include-deleted named it unambiguously
|
|
2715
|
+
(_resolve_tombstone_overrides), which overrides the tombstone skip for
|
|
2716
|
+
that row only. Such a row is marked `overrode_tombstone` and listed in
|
|
2717
|
+
tally["resurrected"] so the report can say what it is about to bring
|
|
2718
|
+
back; tally["deleted"] holds only the rows whose deletion was honoured.
|
|
2719
|
+
|
|
2720
|
+
Note that presence is keyed on FILENAME, not cliSessionId: a destination
|
|
2721
|
+
row for the same conversation under a different local id is not detected.
|
|
2722
|
+
"""
|
|
2723
|
+
tally = {"present": [], "no_transcript": [], "deleted": [], "unreadable": [],
|
|
2724
|
+
"filtered": [], "resurrected": []}
|
|
2725
|
+
have = set(_listdir_or_refuse(dest.path, "the destination store"))
|
|
2726
|
+
tombs = _destination_tombstones(dest)
|
|
2727
|
+
|
|
2728
|
+
# Parse first, decide second: --include-deleted has to be resolved
|
|
2729
|
+
# against the whole set of tombstoned rows to know whether a term is
|
|
2730
|
+
# ambiguous, which a single streaming pass cannot see.
|
|
2731
|
+
entries = []
|
|
2732
|
+
for name in sorted(_listdir_or_refuse(source.path, "the source store")):
|
|
2733
|
+
if not (name.startswith("local_") and name.endswith(".json")):
|
|
2734
|
+
continue # scheduled-tasks.json, deleted_*, *.tmp
|
|
2735
|
+
p = os.path.join(source.path, name)
|
|
2736
|
+
try:
|
|
2737
|
+
d = read_json(p)
|
|
2738
|
+
except LayoutError:
|
|
2739
|
+
tally["unreadable"].append(name)
|
|
2740
|
+
continue
|
|
2741
|
+
if not isinstance(d, dict):
|
|
2742
|
+
tally["unreadable"].append(name)
|
|
2743
|
+
continue
|
|
2744
|
+
entries.append({"name": name, "src_path": p, "data": d,
|
|
2745
|
+
"session_id": d.get("cliSessionId") or "",
|
|
2746
|
+
# The row's OTHER id: the filename stem, which is the
|
|
2747
|
+
# local id, not the session id. Tombstones are written
|
|
2748
|
+
# in both spaces - see _tombstone_ids.
|
|
2749
|
+
"local_id": name[len("local_"):-len(".json")],
|
|
2750
|
+
"title": d.get("title") or "(untitled)",
|
|
2751
|
+
"last_activity": d.get("lastActivityAt") or 0})
|
|
2752
|
+
overridden = _resolve_tombstone_overrides(entries, tombs, flags.include_deleted)
|
|
2753
|
+
|
|
2754
|
+
picked = []
|
|
2755
|
+
for e in entries:
|
|
2756
|
+
name, sid, title = e["name"], e["session_id"], e["title"]
|
|
2757
|
+
if flags.only and flags.only.lower() not in title.lower():
|
|
2758
|
+
tally["filtered"].append(title)
|
|
2759
|
+
continue
|
|
2760
|
+
if name in have:
|
|
2761
|
+
tally["present"].append(title)
|
|
2762
|
+
continue
|
|
2763
|
+
if not sid or not find_transcripts(env.projects_root, sid):
|
|
2764
|
+
tally["no_transcript"].append(title)
|
|
2765
|
+
continue
|
|
2766
|
+
overrode = False
|
|
2767
|
+
if any(i in tombs for i in _tombstone_ids(e)):
|
|
2768
|
+
# E4: the app shows a restored row for a deleted session, so this
|
|
2769
|
+
# skip is the only thing preventing a resurrection.
|
|
2770
|
+
if name not in overridden:
|
|
2771
|
+
tally["deleted"].append(title)
|
|
2772
|
+
continue
|
|
2773
|
+
overrode = True
|
|
2774
|
+
tally["resurrected"].append(title)
|
|
2775
|
+
picked.append({"name": name, "src_path": e["src_path"], "data": e["data"],
|
|
2776
|
+
"session_id": sid, "title": title,
|
|
2777
|
+
"last_activity": e["last_activity"],
|
|
2778
|
+
"overrode_tombstone": overrode})
|
|
2779
|
+
picked.sort(key=lambda r: r["last_activity"], reverse=True)
|
|
2780
|
+
return picked, tally
|
|
2781
|
+
|
|
2782
|
+
|
|
2783
|
+
# E5: stripping these took a real row from 132,264 to 715 bytes with the
|
|
2784
|
+
# sidebar, history, responses and connectors all unaffected - the app sources
|
|
2785
|
+
# connectors from the destination account's own configuration, so the row's
|
|
2786
|
+
# copy is redundant baggage that would otherwise disclose which integrations
|
|
2787
|
+
# the source account has and where their endpoints are.
|
|
2788
|
+
SYNC_STRIP = ("remoteMcpServersConfig", "enabledMcpTools", "bridgeSessionIds",
|
|
2789
|
+
"scheduledTaskId")
|
|
2790
|
+
|
|
2791
|
+
# A permission granted under one login was never granted under the other.
|
|
2792
|
+
# NOT yet measured (the E5 row carried no non-default permission state); if the
|
|
2793
|
+
# app dislikes the defaults the failure mode is a re-prompt, not a leak.
|
|
2794
|
+
SYNC_RESET = {"alwaysAllowedReasons": [], "sessionPermissionUpdates": [],
|
|
2795
|
+
"chromePermissionMode": None, "chromeTabGroupId": None}
|
|
2796
|
+
|
|
2797
|
+
|
|
2798
|
+
def transform_row(data, verbatim=False):
|
|
2799
|
+
"""Serialize a row for the destination account. Returns (bytes, removed, reset).
|
|
2800
|
+
|
|
2801
|
+
Never mutates the caller's dict - selection holds the originals and a
|
|
2802
|
+
dry run must be able to report without changing anything.
|
|
2803
|
+
"""
|
|
2804
|
+
if verbatim:
|
|
2805
|
+
return json.dumps(data, separators=(",", ":")).encode("utf-8"), [], []
|
|
2806
|
+
out = dict(data)
|
|
2807
|
+
# Sorted, not SYNC_STRIP declaration order: `reset` below is sorted() too,
|
|
2808
|
+
# and the JSON manifest (--json) surfaces both lists on the same row - a
|
|
2809
|
+
# reviewer flagged the mismatched conventions as a stability trap for
|
|
2810
|
+
# anything that reads or diffs that output.
|
|
2811
|
+
removed = sorted(k for k in SYNC_STRIP if k in out)
|
|
2812
|
+
for k in removed:
|
|
2813
|
+
out.pop(k)
|
|
2814
|
+
reset = []
|
|
2815
|
+
for k, v in SYNC_RESET.items():
|
|
2816
|
+
if k in out and out[k] != v:
|
|
2817
|
+
out[k] = v
|
|
2818
|
+
reset.append(k)
|
|
2819
|
+
return json.dumps(out, separators=(",", ":")).encode("utf-8"), removed, sorted(reset)
|
|
2820
|
+
|
|
2821
|
+
|
|
2822
|
+
def plan_sync(env, flags):
|
|
2823
|
+
"""Build the sync manifest. Pure planning - writes nothing."""
|
|
2824
|
+
source, dest = resolve_sync_endpoints(env, flags.to or None)
|
|
2825
|
+
picked, tally = select_sync_rows(env, source, dest, flags)
|
|
2826
|
+
rows = []
|
|
2827
|
+
for cand in picked:
|
|
2828
|
+
blob, removed, reset = transform_row(cand["data"], flags.verbatim)
|
|
2829
|
+
rows.append({"name": cand["name"],
|
|
2830
|
+
"dest_path": os.path.join(dest.path, cand["name"]),
|
|
2831
|
+
"post_b64": b64(blob), "session_id": cand["session_id"],
|
|
2832
|
+
"title": cand["title"], "removed": removed, "reset": reset,
|
|
2833
|
+
# carried per row (not just in the tally) so --json and
|
|
2834
|
+
# any later reader can tell which rows only exist because
|
|
2835
|
+
# a deliberate deletion was overridden
|
|
2836
|
+
"overrode_tombstone": bool(cand.get("overrode_tombstone")),
|
|
2837
|
+
"written": False})
|
|
2838
|
+
return {"op_type": "sync",
|
|
2839
|
+
"source_account": source.account_uuid, "source_org": source.org_uuid,
|
|
2840
|
+
"source_email": source.email, "source_path": source.path,
|
|
2841
|
+
# Provenance of the live-account determination ("oauth"/"config"),
|
|
2842
|
+
# so the CLI can say plainly which evidence this plan rests on and
|
|
2843
|
+
# warn that --apply will need the app closed. The executor does
|
|
2844
|
+
# NOT read this key - it re-derives live_account itself, which is
|
|
2845
|
+
# what makes the guard unbypassable by a hand-edited manifest.
|
|
2846
|
+
"source_resolved_from": source.resolved_from,
|
|
2847
|
+
"dest_account": dest.account_uuid, "dest_org": dest.org_uuid,
|
|
2848
|
+
"dest_email": dest.email, "dest_path": dest.path,
|
|
2849
|
+
"verbatim": bool(flags.verbatim), "rows": rows, "tally": tally}
|
|
2850
|
+
|
|
2851
|
+
|
|
2852
|
+
# Journal-write budget for execute_sync_op's row loop.
|
|
2853
|
+
#
|
|
2854
|
+
# save_manifest rewrites and fsyncs the WHOLE manifest, which carries a base64
|
|
2855
|
+
# post-image of every row in the op - so flagging each row `written` with its
|
|
2856
|
+
# own save_manifest costs rows x manifest bytes. Measured: 60 stripped rows at
|
|
2857
|
+
# ~2 KB each produce a 197,693-byte manifest and rewrite 11.9 MB during one
|
|
2858
|
+
# execute. Extrapolated to this machine's real rows under --verbatim (432
|
|
2859
|
+
# rows, the largest 1.36 MB) that is a ~385 MB manifest rewritten 432 times:
|
|
2860
|
+
# over 160 GB of fsynced I/O, i.e. a first `sync --verbatim --apply` into an
|
|
2861
|
+
# empty second account would look like an indefinite hang.
|
|
2862
|
+
#
|
|
2863
|
+
# So spend a fixed byte budget on per-row journaling and stop when it is gone.
|
|
2864
|
+
# How far the budget stretches is a function of manifest size, not row count:
|
|
2865
|
+
# it buys BUDGET / manifest_bytes per-row saves. Small runs - every op in the
|
|
2866
|
+
# test suite, and any modest stripped sync - journal every row individually.
|
|
2867
|
+
# A stripped sync of this machine's own corpus (432 rows, ~1 KB of base64
|
|
2868
|
+
# each, ~417 KB of manifest) journals roughly the first 78 rows individually
|
|
2869
|
+
# and batches the rest; that is the intended shape, not a shortfall. A huge --verbatim manifest is
|
|
2870
|
+
# written once, at the end (or on the way out through an exception - see the
|
|
2871
|
+
# loop). Under-reporting what landed is harmless by construction:
|
|
2872
|
+
# execute_sync_op re-reads every destination row before writing it and
|
|
2873
|
+
# recognises one that already holds exactly the planned bytes, so a resumed op
|
|
2874
|
+
# marks it done rather than duplicating or refusing. The tradeoff bought is
|
|
2875
|
+
# bounded I/O for a coarser - never wrong - record of which rows landed.
|
|
2876
|
+
SYNC_JOURNAL_BYTE_BUDGET = 32 * 1024 * 1024
|
|
2877
|
+
|
|
2878
|
+
|
|
2879
|
+
def execute_sync_op(env, op):
|
|
2880
|
+
"""journaled -> writing -> completed.
|
|
2881
|
+
|
|
2882
|
+
Far simpler than execute_op because nothing is deleted and no transcript
|
|
2883
|
+
moves: the destructive step that dominates a move does not exist here.
|
|
2884
|
+
Rows are journaled as written on a byte budget (SYNC_JOURNAL_BYTE_BUDGET),
|
|
2885
|
+
and always on the way out - normally or through an exception - so any
|
|
2886
|
+
failure this process can observe leaves an exact record of which rows
|
|
2887
|
+
landed. Only a hard kill (power loss, SIGKILL) can lose the tail of that
|
|
2888
|
+
record, and a resumed op recovers from it safely either way.
|
|
2889
|
+
"""
|
|
2890
|
+
m = op.manifest
|
|
2891
|
+
if m.get("status") != "journaled":
|
|
2892
|
+
raise LayoutError("execute_sync_op runs ops from 'journaled'; use recover "
|
|
2893
|
+
"for interrupted ops")
|
|
2894
|
+
|
|
2895
|
+
# Two independent gates. The path comparison below catches a resolvable
|
|
2896
|
+
# live account that IS the destination (a switch the identity files did
|
|
2897
|
+
# register); _guard_mutation catches everything the files cannot prove -
|
|
2898
|
+
# including the E4 case where they are stale or disagree - by refusing
|
|
2899
|
+
# any write while the desktop app itself is running (RULING 4).
|
|
2900
|
+
# realpath on BOTH sides, not normpath: ensure_contained - the other half
|
|
2901
|
+
# of this guarantee, in the row loop below - resolves reparse points, and
|
|
2902
|
+
# this comparison has to agree with it. A junction makes the two disagree
|
|
2903
|
+
# (dest realpath == live realpath while the normpath strings differ).
|
|
2904
|
+
# Everywhere else in this module treats reparse points as hostile; so
|
|
2905
|
+
# does this.
|
|
2906
|
+
live = live_account(env)
|
|
2907
|
+
_refuse_dest_possibly_live(
|
|
2908
|
+
env, live, m["dest_path"], "write to",
|
|
2909
|
+
lambda: "destination resolves to the LIVE account ({0}); refusing - sync must "
|
|
2910
|
+
"never write to the account that is currently live."
|
|
2911
|
+
.format(live.email or live.account_uuid))
|
|
2912
|
+
_guard_mutation(env, "write to")
|
|
2913
|
+
if not os.path.isdir(m["dest_path"]):
|
|
2914
|
+
raise LayoutError("destination store vanished: " + m["dest_path"])
|
|
2915
|
+
|
|
2916
|
+
set_status(op, "writing")
|
|
2917
|
+
rows = m["rows"]
|
|
2918
|
+
# What one save_manifest costs, estimated once rather than measured per
|
|
2919
|
+
# row: the manifest is dominated by the rows' base64 post-images, and
|
|
2920
|
+
# flipping a `written` flag does not change its size materially.
|
|
2921
|
+
per_save = sum(len(r.get("post_b64") or "") for r in rows) + 4096
|
|
2922
|
+
budget = SYNC_JOURNAL_BYTE_BUDGET
|
|
2923
|
+
try:
|
|
2924
|
+
_sync_write_rows(op, m, rows, per_save, budget)
|
|
2925
|
+
except BaseException:
|
|
2926
|
+
# Journal what actually landed before the failure propagates. Every
|
|
2927
|
+
# in-process failure - Refusal, LayoutError, a bare OSError, even
|
|
2928
|
+
# KeyboardInterrupt - therefore still leaves an exact record, which is
|
|
2929
|
+
# what recover's 'back' arm needs to remove exactly the rows this op
|
|
2930
|
+
# wrote. Best-effort: a save that itself fails must never mask the
|
|
2931
|
+
# original failure.
|
|
2932
|
+
try:
|
|
2933
|
+
save_manifest(op)
|
|
2934
|
+
except Exception:
|
|
2935
|
+
pass
|
|
2936
|
+
raise
|
|
2937
|
+
# set_status saves the manifest itself, so it IS the tail-of-batch write -
|
|
2938
|
+
# an explicit save_manifest here would serialize and fsync the whole thing
|
|
2939
|
+
# a second time, which on the very manifest the budget exists to bound
|
|
2940
|
+
# doubles the cost the budget just saved.
|
|
2941
|
+
set_status(op, "completed")
|
|
2942
|
+
return "completed"
|
|
2943
|
+
|
|
2944
|
+
|
|
2945
|
+
def _sync_write_rows(op, m, rows, per_save, budget):
|
|
2946
|
+
"""execute_sync_op's write loop, split out only so its caller can wrap it
|
|
2947
|
+
in the journal-on-the-way-out handler above."""
|
|
2948
|
+
for i, r in enumerate(rows):
|
|
2949
|
+
if r.get("written"):
|
|
2950
|
+
continue
|
|
2951
|
+
# Containment: a hand-edited or simply wrong row dest_path must never
|
|
2952
|
+
# let this loop touch a path outside the destination this op was
|
|
2953
|
+
# verified against above - the dest_path check just above only means
|
|
2954
|
+
# something if every row it is supposed to "cover" is independently
|
|
2955
|
+
# confirmed to actually sit inside it. ensure_contained alone admits
|
|
2956
|
+
# the root itself (real == rreal); a row dest_path equal to the root
|
|
2957
|
+
# would pass that check yet still put atomic_write's <path>.ct-tmp
|
|
2958
|
+
# scratch file one level OUTSIDE the root (a sibling, in its parent)
|
|
2959
|
+
# before the write even fails - so also require every row to be a
|
|
2960
|
+
# direct child of the verified root, not just "under" it.
|
|
2961
|
+
real_dest = ensure_contained(r["dest_path"], [m["dest_path"]])
|
|
2962
|
+
if os.path.dirname(real_dest) != os.path.realpath(m["dest_path"]):
|
|
2963
|
+
raise LayoutError(
|
|
2964
|
+
"row dest_path {0!r} is not a direct child of the destination "
|
|
2965
|
+
"store {1!r}; refusing".format(r["dest_path"], m["dest_path"]))
|
|
2966
|
+
post = unb64(r["post_b64"])
|
|
2967
|
+
try:
|
|
2968
|
+
with open(r["dest_path"], "rb") as fh:
|
|
2969
|
+
current = fh.read()
|
|
2970
|
+
except FileNotFoundError:
|
|
2971
|
+
current = None # not there yet - the common case
|
|
2972
|
+
except OSError as exc:
|
|
2973
|
+
# Anything other than "doesn't exist yet" - permission denied,
|
|
2974
|
+
# the row name resolving to a directory, an I/O error - must
|
|
2975
|
+
# refuse rather than be treated as "absent" and written over:
|
|
2976
|
+
# _row_state elsewhere in this module maps an unreadable current
|
|
2977
|
+
# file to "drifted" (the REFUSING branch), never to "safe to
|
|
2978
|
+
# write". Getting this wrong here would silently overwrite a
|
|
2979
|
+
# destination row this process could not actually verify.
|
|
2980
|
+
raise Refusal(
|
|
2981
|
+
"could not read destination row {0!r} (session {1}) to check "
|
|
2982
|
+
"for changes since planning: {2}. The op is left at 'writing' "
|
|
2983
|
+
"- resolve the row, then re-run.".format(
|
|
2984
|
+
r["name"], r["session_id"], exc))
|
|
2985
|
+
if current is None:
|
|
2986
|
+
try:
|
|
2987
|
+
atomic_write(r["dest_path"], post)
|
|
2988
|
+
except OSError as exc:
|
|
2989
|
+
# The op stays at "writing" either way (non-terminal) -
|
|
2990
|
+
# recover has an accurate record of exactly which rows
|
|
2991
|
+
# landed, same as any other crash mid-loop.
|
|
2992
|
+
raise Refusal(
|
|
2993
|
+
"could not write destination row {0!r} (session {1}): {2}"
|
|
2994
|
+
.format(r["name"], r["session_id"], exc))
|
|
2995
|
+
_maybe_crash("sync-write-before-save")
|
|
2996
|
+
elif current != post:
|
|
2997
|
+
# select_sync_rows only picked rows that were ABSENT at the
|
|
2998
|
+
# destination at plan time. A row now present with DIFFERENT
|
|
2999
|
+
# bytes means the destination account changed it since planning
|
|
3000
|
+
# (e.g. the user signed in and touched that session) - rewriting
|
|
3001
|
+
# over that would silently discard the change. execute_op
|
|
3002
|
+
# refuses on the equivalent drift rather than blindly
|
|
3003
|
+
# overwriting; do the same here, and leave the op non-terminal
|
|
3004
|
+
# (still "writing") so it stays recoverable.
|
|
3005
|
+
raise Refusal(
|
|
3006
|
+
"destination row {0!r} (session {1}) changed since this sync was "
|
|
3007
|
+
"planned; re-running would discard that change. The op is left "
|
|
3008
|
+
"at 'writing' - resolve the row, then re-run.".format(
|
|
3009
|
+
r["name"], r["session_id"]))
|
|
3010
|
+
# else: already byte-identical to the planned post-image - nothing
|
|
3011
|
+
# left to write, just record this row as done.
|
|
3012
|
+
r["written"] = True
|
|
3013
|
+
if budget >= per_save:
|
|
3014
|
+
budget -= per_save
|
|
3015
|
+
save_manifest(op)
|
|
3016
|
+
if i < len(rows) - 1:
|
|
3017
|
+
_maybe_crash("sync-mid-write")
|
|
3018
|
+
|
|
3019
|
+
|
|
3020
|
+
def run_sync(env, manifest):
|
|
3021
|
+
"""Lock, journal, execute, rotate - the same shape as run_move, minus the
|
|
3022
|
+
moved-log append (sync moves nothing, so there is nothing to log there).
|
|
3023
|
+
One difference from run_move worth flagging: run_move's execute_op calls
|
|
3024
|
+
_validate_manifest_paths once, up front, for every path in the manifest.
|
|
3025
|
+
Sync has no such single up-front pass - each row's dest_path is instead
|
|
3026
|
+
validated inline, immediately before that row is touched, inside
|
|
3027
|
+
execute_sync_op's write loop (there is no sidecar inventory or transcript
|
|
3028
|
+
path here for a single shared validator to be worth factoring out).
|
|
3029
|
+
|
|
3030
|
+
Plan-review fix (RULING 4 follow-up): _guard_mutation is checked here
|
|
3031
|
+
too, before acquire_lock/new_op - the earliest clean point, so a refused
|
|
3032
|
+
run creates no lock file and no op directory. execute_sync_op's own copy
|
|
3033
|
+
of this guard fires too late to prevent that: it runs AFTER new_op has
|
|
3034
|
+
already journaled the op, so every refusal there left a stray
|
|
3035
|
+
'journaled' op behind - doctor flags it, recover has to clear it - and
|
|
3036
|
+
the common case triggering this (desktop app left open) is exactly the
|
|
3037
|
+
one RULING 4 made this guard fire on. execute_sync_op's guard still
|
|
3038
|
+
stays, unchanged: it is the ONLY guard recover --forward gets, since
|
|
3039
|
+
resuming a crash-interrupted op re-enters execute_sync_op directly and
|
|
3040
|
+
never calls back through here.
|
|
3041
|
+
"""
|
|
3042
|
+
_guard_mutation(env, "write to")
|
|
3043
|
+
acquire_lock(env, "pending")
|
|
3044
|
+
try:
|
|
3045
|
+
# "tally" is the report's data, not the operation's: it names every
|
|
3046
|
+
# session the run skipped - including the ones the destination account
|
|
3047
|
+
# deliberately DELETED - and nothing in execute/undo/recover reads it.
|
|
3048
|
+
# Journaling it would write those titles to disk in ~/.claude-code-journal
|
|
3049
|
+
# for the lifetime of the op, so strip it from the copy that is
|
|
3050
|
+
# journaled. The "rows" list (and every row dict in it) is still the
|
|
3051
|
+
# same object, so run_sync's caller keeps seeing `written` flags flip.
|
|
3052
|
+
op = new_op(env, dict((k, v) for k, v in manifest.items() if k != "tally"))
|
|
3053
|
+
# Hand the op_id back to the caller: new_op shallow-copies the
|
|
3054
|
+
# manifest and sets op_id on ITS copy, so without this a `sync --apply
|
|
3055
|
+
# --json` run reports a result but no id for `undo --id` to use.
|
|
3056
|
+
manifest["op_id"] = op.manifest["op_id"]
|
|
3057
|
+
# already holding the lock (no O_EXCL needed) - just record the real op_id
|
|
3058
|
+
with open(_lock_path(env), "w") as fh:
|
|
3059
|
+
fh.write("{0} {1}".format(os.getpid(), op.manifest["op_id"]))
|
|
3060
|
+
final = execute_sync_op(env, op)
|
|
3061
|
+
if final == "completed":
|
|
3062
|
+
rotate_ops(env)
|
|
3063
|
+
return final
|
|
3064
|
+
finally:
|
|
3065
|
+
release_lock(env)
|
|
3066
|
+
|
|
3067
|
+
|
|
3068
|
+
def _sync_row_drift(r):
|
|
3069
|
+
"""Compare a single sync row's post-image to whatever currently sits at
|
|
3070
|
+
its dest_path. Returns 'absent' (nothing there), 'match' (present and
|
|
3071
|
+
byte-identical to what this op wrote/would write), 'drifted' (present
|
|
3072
|
+
with different bytes - someone else touched this path since the sync
|
|
3073
|
+
was planned), or 'unreadable' (present but could not be read - a
|
|
3074
|
+
permission or I/O error; fail-closed like every "couldn't look" in this
|
|
3075
|
+
module, never treated as "nothing there"). Never raises: classify_op
|
|
3076
|
+
must never raise (that was this task's original defect - a KeyError on
|
|
3077
|
+
a sync manifest), and undo_sync / recover_op's 'back' arm both need this
|
|
3078
|
+
same read-only per-row classification before deciding what to do.
|
|
3079
|
+
|
|
3080
|
+
"Never raises" has to include the manifest side, not just the disk side:
|
|
3081
|
+
unb64(r["post_b64"]) raises binascii.Error (a ValueError) on a corrupt
|
|
3082
|
+
manifest and KeyError if the field is missing, and main() catches
|
|
3083
|
+
neither. A row whose planned bytes cannot be reconstructed is
|
|
3084
|
+
"unreadable" - fail-closed, and correctly so: it can never be written
|
|
3085
|
+
forward, and it must never be deleted either.
|
|
3086
|
+
"""
|
|
3087
|
+
try:
|
|
3088
|
+
post = unb64(r["post_b64"])
|
|
3089
|
+
except (KeyError, ValueError):
|
|
3090
|
+
return "unreadable"
|
|
3091
|
+
try:
|
|
3092
|
+
with open(r["dest_path"], "rb") as fh:
|
|
3093
|
+
cur = fh.read()
|
|
3094
|
+
except FileNotFoundError:
|
|
3095
|
+
return "absent"
|
|
3096
|
+
except OSError:
|
|
3097
|
+
return "unreadable"
|
|
3098
|
+
return "match" if cur == post else "drifted"
|
|
3099
|
+
|
|
3100
|
+
|
|
3101
|
+
def _sync_drift_titles(rows):
|
|
3102
|
+
"""(changed_titles, unreadable_titles) for ROWS, via _sync_row_drift.
|
|
3103
|
+
Read-only and exception-safe."""
|
|
3104
|
+
changed, unreadable = [], []
|
|
3105
|
+
for r in rows:
|
|
3106
|
+
state = _sync_row_drift(r)
|
|
3107
|
+
if state == "drifted":
|
|
3108
|
+
changed.append(r["title"])
|
|
3109
|
+
elif state == "unreadable":
|
|
3110
|
+
unreadable.append(r["title"])
|
|
3111
|
+
return changed, unreadable
|
|
3112
|
+
|
|
3113
|
+
|
|
3114
|
+
def _drift_clause(changed, unreadable):
|
|
3115
|
+
"""A note/refusal fragment naming drifted vs unreadable rows separately.
|
|
3116
|
+
'Changed' is only true of one of them - conflating the two would falsely
|
|
3117
|
+
claim an unreadable row 'changed' when the real reason is a permission
|
|
3118
|
+
or I/O error."""
|
|
3119
|
+
parts = []
|
|
3120
|
+
if changed:
|
|
3121
|
+
parts.append("changed since this sync was planned ({0})".format(", ".join(changed)))
|
|
3122
|
+
if unreadable:
|
|
3123
|
+
parts.append("could not be read ({0})".format(", ".join(unreadable)))
|
|
3124
|
+
return " and ".join(parts)
|
|
3125
|
+
|
|
3126
|
+
|
|
3127
|
+
def classify_sync_op(env, op):
|
|
3128
|
+
"""Sync's recovery shape. A sync only ever adds rows, so 'back' does not
|
|
3129
|
+
normally exist - forward finishes the remaining writes, and removing
|
|
3130
|
+
what was written is `undo`'s job. The one exception: if a destination
|
|
3131
|
+
row changes underneath a still-in-flight sync, forward can never
|
|
3132
|
+
complete (execute_sync_op refuses on that exact row every time it
|
|
3133
|
+
re-enters), so offering it forever would be a dead end - 'back' becomes
|
|
3134
|
+
the only way off a stuck op. A written row that cannot itself be
|
|
3135
|
+
verified is surfaced here too, before the user even picks a direction -
|
|
3136
|
+
a single event ("the dormant account got opened") can plausibly both
|
|
3137
|
+
block a pending row and rewrite an already-written one, so 'options:
|
|
3138
|
+
back' alone would otherwise promise more than back can deliver.
|
|
3139
|
+
|
|
3140
|
+
Correction from the whole-branch review: 'back' is offered ALWAYS, not
|
|
3141
|
+
only on drift. Destination-row drift is just one of the ways
|
|
3142
|
+
execute_sync_op can leave an op non-terminal - atomic_write raising
|
|
3143
|
+
OSError, the row-containment LayoutError and the vanished-store
|
|
3144
|
+
LayoutError all leave the pending row simply ABSENT, which reads as no
|
|
3145
|
+
drift, which used to classify as "forward only". Forward then raised the
|
|
3146
|
+
same error on every re-entry, 'back' was refused as unsafe, and undo
|
|
3147
|
+
refused the op for not being 'completed': every exit refused and the op
|
|
3148
|
+
was stuck forever - the exact dead end 'back' exists to close, left open
|
|
3149
|
+
for the I/O and layout cases. 'back' is unconditionally safe here (it
|
|
3150
|
+
only deletes rows this op recorded as written AND that are still
|
|
3151
|
+
byte-identical to what it wrote, skipping everything else), so there is
|
|
3152
|
+
no state in which withholding it is right. classify_op's move branches
|
|
3153
|
+
already offer two resolutions where both are safe; follow that shape:
|
|
3154
|
+
["forward", "back"] normally, ["back"] alone when a drifted or
|
|
3155
|
+
unreadable pending row makes forward impossible.
|
|
3156
|
+
"""
|
|
3157
|
+
m = op.manifest
|
|
3158
|
+
written = [r for r in m["rows"] if r.get("written")]
|
|
3159
|
+
pending = [r for r in m["rows"] if not r.get("written")]
|
|
3160
|
+
if m["status"] not in NONTERMINAL:
|
|
3161
|
+
return {"status": m["status"], "source": "n/a", "dest": "n/a",
|
|
3162
|
+
"resolutions": [], "drifted_rows": [],
|
|
3163
|
+
"note": "sync: {0} row(s) written, {1} pending; forward finishes "
|
|
3164
|
+
"them (use undo to remove what was written)"
|
|
3165
|
+
.format(len(written), len(pending))}
|
|
3166
|
+
pend_changed, pend_unreadable = _sync_drift_titles(pending)
|
|
3167
|
+
if pend_changed or pend_unreadable:
|
|
3168
|
+
blocking = pend_changed + pend_unreadable
|
|
3169
|
+
skip_changed, skip_unreadable = _sync_drift_titles(written)
|
|
3170
|
+
skipped = skip_changed + skip_unreadable
|
|
3171
|
+
note = ("sync: destination row(s) {0}; it can no longer be rolled "
|
|
3172
|
+
"forward - 'back' removes the row(s) this op safely can"
|
|
3173
|
+
.format(_drift_clause(pend_changed, pend_unreadable)))
|
|
3174
|
+
if skipped:
|
|
3175
|
+
note += ("; it will also skip {0} already-written row(s) it cannot "
|
|
3176
|
+
"verify ({1})".format(len(skipped),
|
|
3177
|
+
_drift_clause(skip_changed, skip_unreadable)))
|
|
3178
|
+
# dict.fromkeys, not a bare concatenation: a pending row and a
|
|
3179
|
+
# written row can share a title, and this list reaches cmd_recover's
|
|
3180
|
+
# printed line - listing the same title twice reads as two problems.
|
|
3181
|
+
# Order-preserving, unlike set().
|
|
3182
|
+
return {"status": m["status"], "source": "n/a", "dest": "n/a",
|
|
3183
|
+
"resolutions": ["back"], "note": note,
|
|
3184
|
+
"drifted_rows": list(dict.fromkeys(blocking + skipped))}
|
|
3185
|
+
return {"status": m["status"], "source": "n/a", "dest": "n/a",
|
|
3186
|
+
"resolutions": ["forward", "back"], "drifted_rows": [],
|
|
3187
|
+
"note": "sync: {0} row(s) written, {1} pending; forward finishes them, "
|
|
3188
|
+
"back removes the {0} already written (use undo instead once "
|
|
3189
|
+
"the op has completed)".format(len(written), len(pending))}
|
|
3190
|
+
|
|
3191
|
+
|
|
3192
|
+
def _sync_delete_targets(env, m):
|
|
3193
|
+
"""Rows THIS sync op actually wrote, classified for safe deletion.
|
|
3194
|
+
Shared by undo_sync and recover_op's 'back' path, which react
|
|
3195
|
+
differently to a non-empty drifted/unreadable result: undo_sync refuses
|
|
3196
|
+
the whole operation (a surprise during a user-initiated reversal of a
|
|
3197
|
+
completed sync means stop and ask), while back SKIPS those rows and
|
|
3198
|
+
proceeds with the rest, because back is the only exit from a stuck op
|
|
3199
|
+
and must always reach a terminal status - refusing there would recreate
|
|
3200
|
+
the exact dead end it exists to close.
|
|
3201
|
+
|
|
3202
|
+
Two hard gates, checked up front - never "skip and continue", because
|
|
3203
|
+
they mean the op itself cannot be trusted, not just one row:
|
|
3204
|
+
1. The same live-account re-check execute_sync_op makes on every write,
|
|
3205
|
+
plus the same universal running-app guard (_guard_mutation, RULING
|
|
3206
|
+
4) - a delete is exactly as dangerous as a write here, and must
|
|
3207
|
+
carry the identical guarantee.
|
|
3208
|
+
2. The same containment execute_sync_op's write loop uses
|
|
3209
|
+
(ensure_contained plus the direct-child check), so a hand-edited or
|
|
3210
|
+
corrupted manifest row can never point this delete outside the
|
|
3211
|
+
destination store.
|
|
3212
|
+
|
|
3213
|
+
Per row, via _sync_row_drift: a row this op never wrote (`written` is
|
|
3214
|
+
not True) is never considered. 'absent' is skipped as already-undone.
|
|
3215
|
+
'match' is removable. 'drifted' and 'unreadable' are reported
|
|
3216
|
+
separately - neither is ever deleted.
|
|
3217
|
+
|
|
3218
|
+
Returns (drifted_titles, unreadable_titles, removable_paths). Deletes
|
|
3219
|
+
nothing itself.
|
|
3220
|
+
"""
|
|
3221
|
+
# realpath on both sides, matching execute_sync_op's write-side check and
|
|
3222
|
+
# ensure_contained below - see the note there on junctions.
|
|
3223
|
+
live = live_account(env)
|
|
3224
|
+
_refuse_dest_possibly_live(
|
|
3225
|
+
env, live, m["dest_path"], "delete from",
|
|
3226
|
+
lambda: "destination resolves to the LIVE account ({0}); refusing - undo, "
|
|
3227
|
+
"like sync, may only ever touch a dormant store, never the account "
|
|
3228
|
+
"that is currently live.".format(live.email or live.account_uuid))
|
|
3229
|
+
# The same guard as the write side (RULING 4: every mutation route,
|
|
3230
|
+
# regardless of provenance). Both callers of this helper (undo_sync,
|
|
3231
|
+
# recover_op's sync 'back' arm) inherit it from this one place.
|
|
3232
|
+
_guard_mutation(env, "delete from")
|
|
3233
|
+
drifted, unreadable, removable = [], [], []
|
|
3234
|
+
for r in m["rows"]:
|
|
3235
|
+
if not r.get("written"):
|
|
3236
|
+
continue
|
|
3237
|
+
real_dest = ensure_contained(r["dest_path"], [m["dest_path"]])
|
|
3238
|
+
if os.path.dirname(real_dest) != os.path.realpath(m["dest_path"]):
|
|
3239
|
+
raise LayoutError(
|
|
3240
|
+
"row dest_path {0!r} is not a direct child of the destination "
|
|
3241
|
+
"store {1!r}; refusing".format(r["dest_path"], m["dest_path"]))
|
|
3242
|
+
state = _sync_row_drift(r)
|
|
3243
|
+
if state == "absent":
|
|
3244
|
+
continue # confirmed absent - nothing to undo
|
|
3245
|
+
elif state == "match":
|
|
3246
|
+
removable.append(r["dest_path"])
|
|
3247
|
+
elif state == "drifted":
|
|
3248
|
+
drifted.append(r["title"])
|
|
3249
|
+
else: # "unreadable"
|
|
3250
|
+
unreadable.append(r["title"])
|
|
3251
|
+
return drifted, unreadable, removable
|
|
3252
|
+
|
|
3253
|
+
|
|
3254
|
+
def _sync_unlink_all(paths):
|
|
3255
|
+
"""Delete every path, attempting all of them even if some fail, and
|
|
3256
|
+
report every failure together rather than stopping at the first -
|
|
3257
|
+
mirrors _delete_inventoried_files. A bare OSError here (permission
|
|
3258
|
+
denied, a locked file) must never propagate raw: main() only catches
|
|
3259
|
+
Refusal/LayoutError."""
|
|
3260
|
+
failures = []
|
|
3261
|
+
for p in paths:
|
|
3262
|
+
try:
|
|
3263
|
+
os.unlink(p)
|
|
3264
|
+
except OSError as exc:
|
|
3265
|
+
failures.append((p, exc))
|
|
3266
|
+
if failures:
|
|
3267
|
+
raise Refusal("could not remove {0}".format(
|
|
3268
|
+
", ".join("{0} ({1})".format(p, exc) for p, exc in failures)))
|
|
3269
|
+
|
|
3270
|
+
|
|
3271
|
+
def undo_sync(env, op):
|
|
3272
|
+
"""Delete exactly the rows this sync wrote - and only while they are still
|
|
3273
|
+
byte-identical to what it wrote. If the destination account has since
|
|
3274
|
+
opened the session the app rewrites the row, and deleting it would discard
|
|
3275
|
+
that account's own state. A row that cannot even be read is treated the
|
|
3276
|
+
same way - a surprise either way, so undo refuses rather than guessing.
|
|
3277
|
+
This is the deliberate asymmetry with recover_op's 'back' arm: undo is a
|
|
3278
|
+
user-initiated reversal of a *completed* sync, where a surprise means
|
|
3279
|
+
stop and ask; back is the only exit from a stuck op and must always
|
|
3280
|
+
terminate, so it skips instead (see _sync_delete_targets).
|
|
3281
|
+
|
|
3282
|
+
Takes the single-instance lock first, before any of its own checks - the
|
|
3283
|
+
same discipline run_undo/run_move/run_sync/recover_op all use, so two
|
|
3284
|
+
concurrent 'undo --apply' runs can never race their unlinks.
|
|
3285
|
+
"""
|
|
3286
|
+
m = op.manifest
|
|
3287
|
+
acquire_lock(env, "undo-" + m["op_id"])
|
|
3288
|
+
try:
|
|
3289
|
+
if m.get("op_type") != "sync":
|
|
3290
|
+
raise Refusal("not a sync op: " + str(m.get("op_id")))
|
|
3291
|
+
if m.get("status") != "completed":
|
|
3292
|
+
raise Refusal("op {0} is '{1}', not 'completed'".format(
|
|
3293
|
+
m.get("op_id"), m.get("status")))
|
|
3294
|
+
drifted, unreadable, removable = _sync_delete_targets(env, m)
|
|
3295
|
+
if drifted or unreadable:
|
|
3296
|
+
raise Refusal("these synced rows {0}; the other account may have opened "
|
|
3297
|
+
"them. Refusing to delete any of them."
|
|
3298
|
+
.format(_drift_clause(drifted, unreadable)))
|
|
3299
|
+
_sync_unlink_all(removable)
|
|
3300
|
+
set_status(op, "undone")
|
|
3301
|
+
rotate_ops(env)
|
|
3302
|
+
return "undone"
|
|
3303
|
+
finally:
|
|
3304
|
+
release_lock(env)
|
|
3305
|
+
|
|
3306
|
+
|
|
3307
|
+
def cmd_sync(env, ns):
|
|
3308
|
+
flags = SyncFlags(to=ns.to, only=ns.only,
|
|
3309
|
+
include_deleted=tuple(ns.include_deleted or ()),
|
|
3310
|
+
verbatim=ns.verbatim)
|
|
3311
|
+
manifest = plan_sync(env, flags)
|
|
3312
|
+
|
|
3313
|
+
def say(line):
|
|
3314
|
+
print(line if ns.verbose else redact(env, line))
|
|
3315
|
+
|
|
3316
|
+
# Ordering, which the human report and the JSON dump have no reason to
|
|
3317
|
+
# share: the human report prints BOTH ENDPOINTS FIRST, before anything
|
|
3318
|
+
# happens (spec s5 - a recognisable destination is a safety feature, and
|
|
3319
|
+
# a run that dies inside run_sync must still leave a record of which two
|
|
3320
|
+
# accounts were involved rather than a bare "refused: <msg>"). --json
|
|
3321
|
+
# instead has to execute first, because "sync --apply --json" - exactly
|
|
3322
|
+
# the combination automation would use - must report what actually
|
|
3323
|
+
# happened, not the plan it would have run.
|
|
3324
|
+
if not ns.json:
|
|
3325
|
+
_print_sync_report(say, manifest)
|
|
3326
|
+
|
|
3327
|
+
# A zero-row plan skips run_sync regardless of --apply: there's nothing
|
|
3328
|
+
# to journal, and journaling an empty op anyway was a parked finding.
|
|
3329
|
+
final = None
|
|
3330
|
+
if manifest["rows"] and ns.apply:
|
|
3331
|
+
final = run_sync(env, manifest)
|
|
3332
|
+
|
|
3333
|
+
if ns.json:
|
|
3334
|
+
if final is not None:
|
|
3335
|
+
manifest["result"] = final
|
|
3336
|
+
print(json.dumps(manifest, indent=1))
|
|
3337
|
+
return 0 if final in (None, "completed") else 1
|
|
3338
|
+
|
|
3339
|
+
if not manifest["rows"]:
|
|
3340
|
+
say("\nnothing to copy")
|
|
3341
|
+
return 0
|
|
3342
|
+
if final is None:
|
|
3343
|
+
say("\ndry run - pass --apply to copy")
|
|
3344
|
+
return 0
|
|
3345
|
+
|
|
3346
|
+
# "copied: N" reads r["written"], which run_sync's execute loop set on
|
|
3347
|
+
# the row dicts THIS manifest still holds: new_op shallow-copies the
|
|
3348
|
+
# manifest, so the "rows" list and every row dict inside it are shared
|
|
3349
|
+
# between the caller's manifest and the journaled one. That coupling is
|
|
3350
|
+
# load-bearing here and easy to break by deep-copying "for safety".
|
|
3351
|
+
say("\ncopied : {0}".format(sum(1 for r in manifest["rows"] if r.get("written"))))
|
|
3352
|
+
say("result : {0}".format(final))
|
|
3353
|
+
say("Sign into {0} (or restart the app) to see them."
|
|
3354
|
+
.format(manifest["dest_email"] or "the other account"))
|
|
3355
|
+
return 0 if final == "completed" else 1
|
|
3356
|
+
|
|
3357
|
+
|
|
3358
|
+
def _print_sync_report(say, manifest):
|
|
3359
|
+
"""The human-readable plan: both endpoints, then the skip tally, then
|
|
3360
|
+
what --include-deleted is resurrecting, then what would be copied."""
|
|
3361
|
+
# Spec s5: name both endpoints, with emails, before doing anything.
|
|
3362
|
+
# dest_email is "" for every non-live account - the dormant account's
|
|
3363
|
+
# email isn't recorded anywhere on disk, so an unlabelled run always hits
|
|
3364
|
+
# this, not just an edge case - so also print the store path and org
|
|
3365
|
+
# prefix, through the same say()/redact() convention (redacted unless
|
|
3366
|
+
# --verbose), giving a cautious user a physical folder to recognise
|
|
3367
|
+
# instead of eight hex characters. Symmetric for the source.
|
|
3368
|
+
# A source resolved from config.json must never print the same
|
|
3369
|
+
# "(email unknown)" an ordinary dormant-side line prints - that would
|
|
3370
|
+
# look identical to the normal case and hide how the account was
|
|
3371
|
+
# identified. Say where it came from; since RULING 4 that provenance is
|
|
3372
|
+
# a note for the user, not a gate - --apply's guard applies the same way
|
|
3373
|
+
# regardless of resolved_from (see _guard_mutation).
|
|
3374
|
+
weak = manifest.get("source_resolved_from") == "config"
|
|
3375
|
+
say("from {0:24} ({1}/{2}) signed in".format(
|
|
3376
|
+
"(from config.json)" if weak else
|
|
3377
|
+
(manifest["source_email"] or "(email unknown)"),
|
|
3378
|
+
manifest["source_account"][:8], manifest["source_org"][:8]))
|
|
3379
|
+
say(" " + manifest["source_path"])
|
|
3380
|
+
if weak:
|
|
3381
|
+
say(" ! identified from config.json's lastKnownAccountUuid, not from a")
|
|
3382
|
+
say(" signed-in oauthAccount - a provenance note only, not a stronger/")
|
|
3383
|
+
say(" weaker distinction.")
|
|
3384
|
+
# Minor 5: this used to sit inside `if weak:` above, so an ordinary
|
|
3385
|
+
# oauth-resolved dry run never warned that --apply refuses while Claude
|
|
3386
|
+
# is running - even though _guard_mutation (RULING 4) applies exactly
|
|
3387
|
+
# the same way regardless of resolved_from. Every dry run prints it now;
|
|
3388
|
+
# the weak-only provenance note above stays weak-only.
|
|
3389
|
+
say(" --apply will refuse while Claude is running either way (RULING 4).")
|
|
3390
|
+
say("to {0:24} ({1}/{2}) signed out".format(
|
|
3391
|
+
manifest["dest_email"] or "(email unknown)",
|
|
3392
|
+
manifest["dest_account"][:8], manifest["dest_org"][:8]))
|
|
3393
|
+
say(" " + manifest["dest_path"])
|
|
3394
|
+
say("")
|
|
3395
|
+
|
|
3396
|
+
tally = manifest["tally"]
|
|
3397
|
+
LABELS = [("present", "already in the destination"),
|
|
3398
|
+
("no_transcript", "skipped, transcript gone"),
|
|
3399
|
+
("deleted", "skipped, deleted in the destination"),
|
|
3400
|
+
("unreadable", "skipped, unreadable row"),
|
|
3401
|
+
("filtered", "skipped, did not match --only")]
|
|
3402
|
+
for key, label in LABELS:
|
|
3403
|
+
items = tally.get(key) or []
|
|
3404
|
+
if items:
|
|
3405
|
+
say("{0:36}: {1}".format(label, len(items)))
|
|
3406
|
+
# Tombstone skips are named individually: the user deleted these on
|
|
3407
|
+
# purpose and should see the deletion was honoured, not silently
|
|
3408
|
+
# dropped. Capped the same way as the "to copy" list below - a source
|
|
3409
|
+
# account with many deliberate deletions must not produce unbounded
|
|
3410
|
+
# output.
|
|
3411
|
+
deleted_titles = tally.get("deleted") or []
|
|
3412
|
+
for title in deleted_titles[:15]:
|
|
3413
|
+
say(" kept deleted: {0}".format(title))
|
|
3414
|
+
if len(deleted_titles) > 15:
|
|
3415
|
+
say(" ... and {0} more".format(len(deleted_titles) - 15))
|
|
3416
|
+
|
|
3417
|
+
# --include-deleted is the one thing this command does that the user
|
|
3418
|
+
# cannot undo by simply deleting a row again - it brings back a session
|
|
3419
|
+
# they deliberately deleted, the first row of the design's own risk
|
|
3420
|
+
# table. It used to be the LEAST visible thing here: the rescued row
|
|
3421
|
+
# entered the plan with no marker and tally["deleted"] held only the
|
|
3422
|
+
# skips, so the report said nothing at all. Name every resurrection,
|
|
3423
|
+
# under an unmissable label, BEFORE the ordinary "to copy" list.
|
|
3424
|
+
resurrected = tally.get("resurrected") or []
|
|
3425
|
+
if resurrected:
|
|
3426
|
+
say("")
|
|
3427
|
+
say("!! RESURRECTING {0} session(s) the destination account DELETED "
|
|
3428
|
+
"(--include-deleted):".format(len(resurrected)))
|
|
3429
|
+
for title in resurrected[:15]:
|
|
3430
|
+
say(" !! {0}".format(title))
|
|
3431
|
+
if len(resurrected) > 15:
|
|
3432
|
+
say(" ... and {0} more".format(len(resurrected) - 15))
|
|
3433
|
+
say("")
|
|
3434
|
+
|
|
3435
|
+
say("{0:36}: {1}".format("to copy", len(manifest["rows"])))
|
|
3436
|
+
for r in manifest["rows"][:15]:
|
|
3437
|
+
say(" {0}{1}".format("!! " if r.get("overrode_tombstone") else "", r["title"]))
|
|
3438
|
+
if len(manifest["rows"]) > 15:
|
|
3439
|
+
say(" ... and {0} more".format(len(manifest["rows"]) - 15))
|
|
3440
|
+
|
|
3441
|
+
|
|
3442
|
+
if __name__ == "__main__":
|
|
3443
|
+
sys.exit(main())
|