@brandry/claude-jsonl-compressor 1.0.0-rc.1 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +29 -0
- package/README.md +34 -16
- package/SKILL.md +6 -1
- package/package.json +2 -2
- package/references/claude-jsonl-compression-format.md +81 -3
- package/scripts/claude_session_tools.py +34 -4
- package/scripts/compress_claude_jsonl.py +778 -155
- package/scripts/repair_claude_jsonl.py +32 -10
|
@@ -26,13 +26,33 @@ from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
|
|
26
26
|
|
|
27
27
|
|
|
28
28
|
JsonObj = Dict[str, Any]
|
|
29
|
-
PACKAGE_VERSION = "1.0.0
|
|
29
|
+
PACKAGE_VERSION = "1.0.0"
|
|
30
30
|
CODEX_OFFLINE_COMPRESSION_VERSION = "v10"
|
|
31
31
|
MODEL_PACK_SCHEMA_VERSION = 11
|
|
32
32
|
REPORT_SCHEMA_VERSION = 1
|
|
33
33
|
PRIOR_SUMMARY_VERBATIM_BUDGET_FACTOR = 1.5
|
|
34
34
|
MIN_SUMMARY_CHAR_BUDGET = 4000
|
|
35
35
|
DEFAULT_MODEL_PACK_ESTIMATED_TOKEN_BUDGET = 150000
|
|
36
|
+
MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS = 160
|
|
37
|
+
RESUME_REASON_CODE_STATUS = {
|
|
38
|
+
"last-prompt-absent": "absent",
|
|
39
|
+
"duplicate-uuid": "duplicate-uuid",
|
|
40
|
+
"leaf-uuid-malformed": "malformed",
|
|
41
|
+
"chain-missing-uuid": "dangling",
|
|
42
|
+
"chain-loop": "loop",
|
|
43
|
+
"chain-malformed-parent": "malformed-parent",
|
|
44
|
+
"chain-empty": "dangling",
|
|
45
|
+
"chain-non-monotonic": "non-monotonic",
|
|
46
|
+
"lineage-unsafe": "session-mismatch",
|
|
47
|
+
"extension-limit-exceeded": "extension-limit",
|
|
48
|
+
"extension-authority-session-missing": "session-mismatch",
|
|
49
|
+
"extension-record-missing-uuid": "extension-unsafe",
|
|
50
|
+
"extension-record-not-linear-descendant": "extension-branch",
|
|
51
|
+
"extension-record-session-mismatch": "session-mismatch",
|
|
52
|
+
"extension-record-not-safe-closure": "extension-unsafe",
|
|
53
|
+
"extension-pending-tool-ids": "extension-unsafe",
|
|
54
|
+
"ok": "valid",
|
|
55
|
+
}
|
|
36
56
|
|
|
37
57
|
|
|
38
58
|
def configure_stdio() -> None:
|
|
@@ -47,6 +67,7 @@ def configure_stdio() -> None:
|
|
|
47
67
|
|
|
48
68
|
configure_stdio()
|
|
49
69
|
|
|
70
|
+
|
|
50
71
|
DEFAULT_IMPORTANCE_WORDS = tuple(
|
|
51
72
|
[
|
|
52
73
|
"must",
|
|
@@ -1825,9 +1846,13 @@ def json_dump_line(obj: JsonObj) -> str:
|
|
|
1825
1846
|
return json.dumps(obj, ensure_ascii=False, separators=(",", ":"))
|
|
1826
1847
|
|
|
1827
1848
|
|
|
1828
|
-
def
|
|
1849
|
+
def parse_jsonl_bytes_with_lines(
|
|
1850
|
+
data: bytes,
|
|
1851
|
+
source_label: str = "JSONL",
|
|
1852
|
+
) -> Tuple[List[JsonObj], List[str], List[int]]:
|
|
1829
1853
|
records: List[JsonObj] = []
|
|
1830
1854
|
raw_lines: List[str] = []
|
|
1855
|
+
physical_lines: List[int] = []
|
|
1831
1856
|
for line_no, physical_line in enumerate(data.split(b"\n"), 1):
|
|
1832
1857
|
line_bytes = physical_line[:-1] if physical_line.endswith(b"\r") else physical_line
|
|
1833
1858
|
if line_no == 1 and line_bytes.startswith(b"\xef\xbb\xbf"):
|
|
@@ -1846,6 +1871,12 @@ def parse_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Tuple[List[Js
|
|
|
1846
1871
|
raise ValueError(f"line {line_no} is JSON but not an object")
|
|
1847
1872
|
records.append(obj)
|
|
1848
1873
|
raw_lines.append(line)
|
|
1874
|
+
physical_lines.append(line_no)
|
|
1875
|
+
return records, raw_lines, physical_lines
|
|
1876
|
+
|
|
1877
|
+
|
|
1878
|
+
def parse_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Tuple[List[JsonObj], List[str]]:
|
|
1879
|
+
records, raw_lines, _physical_lines = parse_jsonl_bytes_with_lines(data, source_label=source_label)
|
|
1849
1880
|
return records, raw_lines
|
|
1850
1881
|
|
|
1851
1882
|
|
|
@@ -1956,6 +1987,49 @@ def numbered_backup_path(path: pathlib.Path, backup_dir: Optional[pathlib.Path]
|
|
|
1956
1987
|
raise RuntimeError(f"could not find free backup name for {path}")
|
|
1957
1988
|
|
|
1958
1989
|
|
|
1990
|
+
def _filesystem_error_detail(error: Exception) -> str:
|
|
1991
|
+
detail = f"{type(error).__name__}: {error}"
|
|
1992
|
+
errno_value = getattr(error, "errno", None)
|
|
1993
|
+
winerror_value = getattr(error, "winerror", None)
|
|
1994
|
+
if errno_value is not None:
|
|
1995
|
+
detail += f"; errno={errno_value}"
|
|
1996
|
+
if winerror_value is not None:
|
|
1997
|
+
detail += f"; winerror={winerror_value}"
|
|
1998
|
+
return detail
|
|
1999
|
+
|
|
2000
|
+
|
|
2001
|
+
def _cleanup_owned_file(
|
|
2002
|
+
path: pathlib.Path,
|
|
2003
|
+
*,
|
|
2004
|
+
identity: Optional[os.stat_result],
|
|
2005
|
+
expected_bytes: Optional[bytes],
|
|
2006
|
+
owner_label: str,
|
|
2007
|
+
missing_is_error: bool = False,
|
|
2008
|
+
) -> Optional[str]:
|
|
2009
|
+
"""Remove a unique temporary path only while its observed identity remains ours."""
|
|
2010
|
+
if identity is None:
|
|
2011
|
+
try:
|
|
2012
|
+
path.lstat()
|
|
2013
|
+
except FileNotFoundError:
|
|
2014
|
+
return None
|
|
2015
|
+
except OSError as exc:
|
|
2016
|
+
return f"{path.name}: {_filesystem_error_detail(exc)}"
|
|
2017
|
+
return f"{path.name}: {owner_label} identity was not captured"
|
|
2018
|
+
try:
|
|
2019
|
+
if not os.path.samestat(identity, path.lstat()):
|
|
2020
|
+
return f"{path.name}: {owner_label} identity changed before cleanup"
|
|
2021
|
+
if expected_bytes is not None and path.read_bytes() != expected_bytes:
|
|
2022
|
+
return f"{path.name}: {owner_label} bytes changed before cleanup"
|
|
2023
|
+
if not os.path.samestat(identity, path.lstat()):
|
|
2024
|
+
return f"{path.name}: {owner_label} identity changed before cleanup"
|
|
2025
|
+
path.unlink()
|
|
2026
|
+
except FileNotFoundError:
|
|
2027
|
+
return f"{path.name}: disappeared before cleanup" if missing_is_error else None
|
|
2028
|
+
except OSError as exc:
|
|
2029
|
+
return f"{path.name}: {_filesystem_error_detail(exc)}"
|
|
2030
|
+
return None
|
|
2031
|
+
|
|
2032
|
+
|
|
1959
2033
|
def _exclusive_backup_from_bytes(
|
|
1960
2034
|
path: pathlib.Path,
|
|
1961
2035
|
source_bytes: bytes,
|
|
@@ -1970,26 +2044,42 @@ def _exclusive_backup_from_bytes(
|
|
|
1970
2044
|
for backup in candidates:
|
|
1971
2045
|
backup.parent.mkdir(parents=True, exist_ok=True)
|
|
1972
2046
|
stage = backup.with_name(f".{backup.name}.stage-{uuid.uuid4().hex}.tmp")
|
|
2047
|
+
stage_identity: Optional[os.stat_result] = None
|
|
2048
|
+
stage_expected_bytes: Optional[bytes] = None
|
|
2049
|
+
verified_backup: Optional[pathlib.Path] = None
|
|
1973
2050
|
try:
|
|
1974
2051
|
with stage.open("xb") as f:
|
|
2052
|
+
stage_identity = stage.lstat()
|
|
1975
2053
|
f.write(source_bytes)
|
|
1976
2054
|
f.flush()
|
|
1977
2055
|
os.fsync(f.fileno())
|
|
1978
2056
|
if stage.read_bytes() != source_bytes:
|
|
1979
2057
|
raise RuntimeError(f"staged backup verification failed: {backup.name}")
|
|
2058
|
+
stage_expected_bytes = source_bytes
|
|
1980
2059
|
try:
|
|
1981
2060
|
_publish_no_clobber(stage, backup)
|
|
1982
2061
|
except FileExistsError:
|
|
1983
2062
|
continue
|
|
1984
2063
|
if backup.read_bytes() == source_bytes:
|
|
1985
2064
|
fsync_parent_directory(backup)
|
|
2065
|
+
verified_backup = backup
|
|
1986
2066
|
return backup
|
|
1987
2067
|
# A concurrently replaced numbered path is not ours to delete.
|
|
1988
2068
|
finally:
|
|
1989
|
-
|
|
1990
|
-
|
|
1991
|
-
|
|
1992
|
-
|
|
2069
|
+
active_error = sys.exc_info()[1]
|
|
2070
|
+
cleanup_error = _cleanup_owned_file(
|
|
2071
|
+
stage,
|
|
2072
|
+
identity=stage_identity,
|
|
2073
|
+
expected_bytes=stage_expected_bytes,
|
|
2074
|
+
owner_label="backup stage",
|
|
2075
|
+
)
|
|
2076
|
+
if cleanup_error is not None:
|
|
2077
|
+
message = f"backup staging cleanup failed: {cleanup_error}"
|
|
2078
|
+
if verified_backup is not None:
|
|
2079
|
+
message += f"; verified backup retained as {verified_backup.name}"
|
|
2080
|
+
if active_error is not None:
|
|
2081
|
+
raise RuntimeError(f"{active_error}; {message}") from active_error
|
|
2082
|
+
raise RuntimeError(message)
|
|
1993
2083
|
raise RuntimeError(f"could not find free backup name for {path}")
|
|
1994
2084
|
|
|
1995
2085
|
|
|
@@ -1999,6 +2089,107 @@ def create_backup(path: pathlib.Path, backup_dir: Optional[pathlib.Path] = None)
|
|
|
1999
2089
|
return _exclusive_backup_from_bytes(path, path.read_bytes(), backup_dir=backup_dir)
|
|
2000
2090
|
|
|
2001
2091
|
|
|
2092
|
+
def _preflight_hardlink_support(directory: pathlib.Path, label: str) -> None:
|
|
2093
|
+
"""Verify hard links in one publication directory before live replacement writes."""
|
|
2094
|
+
token = uuid.uuid4().hex
|
|
2095
|
+
probe_source = directory / f".cjc-hardlink-probe-{token}.source.tmp"
|
|
2096
|
+
probe_link = directory / f".cjc-hardlink-probe-{token}.link.tmp"
|
|
2097
|
+
marker = f"claude-jsonl-compressor hard-link probe {token}\n".encode("ascii")
|
|
2098
|
+
source_created = False
|
|
2099
|
+
link_created = False
|
|
2100
|
+
source_identity: Optional[os.stat_result] = None
|
|
2101
|
+
link_identity: Optional[os.stat_result] = None
|
|
2102
|
+
operation_error: Optional[Tuple[str, Exception]] = None
|
|
2103
|
+
cleanup_errors: List[str] = []
|
|
2104
|
+
|
|
2105
|
+
try:
|
|
2106
|
+
phase = "probe source creation"
|
|
2107
|
+
with probe_source.open("xb") as stream:
|
|
2108
|
+
source_created = True
|
|
2109
|
+
stream.write(marker)
|
|
2110
|
+
stream.flush()
|
|
2111
|
+
os.fsync(stream.fileno())
|
|
2112
|
+
source_identity = probe_source.lstat()
|
|
2113
|
+
|
|
2114
|
+
phase = "hard-link creation"
|
|
2115
|
+
os.link(probe_source, probe_link)
|
|
2116
|
+
link_created = True
|
|
2117
|
+
link_identity = probe_link.lstat()
|
|
2118
|
+
|
|
2119
|
+
phase = "hard-link verification"
|
|
2120
|
+
if source_identity is None or link_identity is None:
|
|
2121
|
+
raise RuntimeError("probe identity was not captured")
|
|
2122
|
+
if not os.path.samestat(source_identity, link_identity):
|
|
2123
|
+
raise RuntimeError("probe destination does not reference the probe source")
|
|
2124
|
+
if not os.path.samefile(probe_source, probe_link):
|
|
2125
|
+
raise RuntimeError("probe destination does not reference the probe source")
|
|
2126
|
+
if probe_link.read_bytes() != marker:
|
|
2127
|
+
raise RuntimeError("probe hard-link bytes differ from the probe source")
|
|
2128
|
+
except Exception as exc:
|
|
2129
|
+
operation_error = (phase, exc)
|
|
2130
|
+
finally:
|
|
2131
|
+
if link_created:
|
|
2132
|
+
cleanup_error = _cleanup_owned_file(
|
|
2133
|
+
probe_link,
|
|
2134
|
+
identity=link_identity,
|
|
2135
|
+
expected_bytes=marker,
|
|
2136
|
+
owner_label="probe",
|
|
2137
|
+
missing_is_error=True,
|
|
2138
|
+
)
|
|
2139
|
+
if cleanup_error is not None:
|
|
2140
|
+
cleanup_errors.append(cleanup_error)
|
|
2141
|
+
else:
|
|
2142
|
+
try:
|
|
2143
|
+
probe_link.lstat()
|
|
2144
|
+
except FileNotFoundError:
|
|
2145
|
+
pass
|
|
2146
|
+
except OSError as exc:
|
|
2147
|
+
cleanup_errors.append(
|
|
2148
|
+
f"{probe_link.name}: could not inspect destination after link failure; "
|
|
2149
|
+
f"{_filesystem_error_detail(exc)}"
|
|
2150
|
+
)
|
|
2151
|
+
else:
|
|
2152
|
+
cleanup_errors.append(f"{probe_link.name}: destination was claimed after link failure")
|
|
2153
|
+
|
|
2154
|
+
if source_created:
|
|
2155
|
+
cleanup_error = _cleanup_owned_file(
|
|
2156
|
+
probe_source,
|
|
2157
|
+
identity=source_identity,
|
|
2158
|
+
expected_bytes=marker,
|
|
2159
|
+
owner_label="probe",
|
|
2160
|
+
missing_is_error=True,
|
|
2161
|
+
)
|
|
2162
|
+
if cleanup_error is not None:
|
|
2163
|
+
cleanup_errors.append(cleanup_error)
|
|
2164
|
+
|
|
2165
|
+
# Directory durability is best effort everywhere else in this module.
|
|
2166
|
+
fsync_parent_directory(probe_source)
|
|
2167
|
+
|
|
2168
|
+
if cleanup_errors:
|
|
2169
|
+
operation_detail = ""
|
|
2170
|
+
if operation_error is not None:
|
|
2171
|
+
operation_detail = (
|
|
2172
|
+
f"; operation_error during {operation_error[0]}={_filesystem_error_detail(operation_error[1])}"
|
|
2173
|
+
)
|
|
2174
|
+
raise RuntimeError(
|
|
2175
|
+
f"{label} hard-link preflight cleanup failed before live replacement; "
|
|
2176
|
+
"the target file was not moved or replaced; "
|
|
2177
|
+
f"retained probe detail={'; '.join(cleanup_errors)}{operation_detail}"
|
|
2178
|
+
) from (operation_error[1] if operation_error is not None else None)
|
|
2179
|
+
|
|
2180
|
+
if operation_error is not None:
|
|
2181
|
+
phase, error = operation_error
|
|
2182
|
+
raise RuntimeError(
|
|
2183
|
+
f"{label} hard-link preflight failed before live replacement; "
|
|
2184
|
+
"the target file was not moved or replaced; "
|
|
2185
|
+
f"phase={phase}; cause={_filesystem_error_detail(error)}"
|
|
2186
|
+
) from error
|
|
2187
|
+
|
|
2188
|
+
|
|
2189
|
+
def _preflight_target_volume_hardlink_support(target_path: pathlib.Path) -> None:
|
|
2190
|
+
_preflight_hardlink_support(target_path.parent, "target-volume")
|
|
2191
|
+
|
|
2192
|
+
|
|
2002
2193
|
def _publish_no_clobber(source_path: pathlib.Path, destination_path: pathlib.Path) -> None:
|
|
2003
2194
|
"""Atomically create destination without replacing any concurrent claimant."""
|
|
2004
2195
|
if destination_path.exists():
|
|
@@ -2009,7 +2200,8 @@ def _publish_no_clobber(source_path: pathlib.Path, destination_path: pathlib.Pat
|
|
|
2009
2200
|
raise
|
|
2010
2201
|
except OSError as exc:
|
|
2011
2202
|
raise RuntimeError(
|
|
2012
|
-
"atomic no-clobber publication requires same-volume hard-link support"
|
|
2203
|
+
"atomic no-clobber publication requires same-volume hard-link support; "
|
|
2204
|
+
f"os.link={_filesystem_error_detail(exc)}"
|
|
2013
2205
|
) from exc
|
|
2014
2206
|
fsync_parent_directory(destination_path)
|
|
2015
2207
|
|
|
@@ -2039,6 +2231,7 @@ def _restore_capture_no_clobber(
|
|
|
2039
2231
|
*,
|
|
2040
2232
|
validate_jsonl: bool,
|
|
2041
2233
|
) -> None:
|
|
2234
|
+
capture_identity = capture_path.lstat()
|
|
2042
2235
|
_publish_no_clobber(capture_path, target_path)
|
|
2043
2236
|
restored_bytes = target_path.read_bytes()
|
|
2044
2237
|
if restored_bytes != expected_bytes:
|
|
@@ -2047,11 +2240,14 @@ def _restore_capture_no_clobber(
|
|
|
2047
2240
|
validation = validate_jsonl_bytes(restored_bytes, source_label=target_path.name)
|
|
2048
2241
|
if not validation.get("ok"):
|
|
2049
2242
|
raise RuntimeError(f"restored source validation failed: {validation.get('errors')}")
|
|
2050
|
-
|
|
2051
|
-
|
|
2052
|
-
|
|
2053
|
-
|
|
2054
|
-
|
|
2243
|
+
cleanup_error = _cleanup_owned_file(
|
|
2244
|
+
capture_path,
|
|
2245
|
+
identity=capture_identity,
|
|
2246
|
+
expected_bytes=expected_bytes,
|
|
2247
|
+
owner_label="restored capture",
|
|
2248
|
+
)
|
|
2249
|
+
if cleanup_error is not None:
|
|
2250
|
+
raise RuntimeError(f"restored capture cleanup failed: {cleanup_error}")
|
|
2055
2251
|
fsync_parent_directory(target_path)
|
|
2056
2252
|
|
|
2057
2253
|
|
|
@@ -2075,22 +2271,34 @@ def _replace_file_after_validation(
|
|
|
2075
2271
|
source_sha256 = sha256_hex(source_bytes)
|
|
2076
2272
|
if expected_source_sha256 is not None and source_sha256 != expected_source_sha256:
|
|
2077
2273
|
raise RuntimeError("input JSONL changed after candidate generation; original file was not replaced")
|
|
2274
|
+
_preflight_target_volume_hardlink_support(target_path)
|
|
2275
|
+
if backup_dir is not None:
|
|
2276
|
+
backup_dir.mkdir(parents=True, exist_ok=True)
|
|
2277
|
+
_preflight_hardlink_support(backup_dir, "backup-directory")
|
|
2078
2278
|
tmp_replace = target_path.with_name(f".{target_path.name}.replace-{uuid.uuid4().hex}.tmp")
|
|
2079
2279
|
old_capture = target_path.with_name(f".{target_path.name}.old-{uuid.uuid4().hex}.tmp")
|
|
2280
|
+
tmp_replace_identity: Optional[os.stat_result] = None
|
|
2281
|
+
tmp_replace_expected_bytes: Optional[bytes] = None
|
|
2282
|
+
old_capture_identity: Optional[os.stat_result] = None
|
|
2283
|
+
rollback_capture_identity: Optional[os.stat_result] = None
|
|
2080
2284
|
backup: Optional[pathlib.Path] = None
|
|
2081
2285
|
published = False
|
|
2082
2286
|
directory_fsync = False
|
|
2083
2287
|
retain_old_capture = False
|
|
2084
2288
|
external_target_preserved = False
|
|
2085
2289
|
retained_rollback_capture: Optional[pathlib.Path] = None
|
|
2290
|
+
rollback_capture: Optional[pathlib.Path] = None
|
|
2291
|
+
cleanup_errors: List[Dict[str, str]] = []
|
|
2086
2292
|
replaced_validation: Dict[str, Any] = {}
|
|
2087
2293
|
try:
|
|
2088
2294
|
with tmp_replace.open("xb") as f:
|
|
2295
|
+
tmp_replace_identity = tmp_replace.lstat()
|
|
2089
2296
|
f.write(candidate_bytes)
|
|
2090
2297
|
f.flush()
|
|
2091
2298
|
os.fsync(f.fileno())
|
|
2092
2299
|
if sha256_hex(tmp_replace.read_bytes()) != candidate_sha256:
|
|
2093
2300
|
raise RuntimeError("staged replacement bytes do not match the validated candidate snapshot")
|
|
2301
|
+
tmp_replace_expected_bytes = candidate_bytes
|
|
2094
2302
|
if target_path.read_bytes() != source_bytes:
|
|
2095
2303
|
raise RuntimeError("input JSONL changed before backup; original file was not replaced")
|
|
2096
2304
|
backup = _exclusive_backup_from_bytes(target_path, source_bytes, backup_dir=backup_dir)
|
|
@@ -2101,15 +2309,11 @@ def _replace_file_after_validation(
|
|
|
2101
2309
|
captured_bytes = old_capture.read_bytes()
|
|
2102
2310
|
if captured_bytes != source_bytes:
|
|
2103
2311
|
raise RuntimeError("input JSONL changed during replacement capture; candidate was not installed")
|
|
2312
|
+
old_capture_identity = old_capture.lstat()
|
|
2104
2313
|
if target_path.exists():
|
|
2105
2314
|
raise RuntimeError("input JSONL was recreated concurrently; external bytes were left in place")
|
|
2106
2315
|
_publish_no_clobber(tmp_replace, target_path)
|
|
2107
2316
|
published = True
|
|
2108
|
-
try:
|
|
2109
|
-
if os.path.samefile(tmp_replace, target_path):
|
|
2110
|
-
tmp_replace.unlink()
|
|
2111
|
-
except (FileNotFoundError, OSError):
|
|
2112
|
-
pass
|
|
2113
2317
|
directory_fsync = fsync_parent_directory(target_path) or directory_fsync
|
|
2114
2318
|
published_bytes = target_path.read_bytes()
|
|
2115
2319
|
if sha256_hex(published_bytes) != candidate_sha256:
|
|
@@ -2120,8 +2324,6 @@ def _replace_file_after_validation(
|
|
|
2120
2324
|
backup = _ensure_verified_source_backup(target_path, source_bytes, backup, backup_dir)
|
|
2121
2325
|
if old_capture.read_bytes() != source_bytes:
|
|
2122
2326
|
raise RuntimeError("retained source capture changed before successful cleanup")
|
|
2123
|
-
old_capture.unlink()
|
|
2124
|
-
directory_fsync = fsync_parent_directory(target_path) or directory_fsync
|
|
2125
2327
|
except Exception as exc:
|
|
2126
2328
|
rollback_error: Optional[Exception] = None
|
|
2127
2329
|
restored = False
|
|
@@ -2147,6 +2349,7 @@ def _replace_file_after_validation(
|
|
|
2147
2349
|
except FileNotFoundError:
|
|
2148
2350
|
rollback_capture = None
|
|
2149
2351
|
if rollback_capture is not None:
|
|
2352
|
+
rollback_capture_identity = rollback_capture.lstat()
|
|
2150
2353
|
actual_target_bytes = rollback_capture.read_bytes()
|
|
2151
2354
|
if actual_target_bytes == candidate_bytes:
|
|
2152
2355
|
try:
|
|
@@ -2157,11 +2360,12 @@ def _replace_file_after_validation(
|
|
|
2157
2360
|
validate_jsonl=True,
|
|
2158
2361
|
)
|
|
2159
2362
|
restored = True
|
|
2160
|
-
if rollback_capture.read_bytes() == candidate_bytes:
|
|
2161
|
-
rollback_capture.unlink()
|
|
2162
2363
|
except FileExistsError:
|
|
2163
2364
|
external_target_preserved = True
|
|
2164
2365
|
retained_rollback_capture = rollback_capture
|
|
2366
|
+
except Exception:
|
|
2367
|
+
retained_rollback_capture = rollback_capture
|
|
2368
|
+
raise
|
|
2165
2369
|
else:
|
|
2166
2370
|
external_target_preserved = True
|
|
2167
2371
|
try:
|
|
@@ -2173,16 +2377,15 @@ def _replace_file_after_validation(
|
|
|
2173
2377
|
)
|
|
2174
2378
|
except FileExistsError:
|
|
2175
2379
|
retained_rollback_capture = rollback_capture
|
|
2380
|
+
except Exception:
|
|
2381
|
+
retained_rollback_capture = rollback_capture
|
|
2382
|
+
raise
|
|
2176
2383
|
if backup is None or backup.read_bytes() != source_bytes:
|
|
2177
2384
|
raise RuntimeError("verified source backup is unavailable during concurrent-target recovery")
|
|
2178
|
-
if old_capture.read_bytes() == source_bytes:
|
|
2179
|
-
old_capture.unlink()
|
|
2180
2385
|
elif target_path.exists():
|
|
2181
2386
|
external_target_preserved = True
|
|
2182
2387
|
if backup is None or backup.read_bytes() != source_bytes:
|
|
2183
2388
|
raise RuntimeError("verified source backup is unavailable during concurrent-target recovery")
|
|
2184
|
-
if old_capture.read_bytes() == source_bytes:
|
|
2185
|
-
old_capture.unlink()
|
|
2186
2389
|
else:
|
|
2187
2390
|
try:
|
|
2188
2391
|
_restore_capture_no_clobber(
|
|
@@ -2217,14 +2420,37 @@ def _replace_file_after_validation(
|
|
|
2217
2420
|
raise RuntimeError(f"replacement failed and original bytes were restored: {exc}") from exc
|
|
2218
2421
|
raise
|
|
2219
2422
|
finally:
|
|
2220
|
-
|
|
2423
|
+
active_error = sys.exc_info()[1]
|
|
2424
|
+
transients = [
|
|
2425
|
+
(tmp_replace, tmp_replace_identity, tmp_replace_expected_bytes, "replacement stage"),
|
|
2426
|
+
]
|
|
2221
2427
|
if not retain_old_capture:
|
|
2222
|
-
transients.append(old_capture)
|
|
2223
|
-
|
|
2224
|
-
|
|
2225
|
-
|
|
2226
|
-
|
|
2227
|
-
|
|
2428
|
+
transients.append((old_capture, old_capture_identity, source_bytes, "source capture"))
|
|
2429
|
+
if rollback_capture is not None and retained_rollback_capture is None:
|
|
2430
|
+
transients.append((rollback_capture, rollback_capture_identity, candidate_bytes, "rollback capture"))
|
|
2431
|
+
removed_transient = False
|
|
2432
|
+
for transient, identity, expected_bytes, owner_label in transients:
|
|
2433
|
+
cleanup_error = _cleanup_owned_file(
|
|
2434
|
+
transient,
|
|
2435
|
+
identity=identity,
|
|
2436
|
+
expected_bytes=expected_bytes,
|
|
2437
|
+
owner_label=owner_label,
|
|
2438
|
+
)
|
|
2439
|
+
if cleanup_error is None:
|
|
2440
|
+
if identity is not None:
|
|
2441
|
+
removed_transient = True
|
|
2442
|
+
else:
|
|
2443
|
+
cleanup_errors.append(
|
|
2444
|
+
{
|
|
2445
|
+
"path": transient.name,
|
|
2446
|
+
"error": cleanup_error,
|
|
2447
|
+
}
|
|
2448
|
+
)
|
|
2449
|
+
if removed_transient:
|
|
2450
|
+
directory_fsync = fsync_parent_directory(target_path) or directory_fsync
|
|
2451
|
+
if cleanup_errors and active_error is not None:
|
|
2452
|
+
cleanup_detail = json.dumps(cleanup_errors, ensure_ascii=False, separators=(",", ":"))
|
|
2453
|
+
raise RuntimeError(f"{active_error}; replacement cleanup errors={cleanup_detail}") from active_error
|
|
2228
2454
|
if backup is None:
|
|
2229
2455
|
raise AssertionError("replacement completed without an auditable backup")
|
|
2230
2456
|
return {
|
|
@@ -2234,6 +2460,8 @@ def _replace_file_after_validation(
|
|
|
2234
2460
|
"candidate_sha256": candidate_sha256,
|
|
2235
2461
|
"published_sha256": candidate_sha256,
|
|
2236
2462
|
"parent_directory_fsync": directory_fsync,
|
|
2463
|
+
"operation_state": "committed-cleanup-failed" if cleanup_errors else "committed",
|
|
2464
|
+
"cleanup_errors": cleanup_errors,
|
|
2237
2465
|
}
|
|
2238
2466
|
|
|
2239
2467
|
|
|
@@ -2361,6 +2589,109 @@ def stable_digest(text: str) -> str:
|
|
|
2361
2589
|
return hashlib.sha256(text.encode("utf-8", errors="replace")).hexdigest()[:16]
|
|
2362
2590
|
|
|
2363
2591
|
|
|
2592
|
+
def _bounded_diagnostic_label(value: Any, *, missing: bool = False) -> str:
|
|
2593
|
+
"""Render schema metadata for reports without copying arbitrary JSON values."""
|
|
2594
|
+
if missing:
|
|
2595
|
+
return "<missing>"
|
|
2596
|
+
if isinstance(value, str):
|
|
2597
|
+
if len(value) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS:
|
|
2598
|
+
return f"<string:length={len(value)};sha256={stable_digest(value)}>"
|
|
2599
|
+
return f"<{value}" if value.startswith("<") else value
|
|
2600
|
+
if value is None:
|
|
2601
|
+
return "<null>"
|
|
2602
|
+
return f"<invalid:{type(value).__name__}>"
|
|
2603
|
+
|
|
2604
|
+
|
|
2605
|
+
def _long_diagnostic_strings(value: Any) -> List[str]:
|
|
2606
|
+
"""Collect source strings that must not be copied into public diagnostics."""
|
|
2607
|
+
found: set = set()
|
|
2608
|
+
|
|
2609
|
+
def visit(item: Any) -> None:
|
|
2610
|
+
if isinstance(item, str):
|
|
2611
|
+
if len(item) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS:
|
|
2612
|
+
found.add(item)
|
|
2613
|
+
elif isinstance(item, dict):
|
|
2614
|
+
for key, nested in item.items():
|
|
2615
|
+
visit(key)
|
|
2616
|
+
visit(nested)
|
|
2617
|
+
elif isinstance(item, (list, tuple)):
|
|
2618
|
+
for nested in item:
|
|
2619
|
+
visit(nested)
|
|
2620
|
+
|
|
2621
|
+
visit(value)
|
|
2622
|
+
return sorted(found, key=len, reverse=True)
|
|
2623
|
+
|
|
2624
|
+
|
|
2625
|
+
def _redact_public_diagnostic_text(text: str, redactions: Sequence[str]) -> Tuple[str, bool]:
|
|
2626
|
+
"""Replace exact/repr/JSON renderings of oversized source values in text."""
|
|
2627
|
+
redacted = text
|
|
2628
|
+
replaced = False
|
|
2629
|
+
for raw in redactions:
|
|
2630
|
+
label = _bounded_diagnostic_label(raw)
|
|
2631
|
+
for source, replacement in (
|
|
2632
|
+
(raw, label),
|
|
2633
|
+
(repr(raw), repr(label)),
|
|
2634
|
+
(json.dumps(raw, ensure_ascii=False), json.dumps(label, ensure_ascii=False)),
|
|
2635
|
+
):
|
|
2636
|
+
if source in redacted:
|
|
2637
|
+
redacted = redacted.replace(source, replacement)
|
|
2638
|
+
replaced = True
|
|
2639
|
+
return redacted, replaced
|
|
2640
|
+
|
|
2641
|
+
|
|
2642
|
+
def _bounded_public_diagnostic_structure(
|
|
2643
|
+
value: Any,
|
|
2644
|
+
*,
|
|
2645
|
+
redactions: Sequence[str] = (),
|
|
2646
|
+
diagnostic_text: bool = False,
|
|
2647
|
+
redact_unmatched_diagnostic_text: bool = False,
|
|
2648
|
+
) -> Any:
|
|
2649
|
+
"""Bound oversized fields while retaining ordinary human-readable diagnostics."""
|
|
2650
|
+
if isinstance(value, str):
|
|
2651
|
+
if diagnostic_text:
|
|
2652
|
+
redacted, replaced = _redact_public_diagnostic_text(value, redactions)
|
|
2653
|
+
if replaced or not redact_unmatched_diagnostic_text:
|
|
2654
|
+
return redacted
|
|
2655
|
+
return _bounded_diagnostic_label(value) if len(value) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS else value
|
|
2656
|
+
return _bounded_diagnostic_label(value) if len(value) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS else value
|
|
2657
|
+
if isinstance(value, list):
|
|
2658
|
+
return [
|
|
2659
|
+
_bounded_public_diagnostic_structure(
|
|
2660
|
+
item,
|
|
2661
|
+
redactions=redactions,
|
|
2662
|
+
diagnostic_text=diagnostic_text,
|
|
2663
|
+
redact_unmatched_diagnostic_text=redact_unmatched_diagnostic_text,
|
|
2664
|
+
)
|
|
2665
|
+
for item in value
|
|
2666
|
+
]
|
|
2667
|
+
if isinstance(value, tuple):
|
|
2668
|
+
return tuple(
|
|
2669
|
+
_bounded_public_diagnostic_structure(
|
|
2670
|
+
item,
|
|
2671
|
+
redactions=redactions,
|
|
2672
|
+
diagnostic_text=diagnostic_text,
|
|
2673
|
+
redact_unmatched_diagnostic_text=redact_unmatched_diagnostic_text,
|
|
2674
|
+
)
|
|
2675
|
+
for item in value
|
|
2676
|
+
)
|
|
2677
|
+
if isinstance(value, dict):
|
|
2678
|
+
projected: Dict[Any, Any] = {}
|
|
2679
|
+
for key, item in value.items():
|
|
2680
|
+
public_key = (
|
|
2681
|
+
_bounded_diagnostic_label(key)
|
|
2682
|
+
if isinstance(key, str) and len(key) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS
|
|
2683
|
+
else key
|
|
2684
|
+
)
|
|
2685
|
+
projected[public_key] = _bounded_public_diagnostic_structure(
|
|
2686
|
+
item,
|
|
2687
|
+
redactions=redactions,
|
|
2688
|
+
diagnostic_text=diagnostic_text or key in {"errors", "warnings"},
|
|
2689
|
+
redact_unmatched_diagnostic_text=redact_unmatched_diagnostic_text,
|
|
2690
|
+
)
|
|
2691
|
+
return projected
|
|
2692
|
+
return value
|
|
2693
|
+
|
|
2694
|
+
|
|
2364
2695
|
def sha256_hex(data: bytes) -> str:
|
|
2365
2696
|
return hashlib.sha256(data).hexdigest()
|
|
2366
2697
|
|
|
@@ -2438,6 +2769,12 @@ def one_line(text: str, limit: int = 220) -> str:
|
|
|
2438
2769
|
return truncate(text, limit).replace("\n", " ")
|
|
2439
2770
|
|
|
2440
2771
|
|
|
2772
|
+
def _bounded_diagnostic_text(value: Any, limit: int = 1200) -> str:
|
|
2773
|
+
if isinstance(value, str):
|
|
2774
|
+
return one_line(truncate(value, limit), limit)
|
|
2775
|
+
return _bounded_diagnostic_label(value)
|
|
2776
|
+
|
|
2777
|
+
|
|
2441
2778
|
def block_text(block: Any) -> str:
|
|
2442
2779
|
if isinstance(block, str):
|
|
2443
2780
|
return block
|
|
@@ -2554,7 +2891,14 @@ def record_text(obj: JsonObj) -> str:
|
|
|
2554
2891
|
if t == "attachment":
|
|
2555
2892
|
return attachment_text(obj)
|
|
2556
2893
|
if t == "system":
|
|
2557
|
-
|
|
2894
|
+
pieces: List[str] = []
|
|
2895
|
+
if "subtype" in obj and obj.get("subtype") not in (None, ""):
|
|
2896
|
+
pieces.append(_bounded_diagnostic_label(obj.get("subtype")))
|
|
2897
|
+
for key in ("content", "error"):
|
|
2898
|
+
value = obj.get(key)
|
|
2899
|
+
if value not in (None, ""):
|
|
2900
|
+
pieces.append(_bounded_diagnostic_text(value))
|
|
2901
|
+
return " ".join(pieces)
|
|
2558
2902
|
if t == "file-history-snapshot":
|
|
2559
2903
|
snap = obj.get("snapshot")
|
|
2560
2904
|
if isinstance(snap, dict):
|
|
@@ -2694,7 +3038,12 @@ def ordered_subsequence(items: Sequence[str], expected: Sequence[str]) -> bool:
|
|
|
2694
3038
|
|
|
2695
3039
|
|
|
2696
3040
|
def is_api_message(obj: JsonObj) -> bool:
|
|
2697
|
-
|
|
3041
|
+
record_type = obj.get("type")
|
|
3042
|
+
return (
|
|
3043
|
+
isinstance(record_type, str)
|
|
3044
|
+
and record_type in {"user", "assistant"}
|
|
3045
|
+
and isinstance(obj.get("message"), dict)
|
|
3046
|
+
)
|
|
2698
3047
|
|
|
2699
3048
|
|
|
2700
3049
|
def api_role(obj: JsonObj) -> Optional[str]:
|
|
@@ -2913,6 +3262,16 @@ def _allowed_post_prompt_closure(obj: JsonObj, pending_tool_ids: set) -> Tuple[b
|
|
|
2913
3262
|
return False, f"unsupported_record_type:{obj.get('type')}"
|
|
2914
3263
|
|
|
2915
3264
|
|
|
3265
|
+
def _reject_resume_path(info: Dict[str, Any], reason_code: str, error: str) -> Dict[str, Any]:
|
|
3266
|
+
status = RESUME_REASON_CODE_STATUS.get(reason_code)
|
|
3267
|
+
if status is None or status == "valid":
|
|
3268
|
+
raise AssertionError(f"invalid resume-path rejection code: {reason_code}")
|
|
3269
|
+
info["status"] = status
|
|
3270
|
+
info["reasonCode"] = reason_code
|
|
3271
|
+
info["errors"].append(error)
|
|
3272
|
+
return info
|
|
3273
|
+
|
|
3274
|
+
|
|
2916
3275
|
def choose_resume_leaf_info(
|
|
2917
3276
|
records: Sequence[JsonObj],
|
|
2918
3277
|
max_post_prompt_extension: int = 0,
|
|
@@ -2928,7 +3287,8 @@ def choose_resume_leaf_info(
|
|
|
2928
3287
|
)
|
|
2929
3288
|
info: Dict[str, Any] = {
|
|
2930
3289
|
"ok": False,
|
|
2931
|
-
"status": "
|
|
3290
|
+
"status": "unvalidated",
|
|
3291
|
+
"reasonCode": None,
|
|
2932
3292
|
"errors": [],
|
|
2933
3293
|
"warnings": [],
|
|
2934
3294
|
"lastPromptIndex": last_prompt_entry[0] if last_prompt_entry else None,
|
|
@@ -2956,6 +3316,7 @@ def choose_resume_leaf_info(
|
|
|
2956
3316
|
"sessionLineageRunCount": 0,
|
|
2957
3317
|
"sessionLineageTransitionCount": 0,
|
|
2958
3318
|
"sessionLineageRunsDigest": None,
|
|
3319
|
+
"lineageReason": None,
|
|
2959
3320
|
"currentSessionStartPosition": 0,
|
|
2960
3321
|
"currentSessionRecordCount": 0,
|
|
2961
3322
|
"currentSessionId": None,
|
|
@@ -2964,19 +3325,26 @@ def choose_resume_leaf_info(
|
|
|
2964
3325
|
"authoritySessionId": last_prompt_entry[1].get("sessionId") if last_prompt_entry else None,
|
|
2965
3326
|
}
|
|
2966
3327
|
if duplicate_uuids:
|
|
2967
|
-
|
|
2968
|
-
|
|
2969
|
-
|
|
3328
|
+
return _reject_resume_path(
|
|
3329
|
+
info,
|
|
3330
|
+
"duplicate-uuid",
|
|
3331
|
+
f"duplicate uuid values make resume topology ambiguous: {duplicate_uuids[:20]}",
|
|
3332
|
+
)
|
|
2970
3333
|
if last_prompt_entry is None:
|
|
2971
|
-
|
|
2972
|
-
|
|
3334
|
+
return _reject_resume_path(
|
|
3335
|
+
info,
|
|
3336
|
+
"last-prompt-absent",
|
|
3337
|
+
"strict active-chain mode requires a last-prompt record",
|
|
3338
|
+
)
|
|
2973
3339
|
|
|
2974
3340
|
prompt_index, prompt_record = last_prompt_entry
|
|
2975
3341
|
prompt_leaf = resume_leaf_override or prompt_record.get("leafUuid")
|
|
2976
3342
|
if not isinstance(prompt_leaf, str) or not prompt_leaf:
|
|
2977
|
-
|
|
2978
|
-
|
|
2979
|
-
|
|
3343
|
+
return _reject_resume_path(
|
|
3344
|
+
info,
|
|
3345
|
+
"leaf-uuid-malformed",
|
|
3346
|
+
"authoritative last-prompt leafUuid is missing or malformed",
|
|
3347
|
+
)
|
|
2980
3348
|
info["selectedLeafUuid"] = prompt_leaf
|
|
2981
3349
|
trace = chain_trace_from_leaf(records, prompt_leaf)
|
|
2982
3350
|
info["promptChainMissingUuid"] = trace.get("missingUuid")
|
|
@@ -2984,25 +3352,27 @@ def choose_resume_leaf_info(
|
|
|
2984
3352
|
info["promptChainMalformedParentUuid"] = trace.get("malformedParentUuid")
|
|
2985
3353
|
info["promptChainMalformedParentType"] = trace.get("malformedParentType")
|
|
2986
3354
|
if trace.get("missingUuid"):
|
|
2987
|
-
|
|
2988
|
-
|
|
2989
|
-
|
|
3355
|
+
return _reject_resume_path(
|
|
3356
|
+
info,
|
|
3357
|
+
"chain-missing-uuid",
|
|
3358
|
+
f"authoritative resume chain references missing uuid: {trace.get('missingUuid')}",
|
|
3359
|
+
)
|
|
2990
3360
|
if trace.get("loopUuid"):
|
|
2991
|
-
|
|
2992
|
-
|
|
2993
|
-
|
|
3361
|
+
return _reject_resume_path(
|
|
3362
|
+
info,
|
|
3363
|
+
"chain-loop",
|
|
3364
|
+
f"authoritative resume chain contains a loop at uuid: {trace.get('loopUuid')}",
|
|
3365
|
+
)
|
|
2994
3366
|
if trace.get("malformedParentUuid"):
|
|
2995
|
-
|
|
2996
|
-
|
|
3367
|
+
return _reject_resume_path(
|
|
3368
|
+
info,
|
|
3369
|
+
"chain-malformed-parent",
|
|
2997
3370
|
"authoritative resume chain contains a non-null, non-empty-string parentUuid "
|
|
2998
|
-
f"on uuid {trace.get('malformedParentUuid')} (type {trace.get('malformedParentType')})"
|
|
3371
|
+
f"on uuid {trace.get('malformedParentUuid')} (type {trace.get('malformedParentType')})",
|
|
2999
3372
|
)
|
|
3000
|
-
return info
|
|
3001
3373
|
chain = list(trace.get("chain") or [])
|
|
3002
3374
|
if not chain:
|
|
3003
|
-
info
|
|
3004
|
-
info["errors"].append("authoritative resume chain is empty")
|
|
3005
|
-
return info
|
|
3375
|
+
return _reject_resume_path(info, "chain-empty", "authoritative resume chain is empty")
|
|
3006
3376
|
uuid_to_index = {
|
|
3007
3377
|
obj.get("uuid"): idx for idx, obj in enumerate(records)
|
|
3008
3378
|
if isinstance(obj.get("uuid"), str) and obj.get("uuid")
|
|
@@ -3012,12 +3382,12 @@ def choose_resume_leaf_info(
|
|
|
3012
3382
|
info["nonMonotonicEdgeCount"] = order_info["inversionCount"]
|
|
3013
3383
|
info["nonMonotonicCompatibilityEdgeCount"] = order_info["compatibilityEdgeCount"]
|
|
3014
3384
|
if not order_info["ok"]:
|
|
3015
|
-
|
|
3016
|
-
|
|
3385
|
+
return _reject_resume_path(
|
|
3386
|
+
info,
|
|
3387
|
+
"chain-non-monotonic",
|
|
3017
3388
|
"authoritative resume chain has non-monotonic physical parent edges outside the "
|
|
3018
|
-
"same-session attachment compatibility rule"
|
|
3389
|
+
"same-session attachment compatibility rule",
|
|
3019
3390
|
)
|
|
3020
|
-
return info
|
|
3021
3391
|
if order_info["compatibilityEdgeCount"]:
|
|
3022
3392
|
info["warnings"].append(
|
|
3023
3393
|
f"accepted {order_info['compatibilityEdgeCount']} same-session attachment parent edges "
|
|
@@ -3030,16 +3400,17 @@ def choose_resume_leaf_info(
|
|
|
3030
3400
|
info["sessionLineageRunCount"] = lineage_info["runCount"]
|
|
3031
3401
|
info["sessionLineageTransitionCount"] = lineage_info["transitionCount"]
|
|
3032
3402
|
info["sessionLineageRunsDigest"] = lineage_info["runsDigest"]
|
|
3403
|
+
info["lineageReason"] = lineage_info["reason"]
|
|
3033
3404
|
info["currentSessionStartPosition"] = lineage_info["currentSessionStartPosition"]
|
|
3034
3405
|
info["currentSessionRecordCount"] = lineage_info["currentSessionRecordCount"]
|
|
3035
3406
|
info["currentSessionId"] = lineage_info["currentSessionId"]
|
|
3036
3407
|
if not lineage_info["ok"]:
|
|
3037
|
-
|
|
3038
|
-
|
|
3408
|
+
return _reject_resume_path(
|
|
3409
|
+
info,
|
|
3410
|
+
"lineage-unsafe",
|
|
3039
3411
|
"authoritative resume chain has an unsafe sessionId lineage: "
|
|
3040
|
-
f"{lineage_info['reason']}"
|
|
3412
|
+
f"{lineage_info['reason']}",
|
|
3041
3413
|
)
|
|
3042
|
-
return info
|
|
3043
3414
|
if lineage_info["compatibility"]:
|
|
3044
3415
|
info["warnings"].append(
|
|
3045
3416
|
f"accepted one-way session lineage with {lineage_info['transitionCount']} transition(s); "
|
|
@@ -3050,50 +3421,58 @@ def choose_resume_leaf_info(
|
|
|
3050
3421
|
extension_pairs = list(enumerate(records[prompt_index + 1 :], start=prompt_index + 1))
|
|
3051
3422
|
if extension_pairs:
|
|
3052
3423
|
if len(extension_pairs) > max_post_prompt_extension:
|
|
3053
|
-
|
|
3054
|
-
|
|
3055
|
-
|
|
3424
|
+
return _reject_resume_path(
|
|
3425
|
+
info,
|
|
3426
|
+
"extension-limit-exceeded",
|
|
3427
|
+
f"post-last-prompt closure has {len(extension_pairs)} UUID records, "
|
|
3428
|
+
f"exceeding limit {max_post_prompt_extension}",
|
|
3056
3429
|
)
|
|
3057
|
-
return info
|
|
3058
3430
|
expected_parent = prompt_leaf
|
|
3059
3431
|
expected_session = prompt_record.get("sessionId")
|
|
3060
3432
|
if not isinstance(expected_session, str) or not expected_session:
|
|
3061
|
-
|
|
3062
|
-
|
|
3063
|
-
|
|
3433
|
+
return _reject_resume_path(
|
|
3434
|
+
info,
|
|
3435
|
+
"extension-authority-session-missing",
|
|
3436
|
+
"post-last-prompt closure requires a non-empty authority sessionId",
|
|
3437
|
+
)
|
|
3064
3438
|
pending_tool_ids = set(tool_use_ids(chain[-1]))
|
|
3065
3439
|
extension_reasons: List[str] = []
|
|
3066
3440
|
for idx, obj in extension_pairs:
|
|
3067
3441
|
if not isinstance(obj.get("uuid"), str) or not obj.get("uuid"):
|
|
3068
|
-
|
|
3069
|
-
|
|
3070
|
-
|
|
3442
|
+
return _reject_resume_path(
|
|
3443
|
+
info,
|
|
3444
|
+
"extension-record-missing-uuid",
|
|
3445
|
+
f"post-last-prompt record L{idx + 1} has no UUID and breaks the physical closure sequence",
|
|
3071
3446
|
)
|
|
3072
|
-
return info
|
|
3073
3447
|
if obj.get("parentUuid") != expected_parent:
|
|
3074
|
-
|
|
3075
|
-
|
|
3076
|
-
|
|
3448
|
+
return _reject_resume_path(
|
|
3449
|
+
info,
|
|
3450
|
+
"extension-record-not-linear-descendant",
|
|
3451
|
+
f"post-last-prompt record L{idx + 1} is not a direct linear descendant",
|
|
3452
|
+
)
|
|
3077
3453
|
obj_session = obj.get("sessionId")
|
|
3078
3454
|
if not isinstance(obj_session, str) or not obj_session or obj_session != expected_session:
|
|
3079
|
-
|
|
3080
|
-
|
|
3081
|
-
|
|
3455
|
+
return _reject_resume_path(
|
|
3456
|
+
info,
|
|
3457
|
+
"extension-record-session-mismatch",
|
|
3458
|
+
f"post-last-prompt record L{idx + 1} does not have the exact authority sessionId",
|
|
3082
3459
|
)
|
|
3083
|
-
return info
|
|
3084
3460
|
allowed, reason = _allowed_post_prompt_closure(obj, pending_tool_ids)
|
|
3085
3461
|
if not allowed:
|
|
3086
|
-
|
|
3087
|
-
|
|
3088
|
-
|
|
3462
|
+
return _reject_resume_path(
|
|
3463
|
+
info,
|
|
3464
|
+
"extension-record-not-safe-closure",
|
|
3465
|
+
f"post-last-prompt record L{idx + 1} is not a safe closure: {reason}",
|
|
3466
|
+
)
|
|
3089
3467
|
extension_reasons.append(reason)
|
|
3090
3468
|
expected_parent = obj.get("uuid")
|
|
3091
3469
|
if pending_tool_ids:
|
|
3092
|
-
|
|
3093
|
-
|
|
3094
|
-
|
|
3470
|
+
return _reject_resume_path(
|
|
3471
|
+
info,
|
|
3472
|
+
"extension-pending-tool-ids",
|
|
3473
|
+
f"post-last-prompt closure leaves pending tool_use ids unresolved: "
|
|
3474
|
+
f"{sorted(pending_tool_ids)[:20]}",
|
|
3095
3475
|
)
|
|
3096
|
-
return info
|
|
3097
3476
|
selected_extension_records = [obj for _, obj in extension_pairs]
|
|
3098
3477
|
chain.extend(selected_extension_records)
|
|
3099
3478
|
chain_indexes.extend(idx for idx, _ in extension_pairs)
|
|
@@ -3106,7 +3485,8 @@ def choose_resume_leaf_info(
|
|
|
3106
3485
|
info["postLastPromptExtensionUuids"] = [obj.get("uuid") for _, obj in extension_pairs]
|
|
3107
3486
|
info["currentSessionRecordCount"] = int(info.get("currentSessionRecordCount") or 0) + len(extension_pairs)
|
|
3108
3487
|
|
|
3109
|
-
info["status"] = "
|
|
3488
|
+
info["status"] = RESUME_REASON_CODE_STATUS["ok"]
|
|
3489
|
+
info["reasonCode"] = "ok"
|
|
3110
3490
|
info["ok"] = True
|
|
3111
3491
|
info["activeChainIndexes"] = chain_indexes
|
|
3112
3492
|
info["activeChainUuids"] = [obj.get("uuid") for obj in chain]
|
|
@@ -3120,13 +3500,18 @@ def require_resume_leaf_info(
|
|
|
3120
3500
|
) -> Dict[str, Any]:
|
|
3121
3501
|
info = choose_resume_leaf_info(records, max_post_prompt_extension, resume_leaf_override)
|
|
3122
3502
|
if not info.get("ok"):
|
|
3123
|
-
|
|
3503
|
+
reason_code = info.get("reasonCode") or "unknown"
|
|
3504
|
+
raise ValueError(
|
|
3505
|
+
f"strict active-chain topology failed (reasonCode={reason_code}): "
|
|
3506
|
+
+ "; ".join(info.get("errors") or [str(info.get("status"))])
|
|
3507
|
+
)
|
|
3124
3508
|
return info
|
|
3125
3509
|
|
|
3126
3510
|
|
|
3127
3511
|
def public_resume_leaf_info(info: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
|
3128
3512
|
if info is None:
|
|
3129
3513
|
return None
|
|
3514
|
+
redactions = _long_diagnostic_strings(info)
|
|
3130
3515
|
public = copy.deepcopy(info)
|
|
3131
3516
|
public.pop("lastPromptTemplate", None)
|
|
3132
3517
|
active_indexes = public.pop("activeChainIndexes", [])
|
|
@@ -3134,7 +3519,40 @@ def public_resume_leaf_info(info: Optional[Dict[str, Any]]) -> Optional[Dict[str
|
|
|
3134
3519
|
public["activeChainRecordCount"] = len(active_indexes)
|
|
3135
3520
|
public["activeChainLinesDigest"] = stable_digest(",".join(str(int(idx) + 1) for idx in active_indexes))
|
|
3136
3521
|
public["activeChainUuidsDigest"] = stable_digest(",".join(str(uid) for uid in active_uuids))
|
|
3137
|
-
|
|
3522
|
+
for key in (
|
|
3523
|
+
"promptLeafUuid",
|
|
3524
|
+
"physicalLeafUuid",
|
|
3525
|
+
"selectedLeafUuid",
|
|
3526
|
+
"promptChainMissingUuid",
|
|
3527
|
+
"promptChainLoopUuid",
|
|
3528
|
+
"promptChainMalformedParentUuid",
|
|
3529
|
+
"authoritySessionId",
|
|
3530
|
+
"currentSessionId",
|
|
3531
|
+
):
|
|
3532
|
+
if key in public and public[key] is not None:
|
|
3533
|
+
public[key] = _bounded_diagnostic_label(public[key])
|
|
3534
|
+
for key in ("duplicateUuids", "postLastPromptExtensionUuids"):
|
|
3535
|
+
values = public.get(key)
|
|
3536
|
+
if isinstance(values, list):
|
|
3537
|
+
public[key] = [_bounded_diagnostic_label(value) for value in values]
|
|
3538
|
+
return _bounded_public_diagnostic_structure(
|
|
3539
|
+
public,
|
|
3540
|
+
redactions=redactions,
|
|
3541
|
+
redact_unmatched_diagnostic_text=True,
|
|
3542
|
+
)
|
|
3543
|
+
|
|
3544
|
+
|
|
3545
|
+
def public_parent_repair_details(repairs: Sequence[JsonObj]) -> List[JsonObj]:
|
|
3546
|
+
"""Project repair diagnostics without copying arbitrary metadata verbatim."""
|
|
3547
|
+
redactions = _long_diagnostic_strings(repairs)
|
|
3548
|
+
public: List[JsonObj] = []
|
|
3549
|
+
for repair in repairs:
|
|
3550
|
+
projected = copy.deepcopy(repair)
|
|
3551
|
+
for key in ("uuid", "type", "sessionId", "oldParentUuid", "newParentUuid"):
|
|
3552
|
+
if key in projected and projected[key] is not None:
|
|
3553
|
+
projected[key] = _bounded_diagnostic_label(projected[key])
|
|
3554
|
+
public.append(projected)
|
|
3555
|
+
return _bounded_public_diagnostic_structure(public, redactions=redactions)
|
|
3138
3556
|
|
|
3139
3557
|
|
|
3140
3558
|
def choose_resume_leaf(records: Sequence[JsonObj]) -> Optional[str]:
|
|
@@ -3234,7 +3652,7 @@ def select_control_projection_indexes(records: Sequence[JsonObj]) -> List[int]:
|
|
|
3234
3652
|
ui_types = {"mode", "permission-mode", "custom-title", "ai-title", "agent-name"}
|
|
3235
3653
|
selected = [
|
|
3236
3654
|
idx for idx, obj in enumerate(records[:first_uuid_index])
|
|
3237
|
-
if obj.get("type") in ui_types and "uuid" not in obj
|
|
3655
|
+
if isinstance(obj.get("type"), str) and obj.get("type") in ui_types and "uuid" not in obj
|
|
3238
3656
|
]
|
|
3239
3657
|
selected.extend(idx for idx, obj in enumerate(records) if obj.get("type") == "last-prompt")
|
|
3240
3658
|
return sorted(set(selected))
|
|
@@ -3391,7 +3809,10 @@ def merge_assistant_fragments(api_records: Sequence[JsonObj]) -> List[JsonObj]:
|
|
|
3391
3809
|
if isinstance(target_msg, dict) and isinstance(obj_msg, dict):
|
|
3392
3810
|
target_blocks = content_blocks(target)
|
|
3393
3811
|
obj_blocks = content_blocks(obj)
|
|
3812
|
+
target_content_lines = _validation_content_lines(target, target.get("_line"))
|
|
3813
|
+
obj_content_lines = _validation_content_lines(obj, obj.get("_line"))
|
|
3394
3814
|
target_msg["content"] = target_blocks + obj_blocks
|
|
3815
|
+
target["_validationContentLines"] = target_content_lines + obj_content_lines
|
|
3395
3816
|
if isinstance(obj.get("uuid"), str):
|
|
3396
3817
|
target["uuid"] = obj.get("uuid")
|
|
3397
3818
|
merged_lines = list(target.get("_mergedLines", []))
|
|
@@ -3458,10 +3879,12 @@ def merge_split_tool_result_users(api_records: Sequence[JsonObj]) -> List[JsonOb
|
|
|
3458
3879
|
combined_msg = combined.get("message")
|
|
3459
3880
|
if isinstance(combined_msg, dict):
|
|
3460
3881
|
blocks: List[Any] = []
|
|
3882
|
+
content_lines: List[int] = []
|
|
3461
3883
|
merged_lines: List[Any] = []
|
|
3462
3884
|
merged_uuids: List[str] = []
|
|
3463
3885
|
for user_obj in group:
|
|
3464
3886
|
blocks.extend(content_blocks(user_obj))
|
|
3887
|
+
content_lines.extend(_validation_content_lines(user_obj, user_obj.get("_line")))
|
|
3465
3888
|
merged_lines.append(user_obj.get("_line"))
|
|
3466
3889
|
if isinstance(user_obj.get("uuid"), str):
|
|
3467
3890
|
merged_uuids.append(user_obj.get("uuid"))
|
|
@@ -3469,6 +3892,7 @@ def merge_split_tool_result_users(api_records: Sequence[JsonObj]) -> List[JsonOb
|
|
|
3469
3892
|
last_user = group[-1]
|
|
3470
3893
|
if isinstance(last_user.get("uuid"), str):
|
|
3471
3894
|
combined["uuid"] = last_user.get("uuid")
|
|
3895
|
+
combined["_validationContentLines"] = content_lines
|
|
3472
3896
|
combined["_mergedLines"] = merged_lines
|
|
3473
3897
|
if merged_uuids:
|
|
3474
3898
|
combined["_mergedUuids"] = merged_uuids
|
|
@@ -3477,15 +3901,102 @@ def merge_split_tool_result_users(api_records: Sequence[JsonObj]) -> List[JsonOb
|
|
|
3477
3901
|
return merged
|
|
3478
3902
|
|
|
3479
3903
|
|
|
3480
|
-
def
|
|
3904
|
+
def _validated_record_lines(
|
|
3905
|
+
records: Sequence[JsonObj],
|
|
3906
|
+
physical_lines: Optional[Sequence[int]] = None,
|
|
3907
|
+
*,
|
|
3908
|
+
allow_nonmonotonic: bool = False,
|
|
3909
|
+
) -> List[int]:
|
|
3910
|
+
if physical_lines is None:
|
|
3911
|
+
return list(range(1, len(records) + 1))
|
|
3912
|
+
lines = list(physical_lines)
|
|
3913
|
+
if len(lines) != len(records):
|
|
3914
|
+
raise ValueError("physical line map length does not match record count")
|
|
3915
|
+
previous = 0
|
|
3916
|
+
for line in lines:
|
|
3917
|
+
if isinstance(line, bool) or not isinstance(line, int) or line <= 0:
|
|
3918
|
+
raise ValueError("physical line map must contain positive integers")
|
|
3919
|
+
if not allow_nonmonotonic and line <= previous:
|
|
3920
|
+
raise ValueError("physical line map must contain strictly increasing positive integers")
|
|
3921
|
+
previous = line
|
|
3922
|
+
return lines
|
|
3923
|
+
|
|
3924
|
+
|
|
3925
|
+
def _diagnostic_counter_label(obj: JsonObj, key: str) -> str:
|
|
3926
|
+
return _bounded_diagnostic_label(obj.get(key), missing=key not in obj)
|
|
3927
|
+
|
|
3928
|
+
|
|
3929
|
+
def _validation_content_lines(obj: JsonObj, fallback_line: Any) -> List[int]:
|
|
3930
|
+
if isinstance(fallback_line, bool) or not isinstance(fallback_line, int) or fallback_line <= 0:
|
|
3931
|
+
fallback_line = 1
|
|
3932
|
+
blocks = content_blocks(obj)
|
|
3933
|
+
candidate = obj.get("_validationContentLines")
|
|
3934
|
+
if (
|
|
3935
|
+
isinstance(candidate, list)
|
|
3936
|
+
and len(candidate) == len(blocks)
|
|
3937
|
+
and all(isinstance(line, int) and not isinstance(line, bool) and line > 0 for line in candidate)
|
|
3938
|
+
):
|
|
3939
|
+
return list(candidate)
|
|
3940
|
+
return [fallback_line] * len(blocks)
|
|
3941
|
+
|
|
3942
|
+
|
|
3943
|
+
def _first_validation_block_line(obj: JsonObj, block_type: str, fallback_line: int) -> int:
|
|
3944
|
+
block_lines = _validation_content_lines(obj, fallback_line)
|
|
3945
|
+
for index, block in enumerate(content_blocks(obj)):
|
|
3946
|
+
if isinstance(block, dict) and block.get("type") == block_type:
|
|
3947
|
+
return block_lines[index]
|
|
3948
|
+
return fallback_line
|
|
3949
|
+
|
|
3950
|
+
|
|
3951
|
+
def _ordered_subsequence_failure_line(
|
|
3952
|
+
obj: JsonObj,
|
|
3953
|
+
expected: Sequence[str],
|
|
3954
|
+
fallback_line: int,
|
|
3955
|
+
) -> int:
|
|
3956
|
+
block_lines = _validation_content_lines(obj, fallback_line)
|
|
3957
|
+
position = 0
|
|
3958
|
+
for index, block in enumerate(content_blocks(obj)):
|
|
3959
|
+
if not isinstance(block, dict) or block.get("type") != "tool_result":
|
|
3960
|
+
continue
|
|
3961
|
+
tool_id = block.get("tool_use_id") or block.get("toolUseID")
|
|
3962
|
+
if not isinstance(tool_id, str):
|
|
3963
|
+
continue
|
|
3964
|
+
while position < len(expected) and expected[position] != tool_id:
|
|
3965
|
+
position += 1
|
|
3966
|
+
if position >= len(expected):
|
|
3967
|
+
return block_lines[index]
|
|
3968
|
+
position += 1
|
|
3969
|
+
return fallback_line
|
|
3970
|
+
|
|
3971
|
+
|
|
3972
|
+
def active_api_messages_for_validation(
|
|
3973
|
+
records: Sequence[JsonObj],
|
|
3974
|
+
physical_lines: Optional[Sequence[int]] = None,
|
|
3975
|
+
*,
|
|
3976
|
+
allow_nonmonotonic_diagnostic_lines: bool = False,
|
|
3977
|
+
) -> List[Tuple[int, JsonObj]]:
|
|
3978
|
+
record_lines = _validated_record_lines(
|
|
3979
|
+
records,
|
|
3980
|
+
physical_lines,
|
|
3981
|
+
allow_nonmonotonic=allow_nonmonotonic_diagnostic_lines,
|
|
3982
|
+
)
|
|
3983
|
+
resume_info = choose_resume_leaf_info(records)
|
|
3984
|
+
active_indexes = list(resume_info.get("activeChainIndexes") or []) if resume_info.get("ok") else []
|
|
3481
3985
|
api_records: List[JsonObj] = []
|
|
3482
|
-
for
|
|
3986
|
+
for record_index in active_indexes:
|
|
3987
|
+
obj = records[record_index]
|
|
3483
3988
|
if not is_api_message(obj):
|
|
3484
3989
|
continue
|
|
3485
|
-
clone =
|
|
3990
|
+
clone = {
|
|
3991
|
+
key: value
|
|
3992
|
+
for key, value in obj.items()
|
|
3993
|
+
if key not in {"_line", "_mergedLines", "_mergedUuids", "_validationContentLines"}
|
|
3994
|
+
}
|
|
3995
|
+
clone["_line"] = record_lines[record_index]
|
|
3996
|
+
clone["_validationContentLines"] = [record_lines[record_index]] * len(content_blocks(obj))
|
|
3486
3997
|
api_records.append(clone)
|
|
3487
3998
|
merged = merge_split_tool_result_users(merge_assistant_fragments(api_records))
|
|
3488
|
-
return [(
|
|
3999
|
+
return [(obj["_line"], obj) for obj in merged]
|
|
3489
4000
|
|
|
3490
4001
|
|
|
3491
4002
|
def adjust_recent_start_for_tool_pairs(records: Sequence[JsonObj], start: int) -> int:
|
|
@@ -3780,11 +4291,13 @@ def is_assistant_research_decision(obj: JsonObj, text: Optional[str] = None) ->
|
|
|
3780
4291
|
|
|
3781
4292
|
def collect_summary_inputs(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
3782
4293
|
ensure_summary_resources()
|
|
3783
|
-
type_counts = collections.Counter(obj
|
|
4294
|
+
type_counts = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in records)
|
|
3784
4295
|
subtype_counts = collections.Counter(
|
|
3785
|
-
f"{obj
|
|
4296
|
+
f"{_diagnostic_counter_label(obj, 'type')}:{_diagnostic_counter_label(obj, 'subtype')}"
|
|
4297
|
+
for obj in records
|
|
4298
|
+
if "subtype" in obj and obj.get("subtype") not in (None, "")
|
|
3786
4299
|
)
|
|
3787
|
-
session_counts = collections.Counter(obj
|
|
4300
|
+
session_counts = collections.Counter(_diagnostic_counter_label(obj, "sessionId") for obj in records)
|
|
3788
4301
|
cwd_counts = collections.Counter(str(obj.get("cwd")) for obj in records if obj.get("cwd"))
|
|
3789
4302
|
versions = collections.Counter(str(obj.get("version")) for obj in records if obj.get("version"))
|
|
3790
4303
|
tools = collections.Counter()
|
|
@@ -3850,7 +4363,11 @@ def collect_summary_inputs(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
3850
4363
|
user_items.append((idx, one_line(txt, 700 if is_important else 420)))
|
|
3851
4364
|
elif obj.get("type") == "system":
|
|
3852
4365
|
if obj.get("subtype") == "api_error" or obj.get("error"):
|
|
3853
|
-
|
|
4366
|
+
fallback = (
|
|
4367
|
+
f"system-error type={_diagnostic_counter_label(obj, 'type')} "
|
|
4368
|
+
f"subtype={_diagnostic_counter_label(obj, 'subtype')}"
|
|
4369
|
+
)
|
|
4370
|
+
errors.append((idx, one_line(txt or fallback, 520)))
|
|
3854
4371
|
elif obj.get("type") == "attachment":
|
|
3855
4372
|
att = obj.get("attachment")
|
|
3856
4373
|
if isinstance(att, dict):
|
|
@@ -4145,7 +4662,7 @@ def make_summary_text(
|
|
|
4145
4662
|
all_info = collect_summary_inputs(omitted)
|
|
4146
4663
|
early_info = collect_summary_inputs(early)
|
|
4147
4664
|
middle_info = collect_summary_inputs(middle)
|
|
4148
|
-
kept_types = collections.Counter(obj
|
|
4665
|
+
kept_types = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in kept)
|
|
4149
4666
|
first_ts = next((obj.get("timestamp") for obj in omitted if obj.get("timestamp")), "unknown")
|
|
4150
4667
|
last_omitted_ts = next((obj.get("timestamp") for obj in reversed(omitted) if obj.get("timestamp")), "unknown")
|
|
4151
4668
|
first_kept_ts = next((obj.get("timestamp") for obj in kept if obj.get("timestamp")), "unknown")
|
|
@@ -4823,10 +5340,12 @@ def model_evidence_record(obj: JsonObj, original_line: int) -> Optional[Dict[str
|
|
|
4823
5340
|
noisy = is_noisy_text(txt)
|
|
4824
5341
|
if not txt or (noisy and not mandatory_semantic and not is_prior_summary):
|
|
4825
5342
|
return None
|
|
4826
|
-
|
|
5343
|
+
record_type = _diagnostic_counter_label(obj, "type")
|
|
5344
|
+
role_value = api_role(obj)
|
|
5345
|
+
role = _bounded_diagnostic_label(role_value) if role_value is not None else record_type
|
|
4827
5346
|
timestamp = obj.get("timestamp") or ""
|
|
4828
5347
|
uid = obj.get("uuid") or ""
|
|
4829
|
-
prefix = f"L{original_line} type={
|
|
5348
|
+
prefix = f"L{original_line} type={record_type} role={role}"
|
|
4830
5349
|
if timestamp:
|
|
4831
5350
|
prefix += f" ts={timestamp}"
|
|
4832
5351
|
if uid:
|
|
@@ -5826,8 +6345,8 @@ def summarize_active_chain_window(records: Sequence[JsonObj], summary_uuid: str)
|
|
|
5826
6345
|
uuids = [obj.get("uuid") for obj in chain if isinstance(obj.get("uuid"), str)]
|
|
5827
6346
|
summary_pos = next((idx for idx, obj in enumerate(chain) if obj.get("uuid") == summary_uuid), None)
|
|
5828
6347
|
after_summary = chain[summary_pos + 1 :] if isinstance(summary_pos, int) else []
|
|
5829
|
-
session_counts = collections.Counter(obj
|
|
5830
|
-
type_counts = collections.Counter(obj
|
|
6348
|
+
session_counts = collections.Counter(_diagnostic_counter_label(obj, "sessionId") for obj in chain)
|
|
6349
|
+
type_counts = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in chain)
|
|
5831
6350
|
return {
|
|
5832
6351
|
"leafUuid": latest_leaf,
|
|
5833
6352
|
"chainLength": len(chain),
|
|
@@ -5960,8 +6479,11 @@ def select_preservation_plan(
|
|
|
5960
6479
|
def validate_source_active_chain_for_plan(
|
|
5961
6480
|
records: Sequence[JsonObj],
|
|
5962
6481
|
plan: Dict[str, Any],
|
|
6482
|
+
*,
|
|
6483
|
+
physical_lines: Optional[Sequence[int]] = None,
|
|
5963
6484
|
) -> Optional[Dict[str, Any]]:
|
|
5964
6485
|
"""Validate the authoritative logical chain without inspecting inactive branches."""
|
|
6486
|
+
source_lines = _validated_record_lines(records, physical_lines)
|
|
5965
6487
|
preserve_info = plan.get("preserve_info")
|
|
5966
6488
|
if not isinstance(preserve_info, dict):
|
|
5967
6489
|
return None
|
|
@@ -5971,7 +6493,16 @@ def validate_source_active_chain_for_plan(
|
|
|
5971
6493
|
active_indexes = list(resume_info.get("activeChainIndexes") or [])
|
|
5972
6494
|
if not active_indexes:
|
|
5973
6495
|
raise ValueError("source active-chain validation failed: authoritative chain is empty")
|
|
5974
|
-
|
|
6496
|
+
if any(
|
|
6497
|
+
isinstance(index, bool)
|
|
6498
|
+
or not isinstance(index, int)
|
|
6499
|
+
or index < 0
|
|
6500
|
+
or index >= len(records)
|
|
6501
|
+
for index in active_indexes
|
|
6502
|
+
):
|
|
6503
|
+
raise ValueError("source active-chain validation failed: preservation plan has invalid active-chain indexes")
|
|
6504
|
+
active_records = [copy.deepcopy(records[index]) for index in active_indexes]
|
|
6505
|
+
active_record_lines = [source_lines[index] for index in active_indexes]
|
|
5975
6506
|
selected_leaf = plan.get("selected_leaf_uuid")
|
|
5976
6507
|
if not isinstance(selected_leaf, str) or not selected_leaf:
|
|
5977
6508
|
raise ValueError("source active-chain validation failed: selected leaf is missing")
|
|
@@ -5990,7 +6521,17 @@ def validate_source_active_chain_for_plan(
|
|
|
5990
6521
|
if isinstance(leaf_session, str) and leaf_session:
|
|
5991
6522
|
pointer["sessionId"] = leaf_session
|
|
5992
6523
|
active_records.append(pointer)
|
|
5993
|
-
|
|
6524
|
+
pointer_index = resume_info.get("lastPromptIndex")
|
|
6525
|
+
if isinstance(pointer_index, int) and not isinstance(pointer_index, bool) and 0 <= pointer_index < len(records):
|
|
6526
|
+
pointer_line = source_lines[pointer_index]
|
|
6527
|
+
else:
|
|
6528
|
+
pointer_line = max(source_lines, default=0) + 1
|
|
6529
|
+
active_record_lines.append(pointer_line)
|
|
6530
|
+
validation = validate_records(
|
|
6531
|
+
active_records,
|
|
6532
|
+
physical_lines=active_record_lines,
|
|
6533
|
+
allow_nonmonotonic_diagnostic_lines=True,
|
|
6534
|
+
)
|
|
5994
6535
|
if not validation.get("ok"):
|
|
5995
6536
|
raise ValueError(
|
|
5996
6537
|
"source active-chain validation failed: "
|
|
@@ -6021,7 +6562,7 @@ def build_model_summary_pack_for_input(
|
|
|
6021
6562
|
if handoff_summary_path is not None and is_under_claude_root(handoff_summary_path):
|
|
6022
6563
|
raise ValueError("--handoff-summary process files must be outside the entire .claude directory")
|
|
6023
6564
|
source_bytes = input_path.read_bytes()
|
|
6024
|
-
records, raw_lines =
|
|
6565
|
+
records, raw_lines, physical_lines = parse_jsonl_bytes_with_lines(source_bytes, source_label=str(input_path))
|
|
6025
6566
|
if not records:
|
|
6026
6567
|
raise ValueError("input JSONL has no records")
|
|
6027
6568
|
input_bytes = len(source_bytes)
|
|
@@ -6041,7 +6582,11 @@ def build_model_summary_pack_for_input(
|
|
|
6041
6582
|
checkpoint_policy,
|
|
6042
6583
|
resume_leaf_override,
|
|
6043
6584
|
)
|
|
6044
|
-
source_active_chain_validation = validate_source_active_chain_for_plan(
|
|
6585
|
+
source_active_chain_validation = validate_source_active_chain_for_plan(
|
|
6586
|
+
records,
|
|
6587
|
+
plan,
|
|
6588
|
+
physical_lines=physical_lines,
|
|
6589
|
+
)
|
|
6045
6590
|
source_sha256 = sha256_hex(source_bytes)
|
|
6046
6591
|
source_digest = source_sha256
|
|
6047
6592
|
omitted_digest = sha256_hex("\n".join(raw_lines[idx] for idx in plan["omitted_indexes"]).encode("utf-8"))
|
|
@@ -6180,7 +6725,7 @@ def compress_jsonl(
|
|
|
6180
6725
|
if model_summary_path is not None and deterministic_summary:
|
|
6181
6726
|
raise ValueError("model_summary_path and deterministic_summary=True are mutually exclusive")
|
|
6182
6727
|
source_bytes = input_path.read_bytes()
|
|
6183
|
-
records, raw_lines =
|
|
6728
|
+
records, raw_lines, physical_lines = parse_jsonl_bytes_with_lines(source_bytes, source_label=str(input_path))
|
|
6184
6729
|
if not records:
|
|
6185
6730
|
raise ValueError("input JSONL has no records")
|
|
6186
6731
|
input_bytes = len(source_bytes)
|
|
@@ -6218,7 +6763,11 @@ def compress_jsonl(
|
|
|
6218
6763
|
checkpoint_policy,
|
|
6219
6764
|
resume_leaf_override,
|
|
6220
6765
|
)
|
|
6221
|
-
source_active_chain_validation = validate_source_active_chain_for_plan(
|
|
6766
|
+
source_active_chain_validation = validate_source_active_chain_for_plan(
|
|
6767
|
+
records,
|
|
6768
|
+
plan,
|
|
6769
|
+
physical_lines=physical_lines,
|
|
6770
|
+
)
|
|
6222
6771
|
preserve_info = plan["preserve_info"]
|
|
6223
6772
|
omitted = list(plan["omitted"])
|
|
6224
6773
|
kept_original = list(plan["kept_original"])
|
|
@@ -6526,7 +7075,7 @@ def compress_jsonl(
|
|
|
6526
7075
|
"omitted_digest": omitted_text_digest,
|
|
6527
7076
|
"safe_prefix_records": len(safe_prefix),
|
|
6528
7077
|
"parent_repairs": len(parent_repair_details),
|
|
6529
|
-
"parent_repair_details": parent_repair_details,
|
|
7078
|
+
"parent_repair_details": public_parent_repair_details(parent_repair_details),
|
|
6530
7079
|
"last_prompt_updates": last_prompt_updates,
|
|
6531
7080
|
"target_session_id": target_session_id,
|
|
6532
7081
|
"normalized_session_records": normalized_session_records,
|
|
@@ -6545,35 +7094,50 @@ def compress_jsonl(
|
|
|
6545
7094
|
return report
|
|
6546
7095
|
|
|
6547
7096
|
|
|
6548
|
-
def validate_records(
|
|
7097
|
+
def validate_records(
|
|
7098
|
+
records: Sequence[JsonObj],
|
|
7099
|
+
physical_lines: Optional[Sequence[int]] = None,
|
|
7100
|
+
*,
|
|
7101
|
+
allow_nonmonotonic_diagnostic_lines: bool = False,
|
|
7102
|
+
) -> Dict[str, Any]:
|
|
6549
7103
|
errors: List[str] = []
|
|
6550
7104
|
warnings: List[str] = []
|
|
7105
|
+
record_lines = _validated_record_lines(
|
|
7106
|
+
records,
|
|
7107
|
+
physical_lines,
|
|
7108
|
+
allow_nonmonotonic=allow_nonmonotonic_diagnostic_lines,
|
|
7109
|
+
)
|
|
6551
7110
|
uuid_counts = collections.Counter(obj.get("uuid") for obj in records if isinstance(obj.get("uuid"), str))
|
|
6552
7111
|
duplicates = sorted(k for k, v in uuid_counts.items() if v > 1)
|
|
6553
7112
|
if duplicates:
|
|
6554
7113
|
errors.append(f"duplicate UUIDs: {duplicates[:20]}")
|
|
6555
7114
|
uuid_set = set(uuid_counts)
|
|
6556
7115
|
uuid_to_record = {obj.get("uuid"): obj for obj in records if isinstance(obj.get("uuid"), str)}
|
|
6557
|
-
uuid_to_line = {
|
|
7116
|
+
uuid_to_line = {
|
|
7117
|
+
obj.get("uuid"): record_lines[index]
|
|
7118
|
+
for index, obj in enumerate(records)
|
|
7119
|
+
if isinstance(obj.get("uuid"), str)
|
|
7120
|
+
}
|
|
6558
7121
|
missing_parent: List[Tuple[int, str]] = []
|
|
6559
7122
|
malformed_parent: List[JsonObj] = []
|
|
6560
7123
|
cross_session_parent: List[JsonObj] = []
|
|
6561
|
-
for
|
|
7124
|
+
for record_index, obj in enumerate(records):
|
|
7125
|
+
line = record_lines[record_index]
|
|
6562
7126
|
parent = obj.get("parentUuid")
|
|
6563
7127
|
if parent is None:
|
|
6564
7128
|
continue
|
|
6565
7129
|
if not isinstance(parent, str) or not parent:
|
|
6566
7130
|
malformed_parent.append(
|
|
6567
7131
|
{
|
|
6568
|
-
"line":
|
|
7132
|
+
"line": line,
|
|
6569
7133
|
"uuid": obj.get("uuid"),
|
|
6570
|
-
"type": obj
|
|
7134
|
+
"type": _diagnostic_counter_label(obj, "type"),
|
|
6571
7135
|
"parentUuidType": type(parent).__name__,
|
|
6572
7136
|
}
|
|
6573
7137
|
)
|
|
6574
7138
|
continue
|
|
6575
7139
|
if parent not in uuid_set:
|
|
6576
|
-
missing_parent.append((
|
|
7140
|
+
missing_parent.append((line, str(parent)))
|
|
6577
7141
|
else:
|
|
6578
7142
|
parent_obj = uuid_to_record.get(parent)
|
|
6579
7143
|
child_session = obj.get("sessionId")
|
|
@@ -6581,12 +7145,12 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6581
7145
|
if isinstance(child_session, str) and isinstance(parent_session, str) and child_session != parent_session:
|
|
6582
7146
|
cross_session_parent.append(
|
|
6583
7147
|
{
|
|
6584
|
-
"line":
|
|
7148
|
+
"line": line,
|
|
6585
7149
|
"uuid": obj.get("uuid"),
|
|
6586
|
-
"type": obj
|
|
6587
|
-
"sessionId":
|
|
7150
|
+
"type": _diagnostic_counter_label(obj, "type"),
|
|
7151
|
+
"sessionId": _diagnostic_counter_label(obj, "sessionId"),
|
|
6588
7152
|
"parentUuid": parent,
|
|
6589
|
-
"parentSessionId": parent_session,
|
|
7153
|
+
"parentSessionId": _bounded_diagnostic_label(parent_session),
|
|
6590
7154
|
}
|
|
6591
7155
|
)
|
|
6592
7156
|
if malformed_parent:
|
|
@@ -6594,7 +7158,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6594
7158
|
if missing_parent:
|
|
6595
7159
|
errors.append(f"missing parentUuid references: {missing_parent[:30]}")
|
|
6596
7160
|
last_prompt_records = [
|
|
6597
|
-
(
|
|
7161
|
+
(record_lines[index], obj)
|
|
7162
|
+
for index, obj in enumerate(records)
|
|
7163
|
+
if obj.get("type") == "last-prompt"
|
|
6598
7164
|
]
|
|
6599
7165
|
last_prompt_malformed: List[JsonObj] = []
|
|
6600
7166
|
last_prompt_missing: List[JsonObj] = []
|
|
@@ -6603,12 +7169,22 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6603
7169
|
leaf = obj.get("leafUuid")
|
|
6604
7170
|
if not isinstance(leaf, str) or not leaf:
|
|
6605
7171
|
last_prompt_malformed.append(
|
|
6606
|
-
{
|
|
7172
|
+
{
|
|
7173
|
+
"line": idx,
|
|
7174
|
+
"leafUuidType": type(leaf).__name__,
|
|
7175
|
+
"sessionId": _diagnostic_counter_label(obj, "sessionId"),
|
|
7176
|
+
}
|
|
6607
7177
|
)
|
|
6608
7178
|
continue
|
|
6609
7179
|
target = uuid_to_record.get(leaf)
|
|
6610
7180
|
if target is None:
|
|
6611
|
-
last_prompt_missing.append(
|
|
7181
|
+
last_prompt_missing.append(
|
|
7182
|
+
{
|
|
7183
|
+
"line": idx,
|
|
7184
|
+
"leafUuid": _bounded_diagnostic_label(leaf),
|
|
7185
|
+
"sessionId": _diagnostic_counter_label(obj, "sessionId"),
|
|
7186
|
+
}
|
|
7187
|
+
)
|
|
6612
7188
|
continue
|
|
6613
7189
|
lp_session = obj.get("sessionId")
|
|
6614
7190
|
target_session = target.get("sessionId")
|
|
@@ -6616,9 +7192,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6616
7192
|
last_prompt_cross_session.append(
|
|
6617
7193
|
{
|
|
6618
7194
|
"line": idx,
|
|
6619
|
-
"leafUuid": leaf,
|
|
6620
|
-
"sessionId": lp_session,
|
|
6621
|
-
"leafSessionId": target_session,
|
|
7195
|
+
"leafUuid": _bounded_diagnostic_label(leaf),
|
|
7196
|
+
"sessionId": _bounded_diagnostic_label(lp_session),
|
|
7197
|
+
"leafSessionId": _bounded_diagnostic_label(target_session),
|
|
6622
7198
|
}
|
|
6623
7199
|
)
|
|
6624
7200
|
if last_prompt_missing:
|
|
@@ -6626,7 +7202,7 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6626
7202
|
if last_prompt_cross_session:
|
|
6627
7203
|
errors.append(f"last-prompt leafUuid cross-session targets: {last_prompt_cross_session[:20]}")
|
|
6628
7204
|
physical_last_prompt = latest_last_prompt_entry(records)
|
|
6629
|
-
physical_last_prompt_line = physical_last_prompt[0]
|
|
7205
|
+
physical_last_prompt_line = record_lines[physical_last_prompt[0]] if physical_last_prompt else None
|
|
6630
7206
|
physical_last_prompt_malformed = bool(
|
|
6631
7207
|
physical_last_prompt
|
|
6632
7208
|
and (not isinstance(physical_last_prompt[1].get("leafUuid"), str) or not physical_last_prompt[1].get("leafUuid"))
|
|
@@ -6637,32 +7213,38 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6637
7213
|
)
|
|
6638
7214
|
elif last_prompt_malformed:
|
|
6639
7215
|
warnings.append(f"earlier malformed last-prompt records: {last_prompt_malformed[:20]}")
|
|
6640
|
-
api_messages = active_api_messages_for_validation(
|
|
7216
|
+
api_messages = active_api_messages_for_validation(
|
|
7217
|
+
records,
|
|
7218
|
+
physical_lines=record_lines,
|
|
7219
|
+
allow_nonmonotonic_diagnostic_lines=allow_nonmonotonic_diagnostic_lines,
|
|
7220
|
+
)
|
|
6641
7221
|
tool_pair_errors: List[JsonObj] = []
|
|
6642
7222
|
tool_pair_partial_result_count = 0
|
|
6643
7223
|
tool_use_occurrences: Dict[str, List[int]] = collections.defaultdict(list)
|
|
6644
7224
|
tool_result_occurrences: Dict[str, List[int]] = collections.defaultdict(list)
|
|
6645
7225
|
malformed_tool_id_samples: List[JsonObj] = []
|
|
6646
7226
|
for line, api_obj in api_messages:
|
|
6647
|
-
|
|
7227
|
+
block_lines = _validation_content_lines(api_obj, line)
|
|
7228
|
+
for block_index, block in enumerate(content_blocks(api_obj)):
|
|
6648
7229
|
if not isinstance(block, dict):
|
|
6649
7230
|
continue
|
|
7231
|
+
block_line = block_lines[block_index]
|
|
6650
7232
|
if block.get("type") == "tool_use":
|
|
6651
7233
|
tool_id = block.get("id")
|
|
6652
7234
|
if not isinstance(tool_id, str) or not tool_id:
|
|
6653
7235
|
malformed_tool_id_samples.append(
|
|
6654
|
-
{"line":
|
|
7236
|
+
{"line": block_line, "kind": "tool_use", "idType": type(tool_id).__name__}
|
|
6655
7237
|
)
|
|
6656
7238
|
else:
|
|
6657
|
-
tool_use_occurrences[tool_id].append(
|
|
7239
|
+
tool_use_occurrences[tool_id].append(block_line)
|
|
6658
7240
|
elif block.get("type") == "tool_result":
|
|
6659
7241
|
tool_id = block.get("tool_use_id") or block.get("toolUseID")
|
|
6660
7242
|
if not isinstance(tool_id, str) or not tool_id:
|
|
6661
7243
|
malformed_tool_id_samples.append(
|
|
6662
|
-
{"line":
|
|
7244
|
+
{"line": block_line, "kind": "tool_result", "idType": type(tool_id).__name__}
|
|
6663
7245
|
)
|
|
6664
7246
|
else:
|
|
6665
|
-
tool_result_occurrences[tool_id].append(
|
|
7247
|
+
tool_result_occurrences[tool_id].append(block_line)
|
|
6666
7248
|
duplicate_tool_use_ids = {
|
|
6667
7249
|
tool_id: lines for tool_id, lines in tool_use_occurrences.items() if len(lines) > 1
|
|
6668
7250
|
}
|
|
@@ -6699,21 +7281,24 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6699
7281
|
for pos, (line, obj) in enumerate(api_messages):
|
|
6700
7282
|
uses = tool_use_ids(obj)
|
|
6701
7283
|
results = tool_result_ids(obj)
|
|
7284
|
+
tool_use_line = _first_validation_block_line(obj, "tool_use", line)
|
|
7285
|
+
tool_result_line = _first_validation_block_line(obj, "tool_result", line)
|
|
6702
7286
|
if uses:
|
|
6703
7287
|
if pos + 1 >= len(api_messages):
|
|
6704
7288
|
tool_pair_errors.append(
|
|
6705
|
-
{"line":
|
|
7289
|
+
{"line": tool_use_line, "uuid": obj.get("uuid"), "reason": "assistant tool_use is final api message", "toolUseIds": uses}
|
|
6706
7290
|
)
|
|
6707
7291
|
else:
|
|
6708
7292
|
next_line, next_obj = api_messages[pos + 1]
|
|
6709
7293
|
next_results = tool_result_ids(next_obj)
|
|
7294
|
+
next_result_line = _first_validation_block_line(next_obj, "tool_result", next_line)
|
|
6710
7295
|
if api_role(next_obj) != "user":
|
|
6711
7296
|
tool_pair_errors.append(
|
|
6712
7297
|
{
|
|
6713
|
-
"line":
|
|
7298
|
+
"line": tool_use_line,
|
|
6714
7299
|
"uuid": obj.get("uuid"),
|
|
6715
7300
|
"reason": "assistant tool_use not followed by user message",
|
|
6716
|
-
"nextLine":
|
|
7301
|
+
"nextLine": next_result_line,
|
|
6717
7302
|
"nextRole": api_role(next_obj),
|
|
6718
7303
|
"toolUseIds": uses,
|
|
6719
7304
|
}
|
|
@@ -6721,11 +7306,11 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6721
7306
|
elif not ordered_subsequence(next_results, uses):
|
|
6722
7307
|
tool_pair_errors.append(
|
|
6723
7308
|
{
|
|
6724
|
-
"line":
|
|
7309
|
+
"line": tool_use_line,
|
|
6725
7310
|
"uuid": obj.get("uuid"),
|
|
6726
7311
|
"reason": "assistant tool_use ids do not contain next user tool_result ids in order",
|
|
6727
7312
|
"toolUseIds": uses,
|
|
6728
|
-
"nextLine":
|
|
7313
|
+
"nextLine": _ordered_subsequence_failure_line(next_obj, uses, next_result_line),
|
|
6729
7314
|
"nextToolResultIds": next_results,
|
|
6730
7315
|
}
|
|
6731
7316
|
)
|
|
@@ -6734,7 +7319,7 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6734
7319
|
if results:
|
|
6735
7320
|
if pos == 0:
|
|
6736
7321
|
tool_pair_errors.append(
|
|
6737
|
-
{"line":
|
|
7322
|
+
{"line": tool_result_line, "uuid": obj.get("uuid"), "reason": "user tool_result is first api message", "toolResultIds": results}
|
|
6738
7323
|
)
|
|
6739
7324
|
else:
|
|
6740
7325
|
source_uuid = source_tool_assistant_uuid(obj)
|
|
@@ -6744,26 +7329,33 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
6744
7329
|
source_uses = tool_use_ids(source_obj) if isinstance(source_obj, dict) else []
|
|
6745
7330
|
prev_line, prev_obj = api_messages[pos - 1]
|
|
6746
7331
|
prev_uses = tool_use_ids(prev_obj)
|
|
7332
|
+
prev_tool_use_line = _first_validation_block_line(prev_obj, "tool_use", prev_line)
|
|
6747
7333
|
matches_prev = api_role(prev_obj) == "assistant" and ordered_subsequence(results, prev_uses)
|
|
6748
7334
|
matches_source = ordered_subsequence(results, source_uses)
|
|
6749
7335
|
if not matches_prev and not matches_source:
|
|
7336
|
+
expected_uses = source_uses or prev_uses
|
|
6750
7337
|
tool_pair_errors.append(
|
|
6751
7338
|
{
|
|
6752
|
-
"line":
|
|
7339
|
+
"line": _ordered_subsequence_failure_line(obj, expected_uses, tool_result_line),
|
|
6753
7340
|
"uuid": obj.get("uuid"),
|
|
6754
7341
|
"reason": "user tool_result ids are not an ordered subset of previous/linked assistant tool_use ids",
|
|
6755
7342
|
"toolResultIds": results,
|
|
6756
|
-
"prevLine":
|
|
7343
|
+
"prevLine": prev_tool_use_line,
|
|
6757
7344
|
"prevToolUseIds": prev_uses,
|
|
6758
7345
|
"sourceToolAssistantUUID": source_uuid,
|
|
6759
7346
|
"sourceToolUseIds": source_uses,
|
|
6760
7347
|
}
|
|
6761
7348
|
)
|
|
6762
7349
|
types = [block.get("type") if isinstance(block, dict) else type(block).__name__ for block in content_blocks(obj)]
|
|
6763
|
-
|
|
7350
|
+
misplaced_index = next(
|
|
7351
|
+
(index for index, kind in enumerate(types[: len(results)]) if kind != "tool_result"),
|
|
7352
|
+
None,
|
|
7353
|
+
)
|
|
7354
|
+
if misplaced_index is not None:
|
|
7355
|
+
content_lines = _validation_content_lines(obj, line)
|
|
6764
7356
|
tool_pair_errors.append(
|
|
6765
7357
|
{
|
|
6766
|
-
"line":
|
|
7358
|
+
"line": content_lines[misplaced_index],
|
|
6767
7359
|
"uuid": obj.get("uuid"),
|
|
6768
7360
|
"reason": "tool_result blocks are not first in user message content",
|
|
6769
7361
|
"blockTypes": types[:10],
|
|
@@ -7029,7 +7621,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
7029
7621
|
)
|
|
7030
7622
|
active_chain_has_compact_summary = any(obj.get("isCompactSummary") is True for obj in active_chain)
|
|
7031
7623
|
active_chain_uuid_set = {obj.get("uuid") for obj in active_chain if isinstance(obj.get("uuid"), str)}
|
|
7032
|
-
active_chain_sessions = collections.Counter(
|
|
7624
|
+
active_chain_sessions = collections.Counter(
|
|
7625
|
+
_diagnostic_counter_label(obj, "sessionId") for obj in active_chain
|
|
7626
|
+
)
|
|
7033
7627
|
compact_metadata_chain_mismatch: List[JsonObj] = []
|
|
7034
7628
|
for boundary in compact_boundaries:
|
|
7035
7629
|
metadata = boundary.get("compactMetadata")
|
|
@@ -7143,9 +7737,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
7143
7737
|
"compactMetadata historical preserved snapshot diverges from the current active chain: "
|
|
7144
7738
|
f"{len(compact_metadata_chain_mismatch)} item(s)"
|
|
7145
7739
|
)
|
|
7146
|
-
type_counts = collections.Counter(obj
|
|
7147
|
-
session_counts = collections.Counter(obj
|
|
7148
|
-
|
|
7740
|
+
type_counts = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in records)
|
|
7741
|
+
session_counts = collections.Counter(_diagnostic_counter_label(obj, "sessionId") for obj in records)
|
|
7742
|
+
validation = {
|
|
7149
7743
|
"ok": not errors,
|
|
7150
7744
|
"errors": errors,
|
|
7151
7745
|
"warnings": warnings,
|
|
@@ -7200,10 +7794,22 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
7200
7794
|
"compact_metadata_chain_mismatch_count": len(compact_metadata_chain_mismatch),
|
|
7201
7795
|
"compact_metadata_chain_mismatch_samples": compact_metadata_chain_mismatch[:20],
|
|
7202
7796
|
"latest_last_prompt_line": latest_last_prompt_line,
|
|
7203
|
-
"latest_last_prompt_session_id":
|
|
7204
|
-
|
|
7797
|
+
"latest_last_prompt_session_id": (
|
|
7798
|
+
_bounded_diagnostic_label(latest_last_prompt_session)
|
|
7799
|
+
if latest_last_prompt_session is not None
|
|
7800
|
+
else None
|
|
7801
|
+
),
|
|
7802
|
+
"latest_last_prompt_leaf_uuid": (
|
|
7803
|
+
_bounded_diagnostic_label(latest_last_prompt_leaf)
|
|
7804
|
+
if latest_last_prompt_leaf is not None
|
|
7805
|
+
else None
|
|
7806
|
+
),
|
|
7205
7807
|
"active_chain_length": len(active_chain),
|
|
7206
|
-
"active_chain_missing_uuid":
|
|
7808
|
+
"active_chain_missing_uuid": (
|
|
7809
|
+
_bounded_diagnostic_label(active_chain_missing_uuid)
|
|
7810
|
+
if active_chain_missing_uuid is not None
|
|
7811
|
+
else None
|
|
7812
|
+
),
|
|
7207
7813
|
"active_chain_loop": active_chain_loop,
|
|
7208
7814
|
"active_chain_min_line": min(active_chain_lines) if active_chain_lines else None,
|
|
7209
7815
|
"active_chain_max_line": max(active_chain_lines) if active_chain_lines else None,
|
|
@@ -7216,6 +7822,10 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
|
|
|
7216
7822
|
"type_counts": dict(type_counts.most_common()),
|
|
7217
7823
|
"session_counts": dict(session_counts.most_common(20)),
|
|
7218
7824
|
}
|
|
7825
|
+
return _bounded_public_diagnostic_structure(
|
|
7826
|
+
validation,
|
|
7827
|
+
redactions=_long_diagnostic_strings(records),
|
|
7828
|
+
)
|
|
7219
7829
|
|
|
7220
7830
|
|
|
7221
7831
|
def validate_jsonl(path: pathlib.Path) -> Dict[str, Any]:
|
|
@@ -7227,8 +7837,8 @@ def validate_jsonl(path: pathlib.Path) -> Dict[str, Any]:
|
|
|
7227
7837
|
|
|
7228
7838
|
|
|
7229
7839
|
def validate_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Dict[str, Any]:
|
|
7230
|
-
records,
|
|
7231
|
-
result = validate_records(records)
|
|
7840
|
+
records, _raw_lines, physical_lines = parse_jsonl_bytes_with_lines(data, source_label=source_label)
|
|
7841
|
+
result = validate_records(records, physical_lines=physical_lines)
|
|
7232
7842
|
result["path"] = pathlib.Path(source_label).name
|
|
7233
7843
|
result["bytes"] = len(data)
|
|
7234
7844
|
result["sha256"] = sha256_hex(data)
|
|
@@ -7247,6 +7857,7 @@ def write_sidecars(output_path: pathlib.Path, report: Dict[str, Any]) -> None:
|
|
|
7247
7857
|
"",
|
|
7248
7858
|
f"- Input: `{report['input']}`",
|
|
7249
7859
|
f"- Output: `{report['output']}`",
|
|
7860
|
+
f"- Operation state: {report.get('operation_state') or 'candidate'}",
|
|
7250
7861
|
f"- Package version: {report.get('package_version')}",
|
|
7251
7862
|
f"- Compression engine: {report.get('codex_offline_compression_version')}",
|
|
7252
7863
|
f"- Model-pack schema: {report.get('model_pack_schema_version')}",
|
|
@@ -7302,6 +7913,7 @@ def write_sidecars(output_path: pathlib.Path, report: Dict[str, Any]) -> None:
|
|
|
7302
7913
|
f"- Replacement target: `{report.get('replacement_target') or ''}`",
|
|
7303
7914
|
f"- Replacement backup: `{report.get('replacement_backup') or ''}`",
|
|
7304
7915
|
f"- Replacement candidate: `{report.get('replacement_candidate') or ''}`",
|
|
7916
|
+
f"- Replacement cleanup errors: {json.dumps(report.get('replacement_cleanup_errors') or [], ensure_ascii=False)}",
|
|
7305
7917
|
"",
|
|
7306
7918
|
"## Validation",
|
|
7307
7919
|
"",
|
|
@@ -7621,6 +8233,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
|
7621
8233
|
report["replacement_candidate_sha256"] = replacement["candidate_sha256"]
|
|
7622
8234
|
report["replacement_published_sha256"] = replacement["published_sha256"]
|
|
7623
8235
|
report["replacement_parent_directory_fsync"] = replacement["parent_directory_fsync"]
|
|
8236
|
+
report["operation_state"] = replacement["operation_state"]
|
|
8237
|
+
report["replacement_cleanup_errors"] = replacement["cleanup_errors"]
|
|
7624
8238
|
try:
|
|
7625
8239
|
write_sidecars(output_path, report)
|
|
7626
8240
|
except Exception as report_exc:
|
|
@@ -7633,6 +8247,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
|
7633
8247
|
"candidate_sha256": replacement.get("candidate_sha256"),
|
|
7634
8248
|
"published_sha256": replacement.get("published_sha256"),
|
|
7635
8249
|
"replacement_validation_ok": bool(replaced_validation.get("ok")),
|
|
8250
|
+
"prior_operation_state": report.get("operation_state"),
|
|
8251
|
+
"replacement_cleanup_errors": report.get("replacement_cleanup_errors", []),
|
|
7636
8252
|
"report_error": f"{type(report_exc).__name__}: {report_exc}",
|
|
7637
8253
|
}
|
|
7638
8254
|
print(json.dumps(receipt, ensure_ascii=False, indent=2))
|
|
@@ -7641,6 +8257,13 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
|
7641
8257
|
f"{report_exc}"
|
|
7642
8258
|
)
|
|
7643
8259
|
return 3
|
|
8260
|
+
if report.get("operation_state") == "committed-cleanup-failed":
|
|
8261
|
+
print(json.dumps(report, ensure_ascii=False, indent=2))
|
|
8262
|
+
eprint(
|
|
8263
|
+
"ERROR: live replacement committed, but transaction cleanup failed; "
|
|
8264
|
+
"inspect replacement_cleanup_errors and remove only the listed residuals after verification."
|
|
8265
|
+
)
|
|
8266
|
+
return 3
|
|
7644
8267
|
print(json.dumps(report, ensure_ascii=False, indent=2))
|
|
7645
8268
|
return 0 if report["validation"].get("ok") else 2
|
|
7646
8269
|
except Exception as exc:
|