@brandry/claude-jsonl-compressor 1.0.0-rc.1 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,13 +26,33 @@ from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
26
26
 
27
27
 
28
28
  JsonObj = Dict[str, Any]
29
- PACKAGE_VERSION = "1.0.0-rc.1"
29
+ PACKAGE_VERSION = "1.0.0"
30
30
  CODEX_OFFLINE_COMPRESSION_VERSION = "v10"
31
31
  MODEL_PACK_SCHEMA_VERSION = 11
32
32
  REPORT_SCHEMA_VERSION = 1
33
33
  PRIOR_SUMMARY_VERBATIM_BUDGET_FACTOR = 1.5
34
34
  MIN_SUMMARY_CHAR_BUDGET = 4000
35
35
  DEFAULT_MODEL_PACK_ESTIMATED_TOKEN_BUDGET = 150000
36
+ MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS = 160
37
+ RESUME_REASON_CODE_STATUS = {
38
+ "last-prompt-absent": "absent",
39
+ "duplicate-uuid": "duplicate-uuid",
40
+ "leaf-uuid-malformed": "malformed",
41
+ "chain-missing-uuid": "dangling",
42
+ "chain-loop": "loop",
43
+ "chain-malformed-parent": "malformed-parent",
44
+ "chain-empty": "dangling",
45
+ "chain-non-monotonic": "non-monotonic",
46
+ "lineage-unsafe": "session-mismatch",
47
+ "extension-limit-exceeded": "extension-limit",
48
+ "extension-authority-session-missing": "session-mismatch",
49
+ "extension-record-missing-uuid": "extension-unsafe",
50
+ "extension-record-not-linear-descendant": "extension-branch",
51
+ "extension-record-session-mismatch": "session-mismatch",
52
+ "extension-record-not-safe-closure": "extension-unsafe",
53
+ "extension-pending-tool-ids": "extension-unsafe",
54
+ "ok": "valid",
55
+ }
36
56
 
37
57
 
38
58
  def configure_stdio() -> None:
@@ -47,6 +67,7 @@ def configure_stdio() -> None:
47
67
 
48
68
  configure_stdio()
49
69
 
70
+
50
71
  DEFAULT_IMPORTANCE_WORDS = tuple(
51
72
  [
52
73
  "must",
@@ -1825,9 +1846,13 @@ def json_dump_line(obj: JsonObj) -> str:
1825
1846
  return json.dumps(obj, ensure_ascii=False, separators=(",", ":"))
1826
1847
 
1827
1848
 
1828
- def parse_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Tuple[List[JsonObj], List[str]]:
1849
+ def parse_jsonl_bytes_with_lines(
1850
+ data: bytes,
1851
+ source_label: str = "JSONL",
1852
+ ) -> Tuple[List[JsonObj], List[str], List[int]]:
1829
1853
  records: List[JsonObj] = []
1830
1854
  raw_lines: List[str] = []
1855
+ physical_lines: List[int] = []
1831
1856
  for line_no, physical_line in enumerate(data.split(b"\n"), 1):
1832
1857
  line_bytes = physical_line[:-1] if physical_line.endswith(b"\r") else physical_line
1833
1858
  if line_no == 1 and line_bytes.startswith(b"\xef\xbb\xbf"):
@@ -1846,6 +1871,12 @@ def parse_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Tuple[List[Js
1846
1871
  raise ValueError(f"line {line_no} is JSON but not an object")
1847
1872
  records.append(obj)
1848
1873
  raw_lines.append(line)
1874
+ physical_lines.append(line_no)
1875
+ return records, raw_lines, physical_lines
1876
+
1877
+
1878
+ def parse_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Tuple[List[JsonObj], List[str]]:
1879
+ records, raw_lines, _physical_lines = parse_jsonl_bytes_with_lines(data, source_label=source_label)
1849
1880
  return records, raw_lines
1850
1881
 
1851
1882
 
@@ -1956,6 +1987,49 @@ def numbered_backup_path(path: pathlib.Path, backup_dir: Optional[pathlib.Path]
1956
1987
  raise RuntimeError(f"could not find free backup name for {path}")
1957
1988
 
1958
1989
 
1990
+ def _filesystem_error_detail(error: Exception) -> str:
1991
+ detail = f"{type(error).__name__}: {error}"
1992
+ errno_value = getattr(error, "errno", None)
1993
+ winerror_value = getattr(error, "winerror", None)
1994
+ if errno_value is not None:
1995
+ detail += f"; errno={errno_value}"
1996
+ if winerror_value is not None:
1997
+ detail += f"; winerror={winerror_value}"
1998
+ return detail
1999
+
2000
+
2001
+ def _cleanup_owned_file(
2002
+ path: pathlib.Path,
2003
+ *,
2004
+ identity: Optional[os.stat_result],
2005
+ expected_bytes: Optional[bytes],
2006
+ owner_label: str,
2007
+ missing_is_error: bool = False,
2008
+ ) -> Optional[str]:
2009
+ """Remove a unique temporary path only while its observed identity remains ours."""
2010
+ if identity is None:
2011
+ try:
2012
+ path.lstat()
2013
+ except FileNotFoundError:
2014
+ return None
2015
+ except OSError as exc:
2016
+ return f"{path.name}: {_filesystem_error_detail(exc)}"
2017
+ return f"{path.name}: {owner_label} identity was not captured"
2018
+ try:
2019
+ if not os.path.samestat(identity, path.lstat()):
2020
+ return f"{path.name}: {owner_label} identity changed before cleanup"
2021
+ if expected_bytes is not None and path.read_bytes() != expected_bytes:
2022
+ return f"{path.name}: {owner_label} bytes changed before cleanup"
2023
+ if not os.path.samestat(identity, path.lstat()):
2024
+ return f"{path.name}: {owner_label} identity changed before cleanup"
2025
+ path.unlink()
2026
+ except FileNotFoundError:
2027
+ return f"{path.name}: disappeared before cleanup" if missing_is_error else None
2028
+ except OSError as exc:
2029
+ return f"{path.name}: {_filesystem_error_detail(exc)}"
2030
+ return None
2031
+
2032
+
1959
2033
  def _exclusive_backup_from_bytes(
1960
2034
  path: pathlib.Path,
1961
2035
  source_bytes: bytes,
@@ -1970,26 +2044,42 @@ def _exclusive_backup_from_bytes(
1970
2044
  for backup in candidates:
1971
2045
  backup.parent.mkdir(parents=True, exist_ok=True)
1972
2046
  stage = backup.with_name(f".{backup.name}.stage-{uuid.uuid4().hex}.tmp")
2047
+ stage_identity: Optional[os.stat_result] = None
2048
+ stage_expected_bytes: Optional[bytes] = None
2049
+ verified_backup: Optional[pathlib.Path] = None
1973
2050
  try:
1974
2051
  with stage.open("xb") as f:
2052
+ stage_identity = stage.lstat()
1975
2053
  f.write(source_bytes)
1976
2054
  f.flush()
1977
2055
  os.fsync(f.fileno())
1978
2056
  if stage.read_bytes() != source_bytes:
1979
2057
  raise RuntimeError(f"staged backup verification failed: {backup.name}")
2058
+ stage_expected_bytes = source_bytes
1980
2059
  try:
1981
2060
  _publish_no_clobber(stage, backup)
1982
2061
  except FileExistsError:
1983
2062
  continue
1984
2063
  if backup.read_bytes() == source_bytes:
1985
2064
  fsync_parent_directory(backup)
2065
+ verified_backup = backup
1986
2066
  return backup
1987
2067
  # A concurrently replaced numbered path is not ours to delete.
1988
2068
  finally:
1989
- try:
1990
- stage.unlink()
1991
- except FileNotFoundError:
1992
- pass
2069
+ active_error = sys.exc_info()[1]
2070
+ cleanup_error = _cleanup_owned_file(
2071
+ stage,
2072
+ identity=stage_identity,
2073
+ expected_bytes=stage_expected_bytes,
2074
+ owner_label="backup stage",
2075
+ )
2076
+ if cleanup_error is not None:
2077
+ message = f"backup staging cleanup failed: {cleanup_error}"
2078
+ if verified_backup is not None:
2079
+ message += f"; verified backup retained as {verified_backup.name}"
2080
+ if active_error is not None:
2081
+ raise RuntimeError(f"{active_error}; {message}") from active_error
2082
+ raise RuntimeError(message)
1993
2083
  raise RuntimeError(f"could not find free backup name for {path}")
1994
2084
 
1995
2085
 
@@ -1999,6 +2089,107 @@ def create_backup(path: pathlib.Path, backup_dir: Optional[pathlib.Path] = None)
1999
2089
  return _exclusive_backup_from_bytes(path, path.read_bytes(), backup_dir=backup_dir)
2000
2090
 
2001
2091
 
2092
+ def _preflight_hardlink_support(directory: pathlib.Path, label: str) -> None:
2093
+ """Verify hard links in one publication directory before live replacement writes."""
2094
+ token = uuid.uuid4().hex
2095
+ probe_source = directory / f".cjc-hardlink-probe-{token}.source.tmp"
2096
+ probe_link = directory / f".cjc-hardlink-probe-{token}.link.tmp"
2097
+ marker = f"claude-jsonl-compressor hard-link probe {token}\n".encode("ascii")
2098
+ source_created = False
2099
+ link_created = False
2100
+ source_identity: Optional[os.stat_result] = None
2101
+ link_identity: Optional[os.stat_result] = None
2102
+ operation_error: Optional[Tuple[str, Exception]] = None
2103
+ cleanup_errors: List[str] = []
2104
+
2105
+ try:
2106
+ phase = "probe source creation"
2107
+ with probe_source.open("xb") as stream:
2108
+ source_created = True
2109
+ stream.write(marker)
2110
+ stream.flush()
2111
+ os.fsync(stream.fileno())
2112
+ source_identity = probe_source.lstat()
2113
+
2114
+ phase = "hard-link creation"
2115
+ os.link(probe_source, probe_link)
2116
+ link_created = True
2117
+ link_identity = probe_link.lstat()
2118
+
2119
+ phase = "hard-link verification"
2120
+ if source_identity is None or link_identity is None:
2121
+ raise RuntimeError("probe identity was not captured")
2122
+ if not os.path.samestat(source_identity, link_identity):
2123
+ raise RuntimeError("probe destination does not reference the probe source")
2124
+ if not os.path.samefile(probe_source, probe_link):
2125
+ raise RuntimeError("probe destination does not reference the probe source")
2126
+ if probe_link.read_bytes() != marker:
2127
+ raise RuntimeError("probe hard-link bytes differ from the probe source")
2128
+ except Exception as exc:
2129
+ operation_error = (phase, exc)
2130
+ finally:
2131
+ if link_created:
2132
+ cleanup_error = _cleanup_owned_file(
2133
+ probe_link,
2134
+ identity=link_identity,
2135
+ expected_bytes=marker,
2136
+ owner_label="probe",
2137
+ missing_is_error=True,
2138
+ )
2139
+ if cleanup_error is not None:
2140
+ cleanup_errors.append(cleanup_error)
2141
+ else:
2142
+ try:
2143
+ probe_link.lstat()
2144
+ except FileNotFoundError:
2145
+ pass
2146
+ except OSError as exc:
2147
+ cleanup_errors.append(
2148
+ f"{probe_link.name}: could not inspect destination after link failure; "
2149
+ f"{_filesystem_error_detail(exc)}"
2150
+ )
2151
+ else:
2152
+ cleanup_errors.append(f"{probe_link.name}: destination was claimed after link failure")
2153
+
2154
+ if source_created:
2155
+ cleanup_error = _cleanup_owned_file(
2156
+ probe_source,
2157
+ identity=source_identity,
2158
+ expected_bytes=marker,
2159
+ owner_label="probe",
2160
+ missing_is_error=True,
2161
+ )
2162
+ if cleanup_error is not None:
2163
+ cleanup_errors.append(cleanup_error)
2164
+
2165
+ # Directory durability is best effort everywhere else in this module.
2166
+ fsync_parent_directory(probe_source)
2167
+
2168
+ if cleanup_errors:
2169
+ operation_detail = ""
2170
+ if operation_error is not None:
2171
+ operation_detail = (
2172
+ f"; operation_error during {operation_error[0]}={_filesystem_error_detail(operation_error[1])}"
2173
+ )
2174
+ raise RuntimeError(
2175
+ f"{label} hard-link preflight cleanup failed before live replacement; "
2176
+ "the target file was not moved or replaced; "
2177
+ f"retained probe detail={'; '.join(cleanup_errors)}{operation_detail}"
2178
+ ) from (operation_error[1] if operation_error is not None else None)
2179
+
2180
+ if operation_error is not None:
2181
+ phase, error = operation_error
2182
+ raise RuntimeError(
2183
+ f"{label} hard-link preflight failed before live replacement; "
2184
+ "the target file was not moved or replaced; "
2185
+ f"phase={phase}; cause={_filesystem_error_detail(error)}"
2186
+ ) from error
2187
+
2188
+
2189
+ def _preflight_target_volume_hardlink_support(target_path: pathlib.Path) -> None:
2190
+ _preflight_hardlink_support(target_path.parent, "target-volume")
2191
+
2192
+
2002
2193
  def _publish_no_clobber(source_path: pathlib.Path, destination_path: pathlib.Path) -> None:
2003
2194
  """Atomically create destination without replacing any concurrent claimant."""
2004
2195
  if destination_path.exists():
@@ -2009,7 +2200,8 @@ def _publish_no_clobber(source_path: pathlib.Path, destination_path: pathlib.Pat
2009
2200
  raise
2010
2201
  except OSError as exc:
2011
2202
  raise RuntimeError(
2012
- "atomic no-clobber publication requires same-volume hard-link support"
2203
+ "atomic no-clobber publication requires same-volume hard-link support; "
2204
+ f"os.link={_filesystem_error_detail(exc)}"
2013
2205
  ) from exc
2014
2206
  fsync_parent_directory(destination_path)
2015
2207
 
@@ -2039,6 +2231,7 @@ def _restore_capture_no_clobber(
2039
2231
  *,
2040
2232
  validate_jsonl: bool,
2041
2233
  ) -> None:
2234
+ capture_identity = capture_path.lstat()
2042
2235
  _publish_no_clobber(capture_path, target_path)
2043
2236
  restored_bytes = target_path.read_bytes()
2044
2237
  if restored_bytes != expected_bytes:
@@ -2047,11 +2240,14 @@ def _restore_capture_no_clobber(
2047
2240
  validation = validate_jsonl_bytes(restored_bytes, source_label=target_path.name)
2048
2241
  if not validation.get("ok"):
2049
2242
  raise RuntimeError(f"restored source validation failed: {validation.get('errors')}")
2050
- try:
2051
- if os.path.samefile(capture_path, target_path):
2052
- capture_path.unlink()
2053
- except (FileNotFoundError, OSError):
2054
- pass
2243
+ cleanup_error = _cleanup_owned_file(
2244
+ capture_path,
2245
+ identity=capture_identity,
2246
+ expected_bytes=expected_bytes,
2247
+ owner_label="restored capture",
2248
+ )
2249
+ if cleanup_error is not None:
2250
+ raise RuntimeError(f"restored capture cleanup failed: {cleanup_error}")
2055
2251
  fsync_parent_directory(target_path)
2056
2252
 
2057
2253
 
@@ -2075,22 +2271,34 @@ def _replace_file_after_validation(
2075
2271
  source_sha256 = sha256_hex(source_bytes)
2076
2272
  if expected_source_sha256 is not None and source_sha256 != expected_source_sha256:
2077
2273
  raise RuntimeError("input JSONL changed after candidate generation; original file was not replaced")
2274
+ _preflight_target_volume_hardlink_support(target_path)
2275
+ if backup_dir is not None:
2276
+ backup_dir.mkdir(parents=True, exist_ok=True)
2277
+ _preflight_hardlink_support(backup_dir, "backup-directory")
2078
2278
  tmp_replace = target_path.with_name(f".{target_path.name}.replace-{uuid.uuid4().hex}.tmp")
2079
2279
  old_capture = target_path.with_name(f".{target_path.name}.old-{uuid.uuid4().hex}.tmp")
2280
+ tmp_replace_identity: Optional[os.stat_result] = None
2281
+ tmp_replace_expected_bytes: Optional[bytes] = None
2282
+ old_capture_identity: Optional[os.stat_result] = None
2283
+ rollback_capture_identity: Optional[os.stat_result] = None
2080
2284
  backup: Optional[pathlib.Path] = None
2081
2285
  published = False
2082
2286
  directory_fsync = False
2083
2287
  retain_old_capture = False
2084
2288
  external_target_preserved = False
2085
2289
  retained_rollback_capture: Optional[pathlib.Path] = None
2290
+ rollback_capture: Optional[pathlib.Path] = None
2291
+ cleanup_errors: List[Dict[str, str]] = []
2086
2292
  replaced_validation: Dict[str, Any] = {}
2087
2293
  try:
2088
2294
  with tmp_replace.open("xb") as f:
2295
+ tmp_replace_identity = tmp_replace.lstat()
2089
2296
  f.write(candidate_bytes)
2090
2297
  f.flush()
2091
2298
  os.fsync(f.fileno())
2092
2299
  if sha256_hex(tmp_replace.read_bytes()) != candidate_sha256:
2093
2300
  raise RuntimeError("staged replacement bytes do not match the validated candidate snapshot")
2301
+ tmp_replace_expected_bytes = candidate_bytes
2094
2302
  if target_path.read_bytes() != source_bytes:
2095
2303
  raise RuntimeError("input JSONL changed before backup; original file was not replaced")
2096
2304
  backup = _exclusive_backup_from_bytes(target_path, source_bytes, backup_dir=backup_dir)
@@ -2101,15 +2309,11 @@ def _replace_file_after_validation(
2101
2309
  captured_bytes = old_capture.read_bytes()
2102
2310
  if captured_bytes != source_bytes:
2103
2311
  raise RuntimeError("input JSONL changed during replacement capture; candidate was not installed")
2312
+ old_capture_identity = old_capture.lstat()
2104
2313
  if target_path.exists():
2105
2314
  raise RuntimeError("input JSONL was recreated concurrently; external bytes were left in place")
2106
2315
  _publish_no_clobber(tmp_replace, target_path)
2107
2316
  published = True
2108
- try:
2109
- if os.path.samefile(tmp_replace, target_path):
2110
- tmp_replace.unlink()
2111
- except (FileNotFoundError, OSError):
2112
- pass
2113
2317
  directory_fsync = fsync_parent_directory(target_path) or directory_fsync
2114
2318
  published_bytes = target_path.read_bytes()
2115
2319
  if sha256_hex(published_bytes) != candidate_sha256:
@@ -2120,8 +2324,6 @@ def _replace_file_after_validation(
2120
2324
  backup = _ensure_verified_source_backup(target_path, source_bytes, backup, backup_dir)
2121
2325
  if old_capture.read_bytes() != source_bytes:
2122
2326
  raise RuntimeError("retained source capture changed before successful cleanup")
2123
- old_capture.unlink()
2124
- directory_fsync = fsync_parent_directory(target_path) or directory_fsync
2125
2327
  except Exception as exc:
2126
2328
  rollback_error: Optional[Exception] = None
2127
2329
  restored = False
@@ -2147,6 +2349,7 @@ def _replace_file_after_validation(
2147
2349
  except FileNotFoundError:
2148
2350
  rollback_capture = None
2149
2351
  if rollback_capture is not None:
2352
+ rollback_capture_identity = rollback_capture.lstat()
2150
2353
  actual_target_bytes = rollback_capture.read_bytes()
2151
2354
  if actual_target_bytes == candidate_bytes:
2152
2355
  try:
@@ -2157,11 +2360,12 @@ def _replace_file_after_validation(
2157
2360
  validate_jsonl=True,
2158
2361
  )
2159
2362
  restored = True
2160
- if rollback_capture.read_bytes() == candidate_bytes:
2161
- rollback_capture.unlink()
2162
2363
  except FileExistsError:
2163
2364
  external_target_preserved = True
2164
2365
  retained_rollback_capture = rollback_capture
2366
+ except Exception:
2367
+ retained_rollback_capture = rollback_capture
2368
+ raise
2165
2369
  else:
2166
2370
  external_target_preserved = True
2167
2371
  try:
@@ -2173,16 +2377,15 @@ def _replace_file_after_validation(
2173
2377
  )
2174
2378
  except FileExistsError:
2175
2379
  retained_rollback_capture = rollback_capture
2380
+ except Exception:
2381
+ retained_rollback_capture = rollback_capture
2382
+ raise
2176
2383
  if backup is None or backup.read_bytes() != source_bytes:
2177
2384
  raise RuntimeError("verified source backup is unavailable during concurrent-target recovery")
2178
- if old_capture.read_bytes() == source_bytes:
2179
- old_capture.unlink()
2180
2385
  elif target_path.exists():
2181
2386
  external_target_preserved = True
2182
2387
  if backup is None or backup.read_bytes() != source_bytes:
2183
2388
  raise RuntimeError("verified source backup is unavailable during concurrent-target recovery")
2184
- if old_capture.read_bytes() == source_bytes:
2185
- old_capture.unlink()
2186
2389
  else:
2187
2390
  try:
2188
2391
  _restore_capture_no_clobber(
@@ -2217,14 +2420,37 @@ def _replace_file_after_validation(
2217
2420
  raise RuntimeError(f"replacement failed and original bytes were restored: {exc}") from exc
2218
2421
  raise
2219
2422
  finally:
2220
- transients = [tmp_replace]
2423
+ active_error = sys.exc_info()[1]
2424
+ transients = [
2425
+ (tmp_replace, tmp_replace_identity, tmp_replace_expected_bytes, "replacement stage"),
2426
+ ]
2221
2427
  if not retain_old_capture:
2222
- transients.append(old_capture)
2223
- for transient in transients:
2224
- try:
2225
- transient.unlink()
2226
- except FileNotFoundError:
2227
- pass
2428
+ transients.append((old_capture, old_capture_identity, source_bytes, "source capture"))
2429
+ if rollback_capture is not None and retained_rollback_capture is None:
2430
+ transients.append((rollback_capture, rollback_capture_identity, candidate_bytes, "rollback capture"))
2431
+ removed_transient = False
2432
+ for transient, identity, expected_bytes, owner_label in transients:
2433
+ cleanup_error = _cleanup_owned_file(
2434
+ transient,
2435
+ identity=identity,
2436
+ expected_bytes=expected_bytes,
2437
+ owner_label=owner_label,
2438
+ )
2439
+ if cleanup_error is None:
2440
+ if identity is not None:
2441
+ removed_transient = True
2442
+ else:
2443
+ cleanup_errors.append(
2444
+ {
2445
+ "path": transient.name,
2446
+ "error": cleanup_error,
2447
+ }
2448
+ )
2449
+ if removed_transient:
2450
+ directory_fsync = fsync_parent_directory(target_path) or directory_fsync
2451
+ if cleanup_errors and active_error is not None:
2452
+ cleanup_detail = json.dumps(cleanup_errors, ensure_ascii=False, separators=(",", ":"))
2453
+ raise RuntimeError(f"{active_error}; replacement cleanup errors={cleanup_detail}") from active_error
2228
2454
  if backup is None:
2229
2455
  raise AssertionError("replacement completed without an auditable backup")
2230
2456
  return {
@@ -2234,6 +2460,8 @@ def _replace_file_after_validation(
2234
2460
  "candidate_sha256": candidate_sha256,
2235
2461
  "published_sha256": candidate_sha256,
2236
2462
  "parent_directory_fsync": directory_fsync,
2463
+ "operation_state": "committed-cleanup-failed" if cleanup_errors else "committed",
2464
+ "cleanup_errors": cleanup_errors,
2237
2465
  }
2238
2466
 
2239
2467
 
@@ -2361,6 +2589,109 @@ def stable_digest(text: str) -> str:
2361
2589
  return hashlib.sha256(text.encode("utf-8", errors="replace")).hexdigest()[:16]
2362
2590
 
2363
2591
 
2592
+ def _bounded_diagnostic_label(value: Any, *, missing: bool = False) -> str:
2593
+ """Render schema metadata for reports without copying arbitrary JSON values."""
2594
+ if missing:
2595
+ return "<missing>"
2596
+ if isinstance(value, str):
2597
+ if len(value) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS:
2598
+ return f"<string:length={len(value)};sha256={stable_digest(value)}>"
2599
+ return f"<{value}" if value.startswith("<") else value
2600
+ if value is None:
2601
+ return "<null>"
2602
+ return f"<invalid:{type(value).__name__}>"
2603
+
2604
+
2605
+ def _long_diagnostic_strings(value: Any) -> List[str]:
2606
+ """Collect source strings that must not be copied into public diagnostics."""
2607
+ found: set = set()
2608
+
2609
+ def visit(item: Any) -> None:
2610
+ if isinstance(item, str):
2611
+ if len(item) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS:
2612
+ found.add(item)
2613
+ elif isinstance(item, dict):
2614
+ for key, nested in item.items():
2615
+ visit(key)
2616
+ visit(nested)
2617
+ elif isinstance(item, (list, tuple)):
2618
+ for nested in item:
2619
+ visit(nested)
2620
+
2621
+ visit(value)
2622
+ return sorted(found, key=len, reverse=True)
2623
+
2624
+
2625
+ def _redact_public_diagnostic_text(text: str, redactions: Sequence[str]) -> Tuple[str, bool]:
2626
+ """Replace exact/repr/JSON renderings of oversized source values in text."""
2627
+ redacted = text
2628
+ replaced = False
2629
+ for raw in redactions:
2630
+ label = _bounded_diagnostic_label(raw)
2631
+ for source, replacement in (
2632
+ (raw, label),
2633
+ (repr(raw), repr(label)),
2634
+ (json.dumps(raw, ensure_ascii=False), json.dumps(label, ensure_ascii=False)),
2635
+ ):
2636
+ if source in redacted:
2637
+ redacted = redacted.replace(source, replacement)
2638
+ replaced = True
2639
+ return redacted, replaced
2640
+
2641
+
2642
+ def _bounded_public_diagnostic_structure(
2643
+ value: Any,
2644
+ *,
2645
+ redactions: Sequence[str] = (),
2646
+ diagnostic_text: bool = False,
2647
+ redact_unmatched_diagnostic_text: bool = False,
2648
+ ) -> Any:
2649
+ """Bound oversized fields while retaining ordinary human-readable diagnostics."""
2650
+ if isinstance(value, str):
2651
+ if diagnostic_text:
2652
+ redacted, replaced = _redact_public_diagnostic_text(value, redactions)
2653
+ if replaced or not redact_unmatched_diagnostic_text:
2654
+ return redacted
2655
+ return _bounded_diagnostic_label(value) if len(value) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS else value
2656
+ return _bounded_diagnostic_label(value) if len(value) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS else value
2657
+ if isinstance(value, list):
2658
+ return [
2659
+ _bounded_public_diagnostic_structure(
2660
+ item,
2661
+ redactions=redactions,
2662
+ diagnostic_text=diagnostic_text,
2663
+ redact_unmatched_diagnostic_text=redact_unmatched_diagnostic_text,
2664
+ )
2665
+ for item in value
2666
+ ]
2667
+ if isinstance(value, tuple):
2668
+ return tuple(
2669
+ _bounded_public_diagnostic_structure(
2670
+ item,
2671
+ redactions=redactions,
2672
+ diagnostic_text=diagnostic_text,
2673
+ redact_unmatched_diagnostic_text=redact_unmatched_diagnostic_text,
2674
+ )
2675
+ for item in value
2676
+ )
2677
+ if isinstance(value, dict):
2678
+ projected: Dict[Any, Any] = {}
2679
+ for key, item in value.items():
2680
+ public_key = (
2681
+ _bounded_diagnostic_label(key)
2682
+ if isinstance(key, str) and len(key) > MAX_PUBLIC_DIAGNOSTIC_STRING_CHARS
2683
+ else key
2684
+ )
2685
+ projected[public_key] = _bounded_public_diagnostic_structure(
2686
+ item,
2687
+ redactions=redactions,
2688
+ diagnostic_text=diagnostic_text or key in {"errors", "warnings"},
2689
+ redact_unmatched_diagnostic_text=redact_unmatched_diagnostic_text,
2690
+ )
2691
+ return projected
2692
+ return value
2693
+
2694
+
2364
2695
  def sha256_hex(data: bytes) -> str:
2365
2696
  return hashlib.sha256(data).hexdigest()
2366
2697
 
@@ -2438,6 +2769,12 @@ def one_line(text: str, limit: int = 220) -> str:
2438
2769
  return truncate(text, limit).replace("\n", " ")
2439
2770
 
2440
2771
 
2772
+ def _bounded_diagnostic_text(value: Any, limit: int = 1200) -> str:
2773
+ if isinstance(value, str):
2774
+ return one_line(truncate(value, limit), limit)
2775
+ return _bounded_diagnostic_label(value)
2776
+
2777
+
2441
2778
  def block_text(block: Any) -> str:
2442
2779
  if isinstance(block, str):
2443
2780
  return block
@@ -2554,7 +2891,14 @@ def record_text(obj: JsonObj) -> str:
2554
2891
  if t == "attachment":
2555
2892
  return attachment_text(obj)
2556
2893
  if t == "system":
2557
- return " ".join(str(obj.get(k, "")) for k in ("subtype", "content", "error") if obj.get(k))
2894
+ pieces: List[str] = []
2895
+ if "subtype" in obj and obj.get("subtype") not in (None, ""):
2896
+ pieces.append(_bounded_diagnostic_label(obj.get("subtype")))
2897
+ for key in ("content", "error"):
2898
+ value = obj.get(key)
2899
+ if value not in (None, ""):
2900
+ pieces.append(_bounded_diagnostic_text(value))
2901
+ return " ".join(pieces)
2558
2902
  if t == "file-history-snapshot":
2559
2903
  snap = obj.get("snapshot")
2560
2904
  if isinstance(snap, dict):
@@ -2694,7 +3038,12 @@ def ordered_subsequence(items: Sequence[str], expected: Sequence[str]) -> bool:
2694
3038
 
2695
3039
 
2696
3040
  def is_api_message(obj: JsonObj) -> bool:
2697
- return obj.get("type") in {"user", "assistant"} and isinstance(obj.get("message"), dict)
3041
+ record_type = obj.get("type")
3042
+ return (
3043
+ isinstance(record_type, str)
3044
+ and record_type in {"user", "assistant"}
3045
+ and isinstance(obj.get("message"), dict)
3046
+ )
2698
3047
 
2699
3048
 
2700
3049
  def api_role(obj: JsonObj) -> Optional[str]:
@@ -2913,6 +3262,16 @@ def _allowed_post_prompt_closure(obj: JsonObj, pending_tool_ids: set) -> Tuple[b
2913
3262
  return False, f"unsupported_record_type:{obj.get('type')}"
2914
3263
 
2915
3264
 
3265
+ def _reject_resume_path(info: Dict[str, Any], reason_code: str, error: str) -> Dict[str, Any]:
3266
+ status = RESUME_REASON_CODE_STATUS.get(reason_code)
3267
+ if status is None or status == "valid":
3268
+ raise AssertionError(f"invalid resume-path rejection code: {reason_code}")
3269
+ info["status"] = status
3270
+ info["reasonCode"] = reason_code
3271
+ info["errors"].append(error)
3272
+ return info
3273
+
3274
+
2916
3275
  def choose_resume_leaf_info(
2917
3276
  records: Sequence[JsonObj],
2918
3277
  max_post_prompt_extension: int = 0,
@@ -2928,7 +3287,8 @@ def choose_resume_leaf_info(
2928
3287
  )
2929
3288
  info: Dict[str, Any] = {
2930
3289
  "ok": False,
2931
- "status": "absent" if last_prompt_entry is None else "unvalidated",
3290
+ "status": "unvalidated",
3291
+ "reasonCode": None,
2932
3292
  "errors": [],
2933
3293
  "warnings": [],
2934
3294
  "lastPromptIndex": last_prompt_entry[0] if last_prompt_entry else None,
@@ -2956,6 +3316,7 @@ def choose_resume_leaf_info(
2956
3316
  "sessionLineageRunCount": 0,
2957
3317
  "sessionLineageTransitionCount": 0,
2958
3318
  "sessionLineageRunsDigest": None,
3319
+ "lineageReason": None,
2959
3320
  "currentSessionStartPosition": 0,
2960
3321
  "currentSessionRecordCount": 0,
2961
3322
  "currentSessionId": None,
@@ -2964,19 +3325,26 @@ def choose_resume_leaf_info(
2964
3325
  "authoritySessionId": last_prompt_entry[1].get("sessionId") if last_prompt_entry else None,
2965
3326
  }
2966
3327
  if duplicate_uuids:
2967
- info["status"] = "duplicate-uuid"
2968
- info["errors"].append(f"duplicate uuid values make resume topology ambiguous: {duplicate_uuids[:20]}")
2969
- return info
3328
+ return _reject_resume_path(
3329
+ info,
3330
+ "duplicate-uuid",
3331
+ f"duplicate uuid values make resume topology ambiguous: {duplicate_uuids[:20]}",
3332
+ )
2970
3333
  if last_prompt_entry is None:
2971
- info["errors"].append("strict active-chain mode requires a last-prompt record")
2972
- return info
3334
+ return _reject_resume_path(
3335
+ info,
3336
+ "last-prompt-absent",
3337
+ "strict active-chain mode requires a last-prompt record",
3338
+ )
2973
3339
 
2974
3340
  prompt_index, prompt_record = last_prompt_entry
2975
3341
  prompt_leaf = resume_leaf_override or prompt_record.get("leafUuid")
2976
3342
  if not isinstance(prompt_leaf, str) or not prompt_leaf:
2977
- info["status"] = "malformed"
2978
- info["errors"].append("authoritative last-prompt leafUuid is missing or malformed")
2979
- return info
3343
+ return _reject_resume_path(
3344
+ info,
3345
+ "leaf-uuid-malformed",
3346
+ "authoritative last-prompt leafUuid is missing or malformed",
3347
+ )
2980
3348
  info["selectedLeafUuid"] = prompt_leaf
2981
3349
  trace = chain_trace_from_leaf(records, prompt_leaf)
2982
3350
  info["promptChainMissingUuid"] = trace.get("missingUuid")
@@ -2984,25 +3352,27 @@ def choose_resume_leaf_info(
2984
3352
  info["promptChainMalformedParentUuid"] = trace.get("malformedParentUuid")
2985
3353
  info["promptChainMalformedParentType"] = trace.get("malformedParentType")
2986
3354
  if trace.get("missingUuid"):
2987
- info["status"] = "dangling"
2988
- info["errors"].append(f"authoritative resume chain references missing uuid: {trace.get('missingUuid')}")
2989
- return info
3355
+ return _reject_resume_path(
3356
+ info,
3357
+ "chain-missing-uuid",
3358
+ f"authoritative resume chain references missing uuid: {trace.get('missingUuid')}",
3359
+ )
2990
3360
  if trace.get("loopUuid"):
2991
- info["status"] = "loop"
2992
- info["errors"].append(f"authoritative resume chain contains a loop at uuid: {trace.get('loopUuid')}")
2993
- return info
3361
+ return _reject_resume_path(
3362
+ info,
3363
+ "chain-loop",
3364
+ f"authoritative resume chain contains a loop at uuid: {trace.get('loopUuid')}",
3365
+ )
2994
3366
  if trace.get("malformedParentUuid"):
2995
- info["status"] = "malformed-parent"
2996
- info["errors"].append(
3367
+ return _reject_resume_path(
3368
+ info,
3369
+ "chain-malformed-parent",
2997
3370
  "authoritative resume chain contains a non-null, non-empty-string parentUuid "
2998
- f"on uuid {trace.get('malformedParentUuid')} (type {trace.get('malformedParentType')})"
3371
+ f"on uuid {trace.get('malformedParentUuid')} (type {trace.get('malformedParentType')})",
2999
3372
  )
3000
- return info
3001
3373
  chain = list(trace.get("chain") or [])
3002
3374
  if not chain:
3003
- info["status"] = "dangling"
3004
- info["errors"].append("authoritative resume chain is empty")
3005
- return info
3375
+ return _reject_resume_path(info, "chain-empty", "authoritative resume chain is empty")
3006
3376
  uuid_to_index = {
3007
3377
  obj.get("uuid"): idx for idx, obj in enumerate(records)
3008
3378
  if isinstance(obj.get("uuid"), str) and obj.get("uuid")
@@ -3012,12 +3382,12 @@ def choose_resume_leaf_info(
3012
3382
  info["nonMonotonicEdgeCount"] = order_info["inversionCount"]
3013
3383
  info["nonMonotonicCompatibilityEdgeCount"] = order_info["compatibilityEdgeCount"]
3014
3384
  if not order_info["ok"]:
3015
- info["status"] = "non-monotonic"
3016
- info["errors"].append(
3385
+ return _reject_resume_path(
3386
+ info,
3387
+ "chain-non-monotonic",
3017
3388
  "authoritative resume chain has non-monotonic physical parent edges outside the "
3018
- "same-session attachment compatibility rule"
3389
+ "same-session attachment compatibility rule",
3019
3390
  )
3020
- return info
3021
3391
  if order_info["compatibilityEdgeCount"]:
3022
3392
  info["warnings"].append(
3023
3393
  f"accepted {order_info['compatibilityEdgeCount']} same-session attachment parent edges "
@@ -3030,16 +3400,17 @@ def choose_resume_leaf_info(
3030
3400
  info["sessionLineageRunCount"] = lineage_info["runCount"]
3031
3401
  info["sessionLineageTransitionCount"] = lineage_info["transitionCount"]
3032
3402
  info["sessionLineageRunsDigest"] = lineage_info["runsDigest"]
3403
+ info["lineageReason"] = lineage_info["reason"]
3033
3404
  info["currentSessionStartPosition"] = lineage_info["currentSessionStartPosition"]
3034
3405
  info["currentSessionRecordCount"] = lineage_info["currentSessionRecordCount"]
3035
3406
  info["currentSessionId"] = lineage_info["currentSessionId"]
3036
3407
  if not lineage_info["ok"]:
3037
- info["status"] = "session-mismatch"
3038
- info["errors"].append(
3408
+ return _reject_resume_path(
3409
+ info,
3410
+ "lineage-unsafe",
3039
3411
  "authoritative resume chain has an unsafe sessionId lineage: "
3040
- f"{lineage_info['reason']}"
3412
+ f"{lineage_info['reason']}",
3041
3413
  )
3042
- return info
3043
3414
  if lineage_info["compatibility"]:
3044
3415
  info["warnings"].append(
3045
3416
  f"accepted one-way session lineage with {lineage_info['transitionCount']} transition(s); "
@@ -3050,50 +3421,58 @@ def choose_resume_leaf_info(
3050
3421
  extension_pairs = list(enumerate(records[prompt_index + 1 :], start=prompt_index + 1))
3051
3422
  if extension_pairs:
3052
3423
  if len(extension_pairs) > max_post_prompt_extension:
3053
- info["status"] = "extension-limit"
3054
- info["errors"].append(
3055
- f"post-last-prompt closure has {len(extension_pairs)} UUID records, exceeding limit {max_post_prompt_extension}"
3424
+ return _reject_resume_path(
3425
+ info,
3426
+ "extension-limit-exceeded",
3427
+ f"post-last-prompt closure has {len(extension_pairs)} UUID records, "
3428
+ f"exceeding limit {max_post_prompt_extension}",
3056
3429
  )
3057
- return info
3058
3430
  expected_parent = prompt_leaf
3059
3431
  expected_session = prompt_record.get("sessionId")
3060
3432
  if not isinstance(expected_session, str) or not expected_session:
3061
- info["status"] = "session-mismatch"
3062
- info["errors"].append("post-last-prompt closure requires a non-empty authority sessionId")
3063
- return info
3433
+ return _reject_resume_path(
3434
+ info,
3435
+ "extension-authority-session-missing",
3436
+ "post-last-prompt closure requires a non-empty authority sessionId",
3437
+ )
3064
3438
  pending_tool_ids = set(tool_use_ids(chain[-1]))
3065
3439
  extension_reasons: List[str] = []
3066
3440
  for idx, obj in extension_pairs:
3067
3441
  if not isinstance(obj.get("uuid"), str) or not obj.get("uuid"):
3068
- info["status"] = "extension-unsafe"
3069
- info["errors"].append(
3070
- f"post-last-prompt record L{idx + 1} has no UUID and breaks the physical closure sequence"
3442
+ return _reject_resume_path(
3443
+ info,
3444
+ "extension-record-missing-uuid",
3445
+ f"post-last-prompt record L{idx + 1} has no UUID and breaks the physical closure sequence",
3071
3446
  )
3072
- return info
3073
3447
  if obj.get("parentUuid") != expected_parent:
3074
- info["status"] = "extension-branch"
3075
- info["errors"].append(f"post-last-prompt record L{idx + 1} is not a direct linear descendant")
3076
- return info
3448
+ return _reject_resume_path(
3449
+ info,
3450
+ "extension-record-not-linear-descendant",
3451
+ f"post-last-prompt record L{idx + 1} is not a direct linear descendant",
3452
+ )
3077
3453
  obj_session = obj.get("sessionId")
3078
3454
  if not isinstance(obj_session, str) or not obj_session or obj_session != expected_session:
3079
- info["status"] = "session-mismatch"
3080
- info["errors"].append(
3081
- f"post-last-prompt record L{idx + 1} does not have the exact authority sessionId"
3455
+ return _reject_resume_path(
3456
+ info,
3457
+ "extension-record-session-mismatch",
3458
+ f"post-last-prompt record L{idx + 1} does not have the exact authority sessionId",
3082
3459
  )
3083
- return info
3084
3460
  allowed, reason = _allowed_post_prompt_closure(obj, pending_tool_ids)
3085
3461
  if not allowed:
3086
- info["status"] = "extension-unsafe"
3087
- info["errors"].append(f"post-last-prompt record L{idx + 1} is not a safe closure: {reason}")
3088
- return info
3462
+ return _reject_resume_path(
3463
+ info,
3464
+ "extension-record-not-safe-closure",
3465
+ f"post-last-prompt record L{idx + 1} is not a safe closure: {reason}",
3466
+ )
3089
3467
  extension_reasons.append(reason)
3090
3468
  expected_parent = obj.get("uuid")
3091
3469
  if pending_tool_ids:
3092
- info["status"] = "extension-unsafe"
3093
- info["errors"].append(
3094
- f"post-last-prompt closure leaves pending tool_use ids unresolved: {sorted(pending_tool_ids)[:20]}"
3470
+ return _reject_resume_path(
3471
+ info,
3472
+ "extension-pending-tool-ids",
3473
+ f"post-last-prompt closure leaves pending tool_use ids unresolved: "
3474
+ f"{sorted(pending_tool_ids)[:20]}",
3095
3475
  )
3096
- return info
3097
3476
  selected_extension_records = [obj for _, obj in extension_pairs]
3098
3477
  chain.extend(selected_extension_records)
3099
3478
  chain_indexes.extend(idx for idx, _ in extension_pairs)
@@ -3106,7 +3485,8 @@ def choose_resume_leaf_info(
3106
3485
  info["postLastPromptExtensionUuids"] = [obj.get("uuid") for _, obj in extension_pairs]
3107
3486
  info["currentSessionRecordCount"] = int(info.get("currentSessionRecordCount") or 0) + len(extension_pairs)
3108
3487
 
3109
- info["status"] = "valid"
3488
+ info["status"] = RESUME_REASON_CODE_STATUS["ok"]
3489
+ info["reasonCode"] = "ok"
3110
3490
  info["ok"] = True
3111
3491
  info["activeChainIndexes"] = chain_indexes
3112
3492
  info["activeChainUuids"] = [obj.get("uuid") for obj in chain]
@@ -3120,13 +3500,18 @@ def require_resume_leaf_info(
3120
3500
  ) -> Dict[str, Any]:
3121
3501
  info = choose_resume_leaf_info(records, max_post_prompt_extension, resume_leaf_override)
3122
3502
  if not info.get("ok"):
3123
- raise ValueError("strict active-chain topology failed: " + "; ".join(info.get("errors") or [str(info.get("status"))]))
3503
+ reason_code = info.get("reasonCode") or "unknown"
3504
+ raise ValueError(
3505
+ f"strict active-chain topology failed (reasonCode={reason_code}): "
3506
+ + "; ".join(info.get("errors") or [str(info.get("status"))])
3507
+ )
3124
3508
  return info
3125
3509
 
3126
3510
 
3127
3511
  def public_resume_leaf_info(info: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
3128
3512
  if info is None:
3129
3513
  return None
3514
+ redactions = _long_diagnostic_strings(info)
3130
3515
  public = copy.deepcopy(info)
3131
3516
  public.pop("lastPromptTemplate", None)
3132
3517
  active_indexes = public.pop("activeChainIndexes", [])
@@ -3134,7 +3519,40 @@ def public_resume_leaf_info(info: Optional[Dict[str, Any]]) -> Optional[Dict[str
3134
3519
  public["activeChainRecordCount"] = len(active_indexes)
3135
3520
  public["activeChainLinesDigest"] = stable_digest(",".join(str(int(idx) + 1) for idx in active_indexes))
3136
3521
  public["activeChainUuidsDigest"] = stable_digest(",".join(str(uid) for uid in active_uuids))
3137
- return public
3522
+ for key in (
3523
+ "promptLeafUuid",
3524
+ "physicalLeafUuid",
3525
+ "selectedLeafUuid",
3526
+ "promptChainMissingUuid",
3527
+ "promptChainLoopUuid",
3528
+ "promptChainMalformedParentUuid",
3529
+ "authoritySessionId",
3530
+ "currentSessionId",
3531
+ ):
3532
+ if key in public and public[key] is not None:
3533
+ public[key] = _bounded_diagnostic_label(public[key])
3534
+ for key in ("duplicateUuids", "postLastPromptExtensionUuids"):
3535
+ values = public.get(key)
3536
+ if isinstance(values, list):
3537
+ public[key] = [_bounded_diagnostic_label(value) for value in values]
3538
+ return _bounded_public_diagnostic_structure(
3539
+ public,
3540
+ redactions=redactions,
3541
+ redact_unmatched_diagnostic_text=True,
3542
+ )
3543
+
3544
+
3545
+ def public_parent_repair_details(repairs: Sequence[JsonObj]) -> List[JsonObj]:
3546
+ """Project repair diagnostics without copying arbitrary metadata verbatim."""
3547
+ redactions = _long_diagnostic_strings(repairs)
3548
+ public: List[JsonObj] = []
3549
+ for repair in repairs:
3550
+ projected = copy.deepcopy(repair)
3551
+ for key in ("uuid", "type", "sessionId", "oldParentUuid", "newParentUuid"):
3552
+ if key in projected and projected[key] is not None:
3553
+ projected[key] = _bounded_diagnostic_label(projected[key])
3554
+ public.append(projected)
3555
+ return _bounded_public_diagnostic_structure(public, redactions=redactions)
3138
3556
 
3139
3557
 
3140
3558
  def choose_resume_leaf(records: Sequence[JsonObj]) -> Optional[str]:
@@ -3234,7 +3652,7 @@ def select_control_projection_indexes(records: Sequence[JsonObj]) -> List[int]:
3234
3652
  ui_types = {"mode", "permission-mode", "custom-title", "ai-title", "agent-name"}
3235
3653
  selected = [
3236
3654
  idx for idx, obj in enumerate(records[:first_uuid_index])
3237
- if obj.get("type") in ui_types and "uuid" not in obj
3655
+ if isinstance(obj.get("type"), str) and obj.get("type") in ui_types and "uuid" not in obj
3238
3656
  ]
3239
3657
  selected.extend(idx for idx, obj in enumerate(records) if obj.get("type") == "last-prompt")
3240
3658
  return sorted(set(selected))
@@ -3391,7 +3809,10 @@ def merge_assistant_fragments(api_records: Sequence[JsonObj]) -> List[JsonObj]:
3391
3809
  if isinstance(target_msg, dict) and isinstance(obj_msg, dict):
3392
3810
  target_blocks = content_blocks(target)
3393
3811
  obj_blocks = content_blocks(obj)
3812
+ target_content_lines = _validation_content_lines(target, target.get("_line"))
3813
+ obj_content_lines = _validation_content_lines(obj, obj.get("_line"))
3394
3814
  target_msg["content"] = target_blocks + obj_blocks
3815
+ target["_validationContentLines"] = target_content_lines + obj_content_lines
3395
3816
  if isinstance(obj.get("uuid"), str):
3396
3817
  target["uuid"] = obj.get("uuid")
3397
3818
  merged_lines = list(target.get("_mergedLines", []))
@@ -3458,10 +3879,12 @@ def merge_split_tool_result_users(api_records: Sequence[JsonObj]) -> List[JsonOb
3458
3879
  combined_msg = combined.get("message")
3459
3880
  if isinstance(combined_msg, dict):
3460
3881
  blocks: List[Any] = []
3882
+ content_lines: List[int] = []
3461
3883
  merged_lines: List[Any] = []
3462
3884
  merged_uuids: List[str] = []
3463
3885
  for user_obj in group:
3464
3886
  blocks.extend(content_blocks(user_obj))
3887
+ content_lines.extend(_validation_content_lines(user_obj, user_obj.get("_line")))
3465
3888
  merged_lines.append(user_obj.get("_line"))
3466
3889
  if isinstance(user_obj.get("uuid"), str):
3467
3890
  merged_uuids.append(user_obj.get("uuid"))
@@ -3469,6 +3892,7 @@ def merge_split_tool_result_users(api_records: Sequence[JsonObj]) -> List[JsonOb
3469
3892
  last_user = group[-1]
3470
3893
  if isinstance(last_user.get("uuid"), str):
3471
3894
  combined["uuid"] = last_user.get("uuid")
3895
+ combined["_validationContentLines"] = content_lines
3472
3896
  combined["_mergedLines"] = merged_lines
3473
3897
  if merged_uuids:
3474
3898
  combined["_mergedUuids"] = merged_uuids
@@ -3477,15 +3901,102 @@ def merge_split_tool_result_users(api_records: Sequence[JsonObj]) -> List[JsonOb
3477
3901
  return merged
3478
3902
 
3479
3903
 
3480
- def active_api_messages_for_validation(records: Sequence[JsonObj]) -> List[Tuple[int, JsonObj]]:
3904
+ def _validated_record_lines(
3905
+ records: Sequence[JsonObj],
3906
+ physical_lines: Optional[Sequence[int]] = None,
3907
+ *,
3908
+ allow_nonmonotonic: bool = False,
3909
+ ) -> List[int]:
3910
+ if physical_lines is None:
3911
+ return list(range(1, len(records) + 1))
3912
+ lines = list(physical_lines)
3913
+ if len(lines) != len(records):
3914
+ raise ValueError("physical line map length does not match record count")
3915
+ previous = 0
3916
+ for line in lines:
3917
+ if isinstance(line, bool) or not isinstance(line, int) or line <= 0:
3918
+ raise ValueError("physical line map must contain positive integers")
3919
+ if not allow_nonmonotonic and line <= previous:
3920
+ raise ValueError("physical line map must contain strictly increasing positive integers")
3921
+ previous = line
3922
+ return lines
3923
+
3924
+
3925
+ def _diagnostic_counter_label(obj: JsonObj, key: str) -> str:
3926
+ return _bounded_diagnostic_label(obj.get(key), missing=key not in obj)
3927
+
3928
+
3929
+ def _validation_content_lines(obj: JsonObj, fallback_line: Any) -> List[int]:
3930
+ if isinstance(fallback_line, bool) or not isinstance(fallback_line, int) or fallback_line <= 0:
3931
+ fallback_line = 1
3932
+ blocks = content_blocks(obj)
3933
+ candidate = obj.get("_validationContentLines")
3934
+ if (
3935
+ isinstance(candidate, list)
3936
+ and len(candidate) == len(blocks)
3937
+ and all(isinstance(line, int) and not isinstance(line, bool) and line > 0 for line in candidate)
3938
+ ):
3939
+ return list(candidate)
3940
+ return [fallback_line] * len(blocks)
3941
+
3942
+
3943
+ def _first_validation_block_line(obj: JsonObj, block_type: str, fallback_line: int) -> int:
3944
+ block_lines = _validation_content_lines(obj, fallback_line)
3945
+ for index, block in enumerate(content_blocks(obj)):
3946
+ if isinstance(block, dict) and block.get("type") == block_type:
3947
+ return block_lines[index]
3948
+ return fallback_line
3949
+
3950
+
3951
+ def _ordered_subsequence_failure_line(
3952
+ obj: JsonObj,
3953
+ expected: Sequence[str],
3954
+ fallback_line: int,
3955
+ ) -> int:
3956
+ block_lines = _validation_content_lines(obj, fallback_line)
3957
+ position = 0
3958
+ for index, block in enumerate(content_blocks(obj)):
3959
+ if not isinstance(block, dict) or block.get("type") != "tool_result":
3960
+ continue
3961
+ tool_id = block.get("tool_use_id") or block.get("toolUseID")
3962
+ if not isinstance(tool_id, str):
3963
+ continue
3964
+ while position < len(expected) and expected[position] != tool_id:
3965
+ position += 1
3966
+ if position >= len(expected):
3967
+ return block_lines[index]
3968
+ position += 1
3969
+ return fallback_line
3970
+
3971
+
3972
+ def active_api_messages_for_validation(
3973
+ records: Sequence[JsonObj],
3974
+ physical_lines: Optional[Sequence[int]] = None,
3975
+ *,
3976
+ allow_nonmonotonic_diagnostic_lines: bool = False,
3977
+ ) -> List[Tuple[int, JsonObj]]:
3978
+ record_lines = _validated_record_lines(
3979
+ records,
3980
+ physical_lines,
3981
+ allow_nonmonotonic=allow_nonmonotonic_diagnostic_lines,
3982
+ )
3983
+ resume_info = choose_resume_leaf_info(records)
3984
+ active_indexes = list(resume_info.get("activeChainIndexes") or []) if resume_info.get("ok") else []
3481
3985
  api_records: List[JsonObj] = []
3482
- for idx, obj in enumerate(active_chain_records(records), 1):
3986
+ for record_index in active_indexes:
3987
+ obj = records[record_index]
3483
3988
  if not is_api_message(obj):
3484
3989
  continue
3485
- clone = obj if "_line" in obj else {**obj, "_line": idx}
3990
+ clone = {
3991
+ key: value
3992
+ for key, value in obj.items()
3993
+ if key not in {"_line", "_mergedLines", "_mergedUuids", "_validationContentLines"}
3994
+ }
3995
+ clone["_line"] = record_lines[record_index]
3996
+ clone["_validationContentLines"] = [record_lines[record_index]] * len(content_blocks(obj))
3486
3997
  api_records.append(clone)
3487
3998
  merged = merge_split_tool_result_users(merge_assistant_fragments(api_records))
3488
- return [(int(obj.get("_line") or idx), obj) for idx, obj in enumerate(merged, 1)]
3999
+ return [(obj["_line"], obj) for obj in merged]
3489
4000
 
3490
4001
 
3491
4002
  def adjust_recent_start_for_tool_pairs(records: Sequence[JsonObj], start: int) -> int:
@@ -3780,11 +4291,13 @@ def is_assistant_research_decision(obj: JsonObj, text: Optional[str] = None) ->
3780
4291
 
3781
4292
  def collect_summary_inputs(records: Sequence[JsonObj]) -> Dict[str, Any]:
3782
4293
  ensure_summary_resources()
3783
- type_counts = collections.Counter(obj.get("type", "<missing>") for obj in records)
4294
+ type_counts = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in records)
3784
4295
  subtype_counts = collections.Counter(
3785
- f"{obj.get('type')}:{obj.get('subtype')}" for obj in records if obj.get("subtype")
4296
+ f"{_diagnostic_counter_label(obj, 'type')}:{_diagnostic_counter_label(obj, 'subtype')}"
4297
+ for obj in records
4298
+ if "subtype" in obj and obj.get("subtype") not in (None, "")
3786
4299
  )
3787
- session_counts = collections.Counter(obj.get("sessionId", "<missing>") for obj in records)
4300
+ session_counts = collections.Counter(_diagnostic_counter_label(obj, "sessionId") for obj in records)
3788
4301
  cwd_counts = collections.Counter(str(obj.get("cwd")) for obj in records if obj.get("cwd"))
3789
4302
  versions = collections.Counter(str(obj.get("version")) for obj in records if obj.get("version"))
3790
4303
  tools = collections.Counter()
@@ -3850,7 +4363,11 @@ def collect_summary_inputs(records: Sequence[JsonObj]) -> Dict[str, Any]:
3850
4363
  user_items.append((idx, one_line(txt, 700 if is_important else 420)))
3851
4364
  elif obj.get("type") == "system":
3852
4365
  if obj.get("subtype") == "api_error" or obj.get("error"):
3853
- errors.append((idx, one_line(txt or json.dumps(obj, ensure_ascii=False), 520)))
4366
+ fallback = (
4367
+ f"system-error type={_diagnostic_counter_label(obj, 'type')} "
4368
+ f"subtype={_diagnostic_counter_label(obj, 'subtype')}"
4369
+ )
4370
+ errors.append((idx, one_line(txt or fallback, 520)))
3854
4371
  elif obj.get("type") == "attachment":
3855
4372
  att = obj.get("attachment")
3856
4373
  if isinstance(att, dict):
@@ -4145,7 +4662,7 @@ def make_summary_text(
4145
4662
  all_info = collect_summary_inputs(omitted)
4146
4663
  early_info = collect_summary_inputs(early)
4147
4664
  middle_info = collect_summary_inputs(middle)
4148
- kept_types = collections.Counter(obj.get("type", "<missing>") for obj in kept)
4665
+ kept_types = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in kept)
4149
4666
  first_ts = next((obj.get("timestamp") for obj in omitted if obj.get("timestamp")), "unknown")
4150
4667
  last_omitted_ts = next((obj.get("timestamp") for obj in reversed(omitted) if obj.get("timestamp")), "unknown")
4151
4668
  first_kept_ts = next((obj.get("timestamp") for obj in kept if obj.get("timestamp")), "unknown")
@@ -4823,10 +5340,12 @@ def model_evidence_record(obj: JsonObj, original_line: int) -> Optional[Dict[str
4823
5340
  noisy = is_noisy_text(txt)
4824
5341
  if not txt or (noisy and not mandatory_semantic and not is_prior_summary):
4825
5342
  return None
4826
- role = api_role(obj) or str(obj.get("type", "<missing>"))
5343
+ record_type = _diagnostic_counter_label(obj, "type")
5344
+ role_value = api_role(obj)
5345
+ role = _bounded_diagnostic_label(role_value) if role_value is not None else record_type
4827
5346
  timestamp = obj.get("timestamp") or ""
4828
5347
  uid = obj.get("uuid") or ""
4829
- prefix = f"L{original_line} type={obj.get('type')} role={role}"
5348
+ prefix = f"L{original_line} type={record_type} role={role}"
4830
5349
  if timestamp:
4831
5350
  prefix += f" ts={timestamp}"
4832
5351
  if uid:
@@ -5826,8 +6345,8 @@ def summarize_active_chain_window(records: Sequence[JsonObj], summary_uuid: str)
5826
6345
  uuids = [obj.get("uuid") for obj in chain if isinstance(obj.get("uuid"), str)]
5827
6346
  summary_pos = next((idx for idx, obj in enumerate(chain) if obj.get("uuid") == summary_uuid), None)
5828
6347
  after_summary = chain[summary_pos + 1 :] if isinstance(summary_pos, int) else []
5829
- session_counts = collections.Counter(obj.get("sessionId", "<missing>") for obj in chain)
5830
- type_counts = collections.Counter(obj.get("type", "<missing>") for obj in chain)
6348
+ session_counts = collections.Counter(_diagnostic_counter_label(obj, "sessionId") for obj in chain)
6349
+ type_counts = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in chain)
5831
6350
  return {
5832
6351
  "leafUuid": latest_leaf,
5833
6352
  "chainLength": len(chain),
@@ -5960,8 +6479,11 @@ def select_preservation_plan(
5960
6479
  def validate_source_active_chain_for_plan(
5961
6480
  records: Sequence[JsonObj],
5962
6481
  plan: Dict[str, Any],
6482
+ *,
6483
+ physical_lines: Optional[Sequence[int]] = None,
5963
6484
  ) -> Optional[Dict[str, Any]]:
5964
6485
  """Validate the authoritative logical chain without inspecting inactive branches."""
6486
+ source_lines = _validated_record_lines(records, physical_lines)
5965
6487
  preserve_info = plan.get("preserve_info")
5966
6488
  if not isinstance(preserve_info, dict):
5967
6489
  return None
@@ -5971,7 +6493,16 @@ def validate_source_active_chain_for_plan(
5971
6493
  active_indexes = list(resume_info.get("activeChainIndexes") or [])
5972
6494
  if not active_indexes:
5973
6495
  raise ValueError("source active-chain validation failed: authoritative chain is empty")
5974
- active_records = [copy.deepcopy(records[int(idx)]) for idx in active_indexes]
6496
+ if any(
6497
+ isinstance(index, bool)
6498
+ or not isinstance(index, int)
6499
+ or index < 0
6500
+ or index >= len(records)
6501
+ for index in active_indexes
6502
+ ):
6503
+ raise ValueError("source active-chain validation failed: preservation plan has invalid active-chain indexes")
6504
+ active_records = [copy.deepcopy(records[index]) for index in active_indexes]
6505
+ active_record_lines = [source_lines[index] for index in active_indexes]
5975
6506
  selected_leaf = plan.get("selected_leaf_uuid")
5976
6507
  if not isinstance(selected_leaf, str) or not selected_leaf:
5977
6508
  raise ValueError("source active-chain validation failed: selected leaf is missing")
@@ -5990,7 +6521,17 @@ def validate_source_active_chain_for_plan(
5990
6521
  if isinstance(leaf_session, str) and leaf_session:
5991
6522
  pointer["sessionId"] = leaf_session
5992
6523
  active_records.append(pointer)
5993
- validation = validate_records(active_records)
6524
+ pointer_index = resume_info.get("lastPromptIndex")
6525
+ if isinstance(pointer_index, int) and not isinstance(pointer_index, bool) and 0 <= pointer_index < len(records):
6526
+ pointer_line = source_lines[pointer_index]
6527
+ else:
6528
+ pointer_line = max(source_lines, default=0) + 1
6529
+ active_record_lines.append(pointer_line)
6530
+ validation = validate_records(
6531
+ active_records,
6532
+ physical_lines=active_record_lines,
6533
+ allow_nonmonotonic_diagnostic_lines=True,
6534
+ )
5994
6535
  if not validation.get("ok"):
5995
6536
  raise ValueError(
5996
6537
  "source active-chain validation failed: "
@@ -6021,7 +6562,7 @@ def build_model_summary_pack_for_input(
6021
6562
  if handoff_summary_path is not None and is_under_claude_root(handoff_summary_path):
6022
6563
  raise ValueError("--handoff-summary process files must be outside the entire .claude directory")
6023
6564
  source_bytes = input_path.read_bytes()
6024
- records, raw_lines = parse_jsonl_bytes(source_bytes, source_label=str(input_path))
6565
+ records, raw_lines, physical_lines = parse_jsonl_bytes_with_lines(source_bytes, source_label=str(input_path))
6025
6566
  if not records:
6026
6567
  raise ValueError("input JSONL has no records")
6027
6568
  input_bytes = len(source_bytes)
@@ -6041,7 +6582,11 @@ def build_model_summary_pack_for_input(
6041
6582
  checkpoint_policy,
6042
6583
  resume_leaf_override,
6043
6584
  )
6044
- source_active_chain_validation = validate_source_active_chain_for_plan(records, plan)
6585
+ source_active_chain_validation = validate_source_active_chain_for_plan(
6586
+ records,
6587
+ plan,
6588
+ physical_lines=physical_lines,
6589
+ )
6045
6590
  source_sha256 = sha256_hex(source_bytes)
6046
6591
  source_digest = source_sha256
6047
6592
  omitted_digest = sha256_hex("\n".join(raw_lines[idx] for idx in plan["omitted_indexes"]).encode("utf-8"))
@@ -6180,7 +6725,7 @@ def compress_jsonl(
6180
6725
  if model_summary_path is not None and deterministic_summary:
6181
6726
  raise ValueError("model_summary_path and deterministic_summary=True are mutually exclusive")
6182
6727
  source_bytes = input_path.read_bytes()
6183
- records, raw_lines = parse_jsonl_bytes(source_bytes, source_label=str(input_path))
6728
+ records, raw_lines, physical_lines = parse_jsonl_bytes_with_lines(source_bytes, source_label=str(input_path))
6184
6729
  if not records:
6185
6730
  raise ValueError("input JSONL has no records")
6186
6731
  input_bytes = len(source_bytes)
@@ -6218,7 +6763,11 @@ def compress_jsonl(
6218
6763
  checkpoint_policy,
6219
6764
  resume_leaf_override,
6220
6765
  )
6221
- source_active_chain_validation = validate_source_active_chain_for_plan(records, plan)
6766
+ source_active_chain_validation = validate_source_active_chain_for_plan(
6767
+ records,
6768
+ plan,
6769
+ physical_lines=physical_lines,
6770
+ )
6222
6771
  preserve_info = plan["preserve_info"]
6223
6772
  omitted = list(plan["omitted"])
6224
6773
  kept_original = list(plan["kept_original"])
@@ -6526,7 +7075,7 @@ def compress_jsonl(
6526
7075
  "omitted_digest": omitted_text_digest,
6527
7076
  "safe_prefix_records": len(safe_prefix),
6528
7077
  "parent_repairs": len(parent_repair_details),
6529
- "parent_repair_details": parent_repair_details,
7078
+ "parent_repair_details": public_parent_repair_details(parent_repair_details),
6530
7079
  "last_prompt_updates": last_prompt_updates,
6531
7080
  "target_session_id": target_session_id,
6532
7081
  "normalized_session_records": normalized_session_records,
@@ -6545,35 +7094,50 @@ def compress_jsonl(
6545
7094
  return report
6546
7095
 
6547
7096
 
6548
- def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
7097
+ def validate_records(
7098
+ records: Sequence[JsonObj],
7099
+ physical_lines: Optional[Sequence[int]] = None,
7100
+ *,
7101
+ allow_nonmonotonic_diagnostic_lines: bool = False,
7102
+ ) -> Dict[str, Any]:
6549
7103
  errors: List[str] = []
6550
7104
  warnings: List[str] = []
7105
+ record_lines = _validated_record_lines(
7106
+ records,
7107
+ physical_lines,
7108
+ allow_nonmonotonic=allow_nonmonotonic_diagnostic_lines,
7109
+ )
6551
7110
  uuid_counts = collections.Counter(obj.get("uuid") for obj in records if isinstance(obj.get("uuid"), str))
6552
7111
  duplicates = sorted(k for k, v in uuid_counts.items() if v > 1)
6553
7112
  if duplicates:
6554
7113
  errors.append(f"duplicate UUIDs: {duplicates[:20]}")
6555
7114
  uuid_set = set(uuid_counts)
6556
7115
  uuid_to_record = {obj.get("uuid"): obj for obj in records if isinstance(obj.get("uuid"), str)}
6557
- uuid_to_line = {obj.get("uuid"): idx for idx, obj in enumerate(records, 1) if isinstance(obj.get("uuid"), str)}
7116
+ uuid_to_line = {
7117
+ obj.get("uuid"): record_lines[index]
7118
+ for index, obj in enumerate(records)
7119
+ if isinstance(obj.get("uuid"), str)
7120
+ }
6558
7121
  missing_parent: List[Tuple[int, str]] = []
6559
7122
  malformed_parent: List[JsonObj] = []
6560
7123
  cross_session_parent: List[JsonObj] = []
6561
- for idx, obj in enumerate(records, 1):
7124
+ for record_index, obj in enumerate(records):
7125
+ line = record_lines[record_index]
6562
7126
  parent = obj.get("parentUuid")
6563
7127
  if parent is None:
6564
7128
  continue
6565
7129
  if not isinstance(parent, str) or not parent:
6566
7130
  malformed_parent.append(
6567
7131
  {
6568
- "line": idx,
7132
+ "line": line,
6569
7133
  "uuid": obj.get("uuid"),
6570
- "type": obj.get("type"),
7134
+ "type": _diagnostic_counter_label(obj, "type"),
6571
7135
  "parentUuidType": type(parent).__name__,
6572
7136
  }
6573
7137
  )
6574
7138
  continue
6575
7139
  if parent not in uuid_set:
6576
- missing_parent.append((idx, str(parent)))
7140
+ missing_parent.append((line, str(parent)))
6577
7141
  else:
6578
7142
  parent_obj = uuid_to_record.get(parent)
6579
7143
  child_session = obj.get("sessionId")
@@ -6581,12 +7145,12 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6581
7145
  if isinstance(child_session, str) and isinstance(parent_session, str) and child_session != parent_session:
6582
7146
  cross_session_parent.append(
6583
7147
  {
6584
- "line": idx,
7148
+ "line": line,
6585
7149
  "uuid": obj.get("uuid"),
6586
- "type": obj.get("type"),
6587
- "sessionId": child_session,
7150
+ "type": _diagnostic_counter_label(obj, "type"),
7151
+ "sessionId": _diagnostic_counter_label(obj, "sessionId"),
6588
7152
  "parentUuid": parent,
6589
- "parentSessionId": parent_session,
7153
+ "parentSessionId": _bounded_diagnostic_label(parent_session),
6590
7154
  }
6591
7155
  )
6592
7156
  if malformed_parent:
@@ -6594,7 +7158,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6594
7158
  if missing_parent:
6595
7159
  errors.append(f"missing parentUuid references: {missing_parent[:30]}")
6596
7160
  last_prompt_records = [
6597
- (idx, obj) for idx, obj in enumerate(records, 1) if obj.get("type") == "last-prompt"
7161
+ (record_lines[index], obj)
7162
+ for index, obj in enumerate(records)
7163
+ if obj.get("type") == "last-prompt"
6598
7164
  ]
6599
7165
  last_prompt_malformed: List[JsonObj] = []
6600
7166
  last_prompt_missing: List[JsonObj] = []
@@ -6603,12 +7169,22 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6603
7169
  leaf = obj.get("leafUuid")
6604
7170
  if not isinstance(leaf, str) or not leaf:
6605
7171
  last_prompt_malformed.append(
6606
- {"line": idx, "leafUuidType": type(leaf).__name__, "sessionId": obj.get("sessionId")}
7172
+ {
7173
+ "line": idx,
7174
+ "leafUuidType": type(leaf).__name__,
7175
+ "sessionId": _diagnostic_counter_label(obj, "sessionId"),
7176
+ }
6607
7177
  )
6608
7178
  continue
6609
7179
  target = uuid_to_record.get(leaf)
6610
7180
  if target is None:
6611
- last_prompt_missing.append({"line": idx, "leafUuid": leaf, "sessionId": obj.get("sessionId")})
7181
+ last_prompt_missing.append(
7182
+ {
7183
+ "line": idx,
7184
+ "leafUuid": _bounded_diagnostic_label(leaf),
7185
+ "sessionId": _diagnostic_counter_label(obj, "sessionId"),
7186
+ }
7187
+ )
6612
7188
  continue
6613
7189
  lp_session = obj.get("sessionId")
6614
7190
  target_session = target.get("sessionId")
@@ -6616,9 +7192,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6616
7192
  last_prompt_cross_session.append(
6617
7193
  {
6618
7194
  "line": idx,
6619
- "leafUuid": leaf,
6620
- "sessionId": lp_session,
6621
- "leafSessionId": target_session,
7195
+ "leafUuid": _bounded_diagnostic_label(leaf),
7196
+ "sessionId": _bounded_diagnostic_label(lp_session),
7197
+ "leafSessionId": _bounded_diagnostic_label(target_session),
6622
7198
  }
6623
7199
  )
6624
7200
  if last_prompt_missing:
@@ -6626,7 +7202,7 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6626
7202
  if last_prompt_cross_session:
6627
7203
  errors.append(f"last-prompt leafUuid cross-session targets: {last_prompt_cross_session[:20]}")
6628
7204
  physical_last_prompt = latest_last_prompt_entry(records)
6629
- physical_last_prompt_line = physical_last_prompt[0] + 1 if physical_last_prompt else None
7205
+ physical_last_prompt_line = record_lines[physical_last_prompt[0]] if physical_last_prompt else None
6630
7206
  physical_last_prompt_malformed = bool(
6631
7207
  physical_last_prompt
6632
7208
  and (not isinstance(physical_last_prompt[1].get("leafUuid"), str) or not physical_last_prompt[1].get("leafUuid"))
@@ -6637,32 +7213,38 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6637
7213
  )
6638
7214
  elif last_prompt_malformed:
6639
7215
  warnings.append(f"earlier malformed last-prompt records: {last_prompt_malformed[:20]}")
6640
- api_messages = active_api_messages_for_validation(records)
7216
+ api_messages = active_api_messages_for_validation(
7217
+ records,
7218
+ physical_lines=record_lines,
7219
+ allow_nonmonotonic_diagnostic_lines=allow_nonmonotonic_diagnostic_lines,
7220
+ )
6641
7221
  tool_pair_errors: List[JsonObj] = []
6642
7222
  tool_pair_partial_result_count = 0
6643
7223
  tool_use_occurrences: Dict[str, List[int]] = collections.defaultdict(list)
6644
7224
  tool_result_occurrences: Dict[str, List[int]] = collections.defaultdict(list)
6645
7225
  malformed_tool_id_samples: List[JsonObj] = []
6646
7226
  for line, api_obj in api_messages:
6647
- for block in content_blocks(api_obj):
7227
+ block_lines = _validation_content_lines(api_obj, line)
7228
+ for block_index, block in enumerate(content_blocks(api_obj)):
6648
7229
  if not isinstance(block, dict):
6649
7230
  continue
7231
+ block_line = block_lines[block_index]
6650
7232
  if block.get("type") == "tool_use":
6651
7233
  tool_id = block.get("id")
6652
7234
  if not isinstance(tool_id, str) or not tool_id:
6653
7235
  malformed_tool_id_samples.append(
6654
- {"line": line, "kind": "tool_use", "idType": type(tool_id).__name__}
7236
+ {"line": block_line, "kind": "tool_use", "idType": type(tool_id).__name__}
6655
7237
  )
6656
7238
  else:
6657
- tool_use_occurrences[tool_id].append(line)
7239
+ tool_use_occurrences[tool_id].append(block_line)
6658
7240
  elif block.get("type") == "tool_result":
6659
7241
  tool_id = block.get("tool_use_id") or block.get("toolUseID")
6660
7242
  if not isinstance(tool_id, str) or not tool_id:
6661
7243
  malformed_tool_id_samples.append(
6662
- {"line": line, "kind": "tool_result", "idType": type(tool_id).__name__}
7244
+ {"line": block_line, "kind": "tool_result", "idType": type(tool_id).__name__}
6663
7245
  )
6664
7246
  else:
6665
- tool_result_occurrences[tool_id].append(line)
7247
+ tool_result_occurrences[tool_id].append(block_line)
6666
7248
  duplicate_tool_use_ids = {
6667
7249
  tool_id: lines for tool_id, lines in tool_use_occurrences.items() if len(lines) > 1
6668
7250
  }
@@ -6699,21 +7281,24 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6699
7281
  for pos, (line, obj) in enumerate(api_messages):
6700
7282
  uses = tool_use_ids(obj)
6701
7283
  results = tool_result_ids(obj)
7284
+ tool_use_line = _first_validation_block_line(obj, "tool_use", line)
7285
+ tool_result_line = _first_validation_block_line(obj, "tool_result", line)
6702
7286
  if uses:
6703
7287
  if pos + 1 >= len(api_messages):
6704
7288
  tool_pair_errors.append(
6705
- {"line": line, "uuid": obj.get("uuid"), "reason": "assistant tool_use is final api message", "toolUseIds": uses}
7289
+ {"line": tool_use_line, "uuid": obj.get("uuid"), "reason": "assistant tool_use is final api message", "toolUseIds": uses}
6706
7290
  )
6707
7291
  else:
6708
7292
  next_line, next_obj = api_messages[pos + 1]
6709
7293
  next_results = tool_result_ids(next_obj)
7294
+ next_result_line = _first_validation_block_line(next_obj, "tool_result", next_line)
6710
7295
  if api_role(next_obj) != "user":
6711
7296
  tool_pair_errors.append(
6712
7297
  {
6713
- "line": line,
7298
+ "line": tool_use_line,
6714
7299
  "uuid": obj.get("uuid"),
6715
7300
  "reason": "assistant tool_use not followed by user message",
6716
- "nextLine": next_line,
7301
+ "nextLine": next_result_line,
6717
7302
  "nextRole": api_role(next_obj),
6718
7303
  "toolUseIds": uses,
6719
7304
  }
@@ -6721,11 +7306,11 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6721
7306
  elif not ordered_subsequence(next_results, uses):
6722
7307
  tool_pair_errors.append(
6723
7308
  {
6724
- "line": line,
7309
+ "line": tool_use_line,
6725
7310
  "uuid": obj.get("uuid"),
6726
7311
  "reason": "assistant tool_use ids do not contain next user tool_result ids in order",
6727
7312
  "toolUseIds": uses,
6728
- "nextLine": next_line,
7313
+ "nextLine": _ordered_subsequence_failure_line(next_obj, uses, next_result_line),
6729
7314
  "nextToolResultIds": next_results,
6730
7315
  }
6731
7316
  )
@@ -6734,7 +7319,7 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6734
7319
  if results:
6735
7320
  if pos == 0:
6736
7321
  tool_pair_errors.append(
6737
- {"line": line, "uuid": obj.get("uuid"), "reason": "user tool_result is first api message", "toolResultIds": results}
7322
+ {"line": tool_result_line, "uuid": obj.get("uuid"), "reason": "user tool_result is first api message", "toolResultIds": results}
6738
7323
  )
6739
7324
  else:
6740
7325
  source_uuid = source_tool_assistant_uuid(obj)
@@ -6744,26 +7329,33 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
6744
7329
  source_uses = tool_use_ids(source_obj) if isinstance(source_obj, dict) else []
6745
7330
  prev_line, prev_obj = api_messages[pos - 1]
6746
7331
  prev_uses = tool_use_ids(prev_obj)
7332
+ prev_tool_use_line = _first_validation_block_line(prev_obj, "tool_use", prev_line)
6747
7333
  matches_prev = api_role(prev_obj) == "assistant" and ordered_subsequence(results, prev_uses)
6748
7334
  matches_source = ordered_subsequence(results, source_uses)
6749
7335
  if not matches_prev and not matches_source:
7336
+ expected_uses = source_uses or prev_uses
6750
7337
  tool_pair_errors.append(
6751
7338
  {
6752
- "line": line,
7339
+ "line": _ordered_subsequence_failure_line(obj, expected_uses, tool_result_line),
6753
7340
  "uuid": obj.get("uuid"),
6754
7341
  "reason": "user tool_result ids are not an ordered subset of previous/linked assistant tool_use ids",
6755
7342
  "toolResultIds": results,
6756
- "prevLine": prev_line,
7343
+ "prevLine": prev_tool_use_line,
6757
7344
  "prevToolUseIds": prev_uses,
6758
7345
  "sourceToolAssistantUUID": source_uuid,
6759
7346
  "sourceToolUseIds": source_uses,
6760
7347
  }
6761
7348
  )
6762
7349
  types = [block.get("type") if isinstance(block, dict) else type(block).__name__ for block in content_blocks(obj)]
6763
- if any(kind != "tool_result" for kind in types[: len(results)]):
7350
+ misplaced_index = next(
7351
+ (index for index, kind in enumerate(types[: len(results)]) if kind != "tool_result"),
7352
+ None,
7353
+ )
7354
+ if misplaced_index is not None:
7355
+ content_lines = _validation_content_lines(obj, line)
6764
7356
  tool_pair_errors.append(
6765
7357
  {
6766
- "line": line,
7358
+ "line": content_lines[misplaced_index],
6767
7359
  "uuid": obj.get("uuid"),
6768
7360
  "reason": "tool_result blocks are not first in user message content",
6769
7361
  "blockTypes": types[:10],
@@ -7029,7 +7621,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
7029
7621
  )
7030
7622
  active_chain_has_compact_summary = any(obj.get("isCompactSummary") is True for obj in active_chain)
7031
7623
  active_chain_uuid_set = {obj.get("uuid") for obj in active_chain if isinstance(obj.get("uuid"), str)}
7032
- active_chain_sessions = collections.Counter(obj.get("sessionId", "<missing>") for obj in active_chain)
7624
+ active_chain_sessions = collections.Counter(
7625
+ _diagnostic_counter_label(obj, "sessionId") for obj in active_chain
7626
+ )
7033
7627
  compact_metadata_chain_mismatch: List[JsonObj] = []
7034
7628
  for boundary in compact_boundaries:
7035
7629
  metadata = boundary.get("compactMetadata")
@@ -7143,9 +7737,9 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
7143
7737
  "compactMetadata historical preserved snapshot diverges from the current active chain: "
7144
7738
  f"{len(compact_metadata_chain_mismatch)} item(s)"
7145
7739
  )
7146
- type_counts = collections.Counter(obj.get("type", "<missing>") for obj in records)
7147
- session_counts = collections.Counter(obj.get("sessionId", "<missing>") for obj in records)
7148
- return {
7740
+ type_counts = collections.Counter(_diagnostic_counter_label(obj, "type") for obj in records)
7741
+ session_counts = collections.Counter(_diagnostic_counter_label(obj, "sessionId") for obj in records)
7742
+ validation = {
7149
7743
  "ok": not errors,
7150
7744
  "errors": errors,
7151
7745
  "warnings": warnings,
@@ -7200,10 +7794,22 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
7200
7794
  "compact_metadata_chain_mismatch_count": len(compact_metadata_chain_mismatch),
7201
7795
  "compact_metadata_chain_mismatch_samples": compact_metadata_chain_mismatch[:20],
7202
7796
  "latest_last_prompt_line": latest_last_prompt_line,
7203
- "latest_last_prompt_session_id": latest_last_prompt_session,
7204
- "latest_last_prompt_leaf_uuid": latest_last_prompt_leaf,
7797
+ "latest_last_prompt_session_id": (
7798
+ _bounded_diagnostic_label(latest_last_prompt_session)
7799
+ if latest_last_prompt_session is not None
7800
+ else None
7801
+ ),
7802
+ "latest_last_prompt_leaf_uuid": (
7803
+ _bounded_diagnostic_label(latest_last_prompt_leaf)
7804
+ if latest_last_prompt_leaf is not None
7805
+ else None
7806
+ ),
7205
7807
  "active_chain_length": len(active_chain),
7206
- "active_chain_missing_uuid": active_chain_missing_uuid,
7808
+ "active_chain_missing_uuid": (
7809
+ _bounded_diagnostic_label(active_chain_missing_uuid)
7810
+ if active_chain_missing_uuid is not None
7811
+ else None
7812
+ ),
7207
7813
  "active_chain_loop": active_chain_loop,
7208
7814
  "active_chain_min_line": min(active_chain_lines) if active_chain_lines else None,
7209
7815
  "active_chain_max_line": max(active_chain_lines) if active_chain_lines else None,
@@ -7216,6 +7822,10 @@ def validate_records(records: Sequence[JsonObj]) -> Dict[str, Any]:
7216
7822
  "type_counts": dict(type_counts.most_common()),
7217
7823
  "session_counts": dict(session_counts.most_common(20)),
7218
7824
  }
7825
+ return _bounded_public_diagnostic_structure(
7826
+ validation,
7827
+ redactions=_long_diagnostic_strings(records),
7828
+ )
7219
7829
 
7220
7830
 
7221
7831
  def validate_jsonl(path: pathlib.Path) -> Dict[str, Any]:
@@ -7227,8 +7837,8 @@ def validate_jsonl(path: pathlib.Path) -> Dict[str, Any]:
7227
7837
 
7228
7838
 
7229
7839
  def validate_jsonl_bytes(data: bytes, source_label: str = "JSONL") -> Dict[str, Any]:
7230
- records, _ = parse_jsonl_bytes(data, source_label=source_label)
7231
- result = validate_records(records)
7840
+ records, _raw_lines, physical_lines = parse_jsonl_bytes_with_lines(data, source_label=source_label)
7841
+ result = validate_records(records, physical_lines=physical_lines)
7232
7842
  result["path"] = pathlib.Path(source_label).name
7233
7843
  result["bytes"] = len(data)
7234
7844
  result["sha256"] = sha256_hex(data)
@@ -7247,6 +7857,7 @@ def write_sidecars(output_path: pathlib.Path, report: Dict[str, Any]) -> None:
7247
7857
  "",
7248
7858
  f"- Input: `{report['input']}`",
7249
7859
  f"- Output: `{report['output']}`",
7860
+ f"- Operation state: {report.get('operation_state') or 'candidate'}",
7250
7861
  f"- Package version: {report.get('package_version')}",
7251
7862
  f"- Compression engine: {report.get('codex_offline_compression_version')}",
7252
7863
  f"- Model-pack schema: {report.get('model_pack_schema_version')}",
@@ -7302,6 +7913,7 @@ def write_sidecars(output_path: pathlib.Path, report: Dict[str, Any]) -> None:
7302
7913
  f"- Replacement target: `{report.get('replacement_target') or ''}`",
7303
7914
  f"- Replacement backup: `{report.get('replacement_backup') or ''}`",
7304
7915
  f"- Replacement candidate: `{report.get('replacement_candidate') or ''}`",
7916
+ f"- Replacement cleanup errors: {json.dumps(report.get('replacement_cleanup_errors') or [], ensure_ascii=False)}",
7305
7917
  "",
7306
7918
  "## Validation",
7307
7919
  "",
@@ -7621,6 +8233,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
7621
8233
  report["replacement_candidate_sha256"] = replacement["candidate_sha256"]
7622
8234
  report["replacement_published_sha256"] = replacement["published_sha256"]
7623
8235
  report["replacement_parent_directory_fsync"] = replacement["parent_directory_fsync"]
8236
+ report["operation_state"] = replacement["operation_state"]
8237
+ report["replacement_cleanup_errors"] = replacement["cleanup_errors"]
7624
8238
  try:
7625
8239
  write_sidecars(output_path, report)
7626
8240
  except Exception as report_exc:
@@ -7633,6 +8247,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
7633
8247
  "candidate_sha256": replacement.get("candidate_sha256"),
7634
8248
  "published_sha256": replacement.get("published_sha256"),
7635
8249
  "replacement_validation_ok": bool(replaced_validation.get("ok")),
8250
+ "prior_operation_state": report.get("operation_state"),
8251
+ "replacement_cleanup_errors": report.get("replacement_cleanup_errors", []),
7636
8252
  "report_error": f"{type(report_exc).__name__}: {report_exc}",
7637
8253
  }
7638
8254
  print(json.dumps(receipt, ensure_ascii=False, indent=2))
@@ -7641,6 +8257,13 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
7641
8257
  f"{report_exc}"
7642
8258
  )
7643
8259
  return 3
8260
+ if report.get("operation_state") == "committed-cleanup-failed":
8261
+ print(json.dumps(report, ensure_ascii=False, indent=2))
8262
+ eprint(
8263
+ "ERROR: live replacement committed, but transaction cleanup failed; "
8264
+ "inspect replacement_cleanup_errors and remove only the listed residuals after verification."
8265
+ )
8266
+ return 3
7644
8267
  print(json.dumps(report, ensure_ascii=False, indent=2))
7645
8268
  return 0 if report["validation"].get("ok") else 2
7646
8269
  except Exception as exc: