@brandry/claude-jsonl-compressor 1.0.0-rc.1 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -172,9 +172,10 @@ def member_deletion_span(object_node: Node, target: Member) -> Tuple[int, int]:
172
172
  return object_node.members[index - 1].end, target.end
173
173
 
174
174
 
175
- def jsonl_record_spans(data: bytes) -> List[Tuple[int, int, bytes]]:
176
- spans: List[Tuple[int, int, bytes]] = []
175
+ def _jsonl_record_spans_with_lines(data: bytes) -> List[Tuple[int, int, int, bytes]]:
176
+ spans: List[Tuple[int, int, int, bytes]] = []
177
177
  offset = 0
178
+ physical_line = 1
178
179
  while offset < len(data):
179
180
  lf_index = data.find(b"\n", offset)
180
181
  physical_end = len(data) if lf_index < 0 else lf_index
@@ -183,14 +184,19 @@ def jsonl_record_spans(data: bytes) -> List[Tuple[int, int, bytes]]:
183
184
  if start == 0 and data.startswith(b"\xef\xbb\xbf"):
184
185
  start = 3
185
186
  content = data[start:content_end]
186
- if content.strip():
187
- spans.append((start, content_end, content))
187
+ if content.decode("utf-8", errors="strict").strip():
188
+ spans.append((physical_line, start, content_end, content))
188
189
  if lf_index < 0:
189
190
  break
190
191
  offset = lf_index + 1
192
+ physical_line += 1
191
193
  return spans
192
194
 
193
195
 
196
+ def jsonl_record_spans(data: bytes) -> List[Tuple[int, int, bytes]]:
197
+ return [(start, end, content) for _line, start, end, content in _jsonl_record_spans_with_lines(data)]
198
+
199
+
194
200
  def _node_at_tool_input(root: Node, block_index: int) -> Tuple[Node, Member]:
195
201
  message = unique_member(root, "message")
196
202
  if message is None or message.value.kind != "object":
@@ -215,12 +221,17 @@ def plan_read_pages_repairs(
215
221
  scope: str = "active-chain",
216
222
  resume_leaf_override: Optional[str] = None,
217
223
  ) -> Dict[str, Any]:
218
- records, _raw_lines = ccj.parse_jsonl_bytes(source_bytes, source_label="SOURCE_JSONL")
219
- line_spans = jsonl_record_spans(source_bytes)
224
+ records, _raw_lines, physical_lines = ccj.parse_jsonl_bytes_with_lines(
225
+ source_bytes,
226
+ source_label="SOURCE_JSONL",
227
+ )
228
+ line_spans = _jsonl_record_spans_with_lines(source_bytes)
220
229
  if len(line_spans) != len(records):
221
230
  raise ValueError("record/span count mismatch while planning byte repair")
231
+ if [line for line, _start, _end, _bytes in line_spans] != physical_lines:
232
+ raise ValueError("record/span physical line map mismatch while planning byte repair")
222
233
  parsed_roots: List[Node] = []
223
- for _line_start, _line_end, line_bytes in line_spans:
234
+ for _physical_line, _line_start, _line_end, line_bytes in line_spans:
224
235
  root = SpanJsonParser(line_bytes).parse()
225
236
  assert_no_duplicate_keys(root)
226
237
  parsed_roots.append(root)
@@ -251,7 +262,7 @@ def plan_read_pages_repairs(
251
262
  content = message.get("content") if isinstance(message, dict) else None
252
263
  if not isinstance(content, list):
253
264
  continue
254
- line_start, _line_end, line_bytes = line_spans[record_index]
265
+ physical_line, line_start, _line_end, line_bytes = line_spans[record_index]
255
266
  root_node = parsed_roots[record_index]
256
267
  for block_index, block in enumerate(content):
257
268
  if not isinstance(block, dict) or block.get("type") != "tool_use" or block.get("name") != "Read":
@@ -301,7 +312,7 @@ def plan_read_pages_repairs(
301
312
  else:
302
313
  reason = "missing-file-path"
303
314
  match = {
304
- "recordLine": record_index + 1,
315
+ "recordLine": physical_line,
305
316
  "blockIndex": block_index,
306
317
  "toolUseId": tool_id,
307
318
  "pairedToolResult": paired,
@@ -322,7 +333,7 @@ def plan_read_pages_repairs(
322
333
  {
323
334
  "start": line_start + rel_start,
324
335
  "end": line_start + rel_end,
325
- "recordLine": record_index + 1,
336
+ "recordLine": physical_line,
326
337
  "blockIndex": block_index,
327
338
  "toolUseId": tool_id,
328
339
  }
@@ -571,6 +582,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
571
582
  report["replacementCandidateSha256"] = replacement["candidate_sha256"]
572
583
  report["replacementPublishedSha256"] = replacement["published_sha256"]
573
584
  report["replacementParentDirectoryFsync"] = replacement["parent_directory_fsync"]
585
+ report["operationState"] = replacement["operation_state"]
586
+ report["replacementCleanupErrors"] = replacement["cleanup_errors"]
574
587
  report_path = output_path.with_suffix(output_path.suffix + ".repair.json")
575
588
  try:
576
589
  ccj.atomic_write_text(report_path, json.dumps(report, ensure_ascii=False, indent=2) + "\n")
@@ -585,6 +598,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
585
598
  "candidateSha256": report.get("replacementCandidateSha256"),
586
599
  "publishedSha256": report.get("replacementPublishedSha256"),
587
600
  "replacementValidationOk": bool((report.get("replacementValidation") or {}).get("ok")),
601
+ "priorOperationState": report.get("operationState"),
602
+ "replacementCleanupErrors": report.get("replacementCleanupErrors", []),
588
603
  "reportError": f"{type(report_exc).__name__}: {report_exc}",
589
604
  }
590
605
  print(json.dumps(receipt, ensure_ascii=False, indent=2))
@@ -594,6 +609,13 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
594
609
  )
595
610
  return 3
596
611
  raise
612
+ if report.get("operationState") == "committed-cleanup-failed":
613
+ print(json.dumps(report, ensure_ascii=False, indent=2))
614
+ ccj.eprint(
615
+ "ERROR: live repair committed, but transaction cleanup failed; "
616
+ "inspect replacementCleanupErrors and remove only the listed residuals after verification."
617
+ )
618
+ return 3
597
619
  print(json.dumps(report, ensure_ascii=False, indent=2))
598
620
  return 0
599
621
  except Exception as exc: