session-preserve 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_preserve/__init__.py +12 -0
- codex_preserve/__main__.py +8 -0
- codex_preserve/_claude_payload.py +146 -0
- codex_preserve/_claude_readable.py +203 -0
- codex_preserve/_claude_source.py +507 -0
- codex_preserve/_claude_v3.py +82 -0
- codex_preserve/_codex_v3.py +142 -0
- codex_preserve/_kimi_source.py +371 -0
- codex_preserve/_kimi_v3.py +84 -0
- codex_preserve/_shared_core.py +77 -0
- codex_preserve/_v3_package.py +523 -0
- codex_preserve/_zcode_source.py +380 -0
- codex_preserve/_zcode_v3.py +55 -0
- codex_preserve/cli.py +341 -0
- codex_preserve/exporter.py +6295 -0
- codex_preserve/verify.py +512 -0
- session_preserve-0.2.0.dist-info/METADATA +268 -0
- session_preserve-0.2.0.dist-info/RECORD +22 -0
- session_preserve-0.2.0.dist-info/WHEEL +5 -0
- session_preserve-0.2.0.dist-info/entry_points.txt +2 -0
- session_preserve-0.2.0.dist-info/licenses/LICENSE +202 -0
- session_preserve-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Session Preserve: durable local coding-agent session preservation.
|
|
2
|
+
|
|
3
|
+
The public distribution and console command are named session-preserve.
|
|
4
|
+
The internal Python package keeps its historical codex_preserve module name for
|
|
5
|
+
the v0.2.0 transition while schema-2 compatibility remains frozen.
|
|
6
|
+
|
|
7
|
+
This project is independent and unofficial.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
__all__ = ["__version__"]
|
|
11
|
+
|
|
12
|
+
__version__ = "0.2.0"
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Internal Claude payload candidate over persisted, G4a/G4b-safe facts.
|
|
2
|
+
|
|
3
|
+
This is not a package schema or a claim about the current Claude UI. G4a's
|
|
4
|
+
``COMPLETE`` classifies the selected persisted source under its parser contract;
|
|
5
|
+
it does not attest that all visible replies were flushed or that a session ended.
|
|
6
|
+
No source path or raw JSONL is accepted or read here.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
from dataclasses import asdict, dataclass, field
|
|
13
|
+
from typing import Optional, Tuple
|
|
14
|
+
|
|
15
|
+
from ._claude_readable import (
|
|
16
|
+
ReadableBranchPoint, ReadableDiagnosticSummary, ReadableEvent,
|
|
17
|
+
ReadableGraph, ReadableLeafHint, ReadableNode, build_readable_graph,
|
|
18
|
+
)
|
|
19
|
+
from ._claude_source import ParseResult
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class ClaudeProvenance:
|
|
24
|
+
"""Content identity of a stable selected top-level JSONL snapshot, not authenticity."""
|
|
25
|
+
|
|
26
|
+
source_sha256: Optional[str]
|
|
27
|
+
source_size_bytes: Optional[int]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class ClaudeCoverage:
|
|
32
|
+
"""Separate source stability, parser classification, persisted text, and UI claims."""
|
|
33
|
+
|
|
34
|
+
source_stable: bool
|
|
35
|
+
parser_source_classification: str
|
|
36
|
+
source_entrypoint: str
|
|
37
|
+
readable_node_count: int
|
|
38
|
+
rendered_text_block_count: int
|
|
39
|
+
branch_count: int
|
|
40
|
+
graph_gap_count: int
|
|
41
|
+
unknown_count: int
|
|
42
|
+
duplicate_uuid_count: int
|
|
43
|
+
diagnostic_counts_by_code: Tuple[Tuple[str, int], ...]
|
|
44
|
+
source_scope: str = field(init=False, default="selected_top_level_jsonl")
|
|
45
|
+
sidecar_bodies_included: bool = field(init=False, default=False)
|
|
46
|
+
ui_completeness_attested: bool = field(init=False, default=False)
|
|
47
|
+
session_terminal_attested: bool = field(init=False, default=False)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class ClaudeGraphGap:
|
|
52
|
+
record_id: str
|
|
53
|
+
source_line: int
|
|
54
|
+
node_id: Optional[str]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True)
|
|
58
|
+
class ClaudeDuplicateOccurrence:
|
|
59
|
+
node_id: str
|
|
60
|
+
record_ids: Tuple[str, ...]
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass(frozen=True)
|
|
64
|
+
class ClaudeProviderPayload:
|
|
65
|
+
"""Loss-averse internal candidate; occurrence order is persisted source order."""
|
|
66
|
+
|
|
67
|
+
readable_nodes: Tuple[ReadableNode, ...]
|
|
68
|
+
events: Tuple[ReadableEvent, ...]
|
|
69
|
+
branch_points: Tuple[ReadableBranchPoint, ...]
|
|
70
|
+
leaf_hints: Tuple[ReadableLeafHint, ...]
|
|
71
|
+
graph_gaps: Tuple[ClaudeGraphGap, ...]
|
|
72
|
+
duplicate_occurrences: Tuple[ClaudeDuplicateOccurrence, ...]
|
|
73
|
+
diagnostics: ReadableDiagnosticSummary
|
|
74
|
+
coverage: ClaudeCoverage
|
|
75
|
+
provenance: ClaudeProvenance
|
|
76
|
+
presentation_order: str
|
|
77
|
+
format_status: str = field(init=False, default="INTERNAL_CANDIDATE")
|
|
78
|
+
|
|
79
|
+
def as_dict(self) -> dict:
|
|
80
|
+
"""Deterministic safe candidate shape; G5 may replace it without migration."""
|
|
81
|
+
def json_shape(value):
|
|
82
|
+
if isinstance(value, dict):
|
|
83
|
+
return {key: json_shape(item) for key, item in value.items()}
|
|
84
|
+
if isinstance(value, tuple):
|
|
85
|
+
return [json_shape(item) for item in value]
|
|
86
|
+
return value
|
|
87
|
+
|
|
88
|
+
return json_shape(asdict(self))
|
|
89
|
+
|
|
90
|
+
def candidate_json_bytes(self) -> bytes:
|
|
91
|
+
"""Deterministic internal bytes, with no promised package member name."""
|
|
92
|
+
return (json.dumps(self.as_dict(), ensure_ascii=False, sort_keys=True,
|
|
93
|
+
separators=(",", ":")) + "\n").encode("utf-8")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def build_claude_payload(graph: ReadableGraph,
|
|
97
|
+
source: ParseResult) -> ClaudeProviderPayload:
|
|
98
|
+
"""Project G4b graph plus only G4a's attested snapshot facts.
|
|
99
|
+
|
|
100
|
+
The caller supplies the G4a result used to build ``graph``. This function
|
|
101
|
+
validates the graph against that already-safe ParseResult so snapshot
|
|
102
|
+
provenance cannot be paired with content from another source. It never
|
|
103
|
+
reopens a source path or reads raw JSONL bytes.
|
|
104
|
+
"""
|
|
105
|
+
if not isinstance(graph, ReadableGraph) or not isinstance(source, ParseResult):
|
|
106
|
+
raise TypeError("expected Claude ReadableGraph and ParseResult")
|
|
107
|
+
expected_graph = build_readable_graph(source)
|
|
108
|
+
if graph != expected_graph:
|
|
109
|
+
raise ValueError("graph does not match the supplied source ParseResult")
|
|
110
|
+
if source.source_stable:
|
|
111
|
+
if source.source_sha256 is None or source.source_size_bytes is None:
|
|
112
|
+
raise ValueError("stable source lacks snapshot identity")
|
|
113
|
+
elif source.source_sha256 is not None or source.source_size_bytes is not None:
|
|
114
|
+
raise ValueError("unstable source cannot attest snapshot identity")
|
|
115
|
+
|
|
116
|
+
occurrences = {}
|
|
117
|
+
for node in graph.nodes:
|
|
118
|
+
if node.node_id is not None:
|
|
119
|
+
occurrences.setdefault(node.node_id, []).append(node.record_id)
|
|
120
|
+
duplicate_occurrences = tuple(
|
|
121
|
+
ClaudeDuplicateOccurrence(node_id, tuple(record_ids))
|
|
122
|
+
for node_id, record_ids in occurrences.items() if len(record_ids) > 1
|
|
123
|
+
)
|
|
124
|
+
graph_gaps = tuple(
|
|
125
|
+
ClaudeGraphGap(node.record_id, node.source_line, node.node_id)
|
|
126
|
+
for node in graph.nodes if node.parent_link == "missing"
|
|
127
|
+
)
|
|
128
|
+
diagnostic_counts = dict(graph.diagnostics.counts_by_code)
|
|
129
|
+
coverage = ClaudeCoverage(
|
|
130
|
+
source_stable=source.source_stable,
|
|
131
|
+
parser_source_classification=graph.source_completeness,
|
|
132
|
+
source_entrypoint=graph.source_entrypoint,
|
|
133
|
+
readable_node_count=graph.readable_node_count,
|
|
134
|
+
rendered_text_block_count=graph.rendered_text_block_count,
|
|
135
|
+
branch_count=graph.branch_count,
|
|
136
|
+
graph_gap_count=graph.graph_gap_count,
|
|
137
|
+
unknown_count=graph.unknown_count,
|
|
138
|
+
duplicate_uuid_count=diagnostic_counts.get("DUPLICATE_UUID", 0),
|
|
139
|
+
diagnostic_counts_by_code=graph.diagnostics.counts_by_code,
|
|
140
|
+
)
|
|
141
|
+
return ClaudeProviderPayload(
|
|
142
|
+
graph.nodes, graph.events, graph.branch_points, graph.leaf_hints,
|
|
143
|
+
graph_gaps, duplicate_occurrences, graph.diagnostics, coverage,
|
|
144
|
+
ClaudeProvenance(source.source_sha256, source.source_size_bytes),
|
|
145
|
+
graph.presentation_order,
|
|
146
|
+
)
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
"""Internal, loss-averse readable projection of a Claude ParseResult.
|
|
2
|
+
|
|
3
|
+
This module accepts only G4a's already classified data. Source line order is
|
|
4
|
+
the deterministic presentation order, not an active conversation path. No
|
|
5
|
+
leaf hint, branch, or graph gap removes persisted safe text.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections import Counter, defaultdict
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Optional, Tuple
|
|
13
|
+
|
|
14
|
+
from ._claude_source import (
|
|
15
|
+
KNOWN_IGNORED_OR_SUMMARIZED, RENDER, UNKNOWN, ParseResult,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class ReadableBlock:
|
|
21
|
+
index: int
|
|
22
|
+
semantic_kind: Optional[str]
|
|
23
|
+
text: str
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class ReadableNode:
|
|
28
|
+
# A record occurrence has its own identity even if a graph UUID repeats.
|
|
29
|
+
record_id: str
|
|
30
|
+
source_line: int
|
|
31
|
+
node_id: Optional[str]
|
|
32
|
+
parent_id: Optional[str]
|
|
33
|
+
parent_record_id: Optional[str]
|
|
34
|
+
parent_link: Optional[str]
|
|
35
|
+
sidechain: Optional[bool]
|
|
36
|
+
role: Optional[str]
|
|
37
|
+
semantic_kind: Optional[str]
|
|
38
|
+
source_policy: str
|
|
39
|
+
text_blocks: Tuple[ReadableBlock, ...]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class ReadableEvent:
|
|
44
|
+
record_id: str
|
|
45
|
+
source_line: int
|
|
46
|
+
kind: str
|
|
47
|
+
block_index: Optional[int] = None
|
|
48
|
+
tool_name: Optional[str] = None
|
|
49
|
+
tool_ref: Optional[str] = None
|
|
50
|
+
result_present: Optional[bool] = None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class ReadableLeafHint:
|
|
55
|
+
record_id: str
|
|
56
|
+
source_line: int
|
|
57
|
+
leaf_hint_id: Optional[str]
|
|
58
|
+
leaf_hint_explicit: Optional[bool]
|
|
59
|
+
leaf_hint_rewound: Optional[bool]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True)
|
|
63
|
+
class ReadableBranchPoint:
|
|
64
|
+
parent_node_id: str
|
|
65
|
+
# None means the normalized node ID has multiple persisted occurrences.
|
|
66
|
+
parent_record_id: Optional[str]
|
|
67
|
+
child_record_ids: Tuple[str, ...]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True)
|
|
71
|
+
class ReadableDiagnosticSummary:
|
|
72
|
+
# Codes and coordinates come from G4a, never from raw source values.
|
|
73
|
+
counts_by_code: Tuple[Tuple[str, int], ...]
|
|
74
|
+
graph_gap_count: int
|
|
75
|
+
unknown_count: int # unknown blocks plus unknown records without such blocks
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@dataclass(frozen=True)
|
|
79
|
+
class ReadableGraph:
|
|
80
|
+
source_completeness: str
|
|
81
|
+
source_entrypoint: str
|
|
82
|
+
nodes: Tuple[ReadableNode, ...]
|
|
83
|
+
events: Tuple[ReadableEvent, ...]
|
|
84
|
+
leaf_hints: Tuple[ReadableLeafHint, ...]
|
|
85
|
+
branch_points: Tuple[ReadableBranchPoint, ...]
|
|
86
|
+
diagnostics: ReadableDiagnosticSummary
|
|
87
|
+
readable_node_count: int
|
|
88
|
+
rendered_text_block_count: int
|
|
89
|
+
branch_count: int
|
|
90
|
+
graph_gap_count: int
|
|
91
|
+
unknown_count: int
|
|
92
|
+
active_head_id: None = None
|
|
93
|
+
presentation_order: str = "persisted_source_line"
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def build_readable_graph(source: ParseResult) -> ReadableGraph:
|
|
97
|
+
"""Preserve every G4a-approved text block, without choosing a path."""
|
|
98
|
+
if not isinstance(source, ParseResult):
|
|
99
|
+
raise TypeError("source must be a Claude ParseResult")
|
|
100
|
+
|
|
101
|
+
# G4a records are in physical source order. The ordinal identifies an
|
|
102
|
+
# occurrence independently of its (possibly duplicated) normalized UUID.
|
|
103
|
+
ordered = sorted(enumerate(source.records, 1),
|
|
104
|
+
key=lambda pair: (pair[1].line, pair[0]))
|
|
105
|
+
record_ids = {ordinal: "record-%06d" % ordinal for ordinal, _ in ordered}
|
|
106
|
+
occurrences = defaultdict(list)
|
|
107
|
+
for ordinal, record in ordered:
|
|
108
|
+
if record.node_id is not None:
|
|
109
|
+
occurrences[record.node_id].append(record_ids[ordinal])
|
|
110
|
+
|
|
111
|
+
nodes = []
|
|
112
|
+
events = []
|
|
113
|
+
hints = []
|
|
114
|
+
children = defaultdict(lambda: defaultdict(list))
|
|
115
|
+
for ordinal, record in ordered:
|
|
116
|
+
record_id = record_ids[ordinal]
|
|
117
|
+
text_blocks = tuple(
|
|
118
|
+
ReadableBlock(block.index, block.semantic_kind, block.text)
|
|
119
|
+
for block in sorted(record.blocks, key=lambda item: item.index)
|
|
120
|
+
if block.policy == RENDER and block.text is not None
|
|
121
|
+
)
|
|
122
|
+
if record.kind in ("user", "assistant", "attachment", "system") or \
|
|
123
|
+
record.node_id is not None or text_blocks:
|
|
124
|
+
parent_occurrences = occurrences.get(record.parent_id, ())
|
|
125
|
+
parent_record_id = (parent_occurrences[0]
|
|
126
|
+
if record.parent_link == "linked" and
|
|
127
|
+
len(parent_occurrences) == 1 else None)
|
|
128
|
+
nodes.append(ReadableNode(
|
|
129
|
+
record_id, record.line, record.node_id, record.parent_id,
|
|
130
|
+
parent_record_id, record.parent_link, record.sidechain,
|
|
131
|
+
record.kind if record.kind in ("user", "assistant") else None,
|
|
132
|
+
record.semantic_kind, record.policy, text_blocks,
|
|
133
|
+
))
|
|
134
|
+
if (record.parent_link == "linked" and
|
|
135
|
+
record.parent_id is not None and record.node_id is not None):
|
|
136
|
+
children[record.parent_id][record.node_id].append(record_id)
|
|
137
|
+
|
|
138
|
+
if record.kind == "last-prompt":
|
|
139
|
+
hints.append(ReadableLeafHint(
|
|
140
|
+
record_id, record.line, record.leaf_hint_id,
|
|
141
|
+
record.leaf_hint_explicit, record.leaf_hint_rewound,
|
|
142
|
+
))
|
|
143
|
+
if record.semantic_kind == "compact_boundary" and \
|
|
144
|
+
record.policy == KNOWN_IGNORED_OR_SUMMARIZED:
|
|
145
|
+
events.append(ReadableEvent(record_id, record.line, "compact_boundary"))
|
|
146
|
+
for block in sorted(record.blocks, key=lambda item: item.index):
|
|
147
|
+
if block.policy != KNOWN_IGNORED_OR_SUMMARIZED:
|
|
148
|
+
continue
|
|
149
|
+
if block.kind == "tool_use":
|
|
150
|
+
events.append(ReadableEvent(
|
|
151
|
+
record_id, record.line, "tool_use", block.index,
|
|
152
|
+
tool_name=block.tool_name, tool_ref=block.tool_ref,
|
|
153
|
+
))
|
|
154
|
+
elif block.kind == "tool_result":
|
|
155
|
+
events.append(ReadableEvent(
|
|
156
|
+
record_id, record.line, "tool_result_present", block.index,
|
|
157
|
+
tool_ref=block.tool_ref,
|
|
158
|
+
result_present=block.result_present,
|
|
159
|
+
))
|
|
160
|
+
|
|
161
|
+
source_line_by_record_id = {node.record_id: node.source_line for node in nodes}
|
|
162
|
+
branch_rows = []
|
|
163
|
+
for parent_id, child_groups in children.items():
|
|
164
|
+
if len(child_groups) <= 1:
|
|
165
|
+
continue
|
|
166
|
+
distinct_children = sorted(
|
|
167
|
+
child_groups.items(),
|
|
168
|
+
key=lambda pair: (
|
|
169
|
+
source_line_by_record_id[pair[1][0]], pair[0]
|
|
170
|
+
),
|
|
171
|
+
)
|
|
172
|
+
branch_rows.append((
|
|
173
|
+
min(source_line_by_record_id[group[0]]
|
|
174
|
+
for _, group in distinct_children),
|
|
175
|
+
parent_id,
|
|
176
|
+
tuple(group[0] for _, group in distinct_children),
|
|
177
|
+
))
|
|
178
|
+
branches = tuple(
|
|
179
|
+
ReadableBranchPoint(
|
|
180
|
+
parent_id,
|
|
181
|
+
occurrences[parent_id][0]
|
|
182
|
+
if len(occurrences[parent_id]) == 1 else None,
|
|
183
|
+
child_record_ids,
|
|
184
|
+
)
|
|
185
|
+
for _, parent_id, child_record_ids in sorted(branch_rows)
|
|
186
|
+
)
|
|
187
|
+
counts = Counter(item.code for item in source.diagnostics)
|
|
188
|
+
gap_count = counts["GRAPH_LINK_GAP"]
|
|
189
|
+
unknown_count = sum(
|
|
190
|
+
sum(block.policy == UNKNOWN for block in record.blocks) or
|
|
191
|
+
(record.policy == UNKNOWN)
|
|
192
|
+
for record in source.records
|
|
193
|
+
)
|
|
194
|
+
summary = ReadableDiagnosticSummary(
|
|
195
|
+
tuple(sorted(counts.items())), gap_count, unknown_count,
|
|
196
|
+
)
|
|
197
|
+
return ReadableGraph(
|
|
198
|
+
source.completeness, source.entrypoint, tuple(nodes), tuple(events),
|
|
199
|
+
tuple(hints), branches, summary,
|
|
200
|
+
sum(bool(node.text_blocks) for node in nodes),
|
|
201
|
+
sum(len(node.text_blocks) for node in nodes),
|
|
202
|
+
source.branch_count, gap_count, unknown_count,
|
|
203
|
+
)
|