agent-session-otel 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_session_otel/__init__.py +10 -0
- agent_session_otel/adapters/__init__.py +12 -0
- agent_session_otel/adapters/base.py +128 -0
- agent_session_otel/adapters/claude_code.py +248 -0
- agent_session_otel/adapters/codex.py +417 -0
- agent_session_otel/cli.py +272 -0
- agent_session_otel/discovery.py +94 -0
- agent_session_otel/doctor.py +160 -0
- agent_session_otel/otel_export.py +594 -0
- agent_session_otel/redaction.py +268 -0
- agent_session_otel/schema.py +118 -0
- agent_session_otel-0.2.0.dist-info/METADATA +332 -0
- agent_session_otel-0.2.0.dist-info/RECORD +16 -0
- agent_session_otel-0.2.0.dist-info/WHEEL +4 -0
- agent_session_otel-0.2.0.dist-info/entry_points.txt +2 -0
- agent_session_otel-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""agent-session-otel: normalize local Claude Code / Codex session logs into
|
|
2
|
+
OpenTelemetry GenAI-compatible traces."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
__version__ = "0.1.0"
|
|
7
|
+
|
|
8
|
+
from .schema import EventType, NormalizedEvent
|
|
9
|
+
|
|
10
|
+
__all__ = ["__version__", "EventType", "NormalizedEvent"]
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from .base import SessionAdapter
|
|
4
|
+
from .claude_code import ClaudeCodeAdapter
|
|
5
|
+
from .codex import CodexAdapter
|
|
6
|
+
|
|
7
|
+
ADAPTERS = {
|
|
8
|
+
ClaudeCodeAdapter.vendor: ClaudeCodeAdapter(),
|
|
9
|
+
CodexAdapter.vendor: CodexAdapter(),
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
__all__ = ["SessionAdapter", "ClaudeCodeAdapter", "CodexAdapter", "ADAPTERS"]
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Adapter interface shared by all vendor-specific parsers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import abc
|
|
6
|
+
import json
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Iterator
|
|
9
|
+
|
|
10
|
+
from ..schema import EventType, NormalizedEvent
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class SessionAdapter(abc.ABC):
|
|
14
|
+
"""Parses one vendor's session JSONL format into NormalizedEvent objects.
|
|
15
|
+
|
|
16
|
+
Adapters must never raise on malformed input: a line that fails to
|
|
17
|
+
parse as JSON, or a record whose shape doesn't match what the adapter
|
|
18
|
+
expects, is turned into an ``EventType.UNKNOWN`` event carrying the
|
|
19
|
+
original text/record so nothing is silently dropped. One bad line does
|
|
20
|
+
not lose the rest of an otherwise-recoverable session, and a file that
|
|
21
|
+
can't be opened at all (permissions, mid-write on some platforms,
|
|
22
|
+
disappeared between discovery and parsing) becomes a single UNKNOWN
|
|
23
|
+
event instead of an unhandled exception.
|
|
24
|
+
|
|
25
|
+
Files are streamed line-by-line rather than loaded into memory. Real
|
|
26
|
+
session logs -- especially Codex rollouts after heavy compaction --
|
|
27
|
+
have been observed in the hundreds of MB to low GB range.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
vendor: str
|
|
31
|
+
|
|
32
|
+
@abc.abstractmethod
|
|
33
|
+
def discover(self, roots=None):
|
|
34
|
+
"""Return a list of session files for this vendor."""
|
|
35
|
+
|
|
36
|
+
def resolve_session_id(self, path: Path, first_line: str) -> str:
|
|
37
|
+
"""Determine the one session id used for every event parsed from
|
|
38
|
+
``path``, given only its first non-empty line. Default: the
|
|
39
|
+
filename stem. Adapters whose filenames don't reliably match the
|
|
40
|
+
vendor's own session/thread id (e.g. Codex's
|
|
41
|
+
``rollout-<timestamp>-<uuid>.jsonl``) should override this so that
|
|
42
|
+
every event from one file lands in the same exported trace.
|
|
43
|
+
|
|
44
|
+
Only the first line is available (not the whole file) so that
|
|
45
|
+
resolving a session id never requires a second full read of a
|
|
46
|
+
potentially multi-GB file.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
return path.stem
|
|
50
|
+
|
|
51
|
+
def parse_file(self, path: Path) -> Iterator[NormalizedEvent]:
|
|
52
|
+
path = Path(path)
|
|
53
|
+
try:
|
|
54
|
+
fh = path.open("r", encoding="utf-8", errors="replace")
|
|
55
|
+
except OSError as exc:
|
|
56
|
+
yield NormalizedEvent(
|
|
57
|
+
vendor=self.vendor,
|
|
58
|
+
event_type=EventType.UNKNOWN,
|
|
59
|
+
session_id=path.stem,
|
|
60
|
+
source_file=str(path),
|
|
61
|
+
raw_type="_unreadable_file",
|
|
62
|
+
extra={"error": str(exc)},
|
|
63
|
+
)
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
try:
|
|
67
|
+
first_line = ""
|
|
68
|
+
for probe in fh:
|
|
69
|
+
stripped = probe.strip()
|
|
70
|
+
if stripped:
|
|
71
|
+
first_line = stripped
|
|
72
|
+
break
|
|
73
|
+
# Always safe: seeking to 0 on a text file is well-defined even
|
|
74
|
+
# after partial iteration (unlike seeking to an arbitrary tell()
|
|
75
|
+
# offset captured mid-iteration, which text-mode files forbid).
|
|
76
|
+
fh.seek(0)
|
|
77
|
+
try:
|
|
78
|
+
session_id = self.resolve_session_id(path, first_line)
|
|
79
|
+
except Exception: # noqa: BLE001 - never let session-id resolution crash the parse
|
|
80
|
+
session_id = path.stem
|
|
81
|
+
if not session_id:
|
|
82
|
+
session_id = path.stem
|
|
83
|
+
|
|
84
|
+
for lineno, raw_line in enumerate(fh, start=1):
|
|
85
|
+
line = raw_line.strip()
|
|
86
|
+
if not line:
|
|
87
|
+
continue
|
|
88
|
+
try:
|
|
89
|
+
record = json.loads(line)
|
|
90
|
+
except json.JSONDecodeError as exc:
|
|
91
|
+
yield NormalizedEvent(
|
|
92
|
+
vendor=self.vendor,
|
|
93
|
+
event_type=EventType.UNKNOWN,
|
|
94
|
+
session_id=session_id,
|
|
95
|
+
source_file=str(path),
|
|
96
|
+
raw_type="_invalid_json",
|
|
97
|
+
line_number=lineno,
|
|
98
|
+
extra={"parse_error": str(exc)},
|
|
99
|
+
raw={"raw_line": line[:4000]},
|
|
100
|
+
)
|
|
101
|
+
continue
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
for sub_index, event in enumerate(
|
|
105
|
+
self.parse_record(record, path=path, session_id=session_id)
|
|
106
|
+
):
|
|
107
|
+
if event.line_number is None:
|
|
108
|
+
event.line_number = lineno
|
|
109
|
+
if event.line_sub_index is None:
|
|
110
|
+
event.line_sub_index = sub_index
|
|
111
|
+
yield event
|
|
112
|
+
except Exception as exc: # noqa: BLE001 - adapters must never crash the pipeline
|
|
113
|
+
yield NormalizedEvent(
|
|
114
|
+
vendor=self.vendor,
|
|
115
|
+
event_type=EventType.UNKNOWN,
|
|
116
|
+
session_id=session_id,
|
|
117
|
+
source_file=str(path),
|
|
118
|
+
raw_type="_adapter_error",
|
|
119
|
+
line_number=lineno,
|
|
120
|
+
extra={"error": str(exc)},
|
|
121
|
+
raw=record if isinstance(record, dict) else {"value": record},
|
|
122
|
+
)
|
|
123
|
+
finally:
|
|
124
|
+
fh.close()
|
|
125
|
+
|
|
126
|
+
@abc.abstractmethod
|
|
127
|
+
def parse_record(self, record: dict, *, path: Path, session_id: str) -> Iterator[NormalizedEvent]:
|
|
128
|
+
"""Yield zero or more NormalizedEvent for a single decoded JSON line."""
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Adapter for Claude Code session transcripts.
|
|
2
|
+
|
|
3
|
+
Claude Code writes one append-only JSONL file per session under
|
|
4
|
+
``~/.claude/projects/<encoded-cwd>/<session-id>.jsonl`` (subagent
|
|
5
|
+
transcripts are named ``agent-<id>.jsonl``). Each line has a ``type``
|
|
6
|
+
discriminator (``summary``, ``user``, ``assistant``, ``system``,
|
|
7
|
+
``file-history-snapshot``, ...). This module is intentionally defensive:
|
|
8
|
+
the format is not officially specified and has evolved over time, so
|
|
9
|
+
every lookup is best-effort and anything we don't recognize is preserved
|
|
10
|
+
via ``EventType.UNKNOWN`` rather than dropped.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Iterator, List
|
|
17
|
+
|
|
18
|
+
from .. import discovery
|
|
19
|
+
from ..schema import EventType, NormalizedEvent, normalize_usage
|
|
20
|
+
from .base import SessionAdapter
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ClaudeCodeAdapter(SessionAdapter):
|
|
24
|
+
vendor = "claude_code"
|
|
25
|
+
|
|
26
|
+
def discover(self, roots=None) -> List[Path]:
|
|
27
|
+
return discovery.discover_claude_code_files(roots)
|
|
28
|
+
|
|
29
|
+
def parse_record(self, record: dict, *, path: Path, session_id: str) -> Iterator[NormalizedEvent]:
|
|
30
|
+
if not isinstance(record, dict):
|
|
31
|
+
yield NormalizedEvent(
|
|
32
|
+
vendor=self.vendor,
|
|
33
|
+
event_type=EventType.UNKNOWN,
|
|
34
|
+
session_id=session_id,
|
|
35
|
+
source_file=str(path),
|
|
36
|
+
raw_type="_non_object_record",
|
|
37
|
+
raw={"value": record},
|
|
38
|
+
)
|
|
39
|
+
return
|
|
40
|
+
|
|
41
|
+
session_id = record.get("sessionId") or session_id
|
|
42
|
+
rtype = record.get("type")
|
|
43
|
+
common = dict(
|
|
44
|
+
vendor=self.vendor,
|
|
45
|
+
session_id=session_id,
|
|
46
|
+
source_file=str(path),
|
|
47
|
+
timestamp=record.get("timestamp"),
|
|
48
|
+
event_id=record.get("uuid"),
|
|
49
|
+
parent_id=record.get("parentUuid"),
|
|
50
|
+
cwd=record.get("cwd"),
|
|
51
|
+
raw_type=rtype,
|
|
52
|
+
extra={
|
|
53
|
+
k: record.get(k)
|
|
54
|
+
for k in ("gitBranch", "version", "isSidechain", "userType")
|
|
55
|
+
if record.get(k) is not None
|
|
56
|
+
},
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
if rtype == "summary":
|
|
60
|
+
yield NormalizedEvent(
|
|
61
|
+
event_type=EventType.SESSION,
|
|
62
|
+
content=record.get("summary"),
|
|
63
|
+
extra={**common["extra"], "kind": "summary", "leaf_uuid": record.get("leafUuid")},
|
|
64
|
+
raw=record,
|
|
65
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
66
|
+
)
|
|
67
|
+
return
|
|
68
|
+
|
|
69
|
+
if rtype == "file-history-snapshot":
|
|
70
|
+
yield NormalizedEvent(
|
|
71
|
+
event_type=EventType.SESSION,
|
|
72
|
+
extra={**common["extra"], "kind": "file_history_snapshot"},
|
|
73
|
+
raw=record,
|
|
74
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
75
|
+
)
|
|
76
|
+
return
|
|
77
|
+
|
|
78
|
+
if rtype == "system":
|
|
79
|
+
level = record.get("level")
|
|
80
|
+
if level == "error":
|
|
81
|
+
yield NormalizedEvent(
|
|
82
|
+
event_type=EventType.ERROR,
|
|
83
|
+
error_message=record.get("content"),
|
|
84
|
+
extra={**common["extra"], "subtype": record.get("subtype")},
|
|
85
|
+
raw=record,
|
|
86
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
87
|
+
)
|
|
88
|
+
else:
|
|
89
|
+
yield NormalizedEvent(
|
|
90
|
+
event_type=EventType.SESSION,
|
|
91
|
+
content=record.get("content") if isinstance(record.get("content"), str) else None,
|
|
92
|
+
extra={**common["extra"], "kind": "system", "subtype": record.get("subtype"), "level": level},
|
|
93
|
+
raw=record,
|
|
94
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
95
|
+
)
|
|
96
|
+
return
|
|
97
|
+
|
|
98
|
+
if rtype == "user":
|
|
99
|
+
yield from self._parse_user(record, common)
|
|
100
|
+
return
|
|
101
|
+
|
|
102
|
+
if rtype == "assistant":
|
|
103
|
+
yield from self._parse_assistant(record, common)
|
|
104
|
+
return
|
|
105
|
+
|
|
106
|
+
# Unrecognized top-level type: preserve, don't drop.
|
|
107
|
+
yield NormalizedEvent(
|
|
108
|
+
event_type=EventType.UNKNOWN,
|
|
109
|
+
raw=record,
|
|
110
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
111
|
+
extra=common["extra"],
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
def _parse_user(self, record: dict, common: dict) -> Iterator[NormalizedEvent]:
|
|
115
|
+
message = record.get("message") or {}
|
|
116
|
+
content = message.get("content")
|
|
117
|
+
role = message.get("role", "user")
|
|
118
|
+
tool_use_result = record.get("toolUseResult")
|
|
119
|
+
|
|
120
|
+
if isinstance(content, str):
|
|
121
|
+
yield NormalizedEvent(
|
|
122
|
+
event_type=EventType.TURN,
|
|
123
|
+
role=role,
|
|
124
|
+
content=content,
|
|
125
|
+
raw=record,
|
|
126
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
127
|
+
extra=common["extra"],
|
|
128
|
+
)
|
|
129
|
+
return
|
|
130
|
+
|
|
131
|
+
if isinstance(content, list):
|
|
132
|
+
emitted = False
|
|
133
|
+
for block in content:
|
|
134
|
+
if not isinstance(block, dict):
|
|
135
|
+
continue
|
|
136
|
+
btype = block.get("type")
|
|
137
|
+
if btype == "tool_result":
|
|
138
|
+
emitted = True
|
|
139
|
+
output = block.get("content")
|
|
140
|
+
yield NormalizedEvent(
|
|
141
|
+
event_type=EventType.TOOL,
|
|
142
|
+
tool_phase="result",
|
|
143
|
+
tool_call_id=block.get("tool_use_id"),
|
|
144
|
+
tool_output=output if output is not None else tool_use_result,
|
|
145
|
+
raw=record,
|
|
146
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
147
|
+
extra={**common["extra"], "is_error": block.get("is_error", False)},
|
|
148
|
+
)
|
|
149
|
+
elif btype == "text" and isinstance(block.get("text"), str):
|
|
150
|
+
emitted = True
|
|
151
|
+
yield NormalizedEvent(
|
|
152
|
+
event_type=EventType.TURN,
|
|
153
|
+
role=role,
|
|
154
|
+
content=block["text"],
|
|
155
|
+
raw=record,
|
|
156
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
157
|
+
extra=common["extra"],
|
|
158
|
+
)
|
|
159
|
+
if not emitted:
|
|
160
|
+
yield NormalizedEvent(
|
|
161
|
+
event_type=EventType.TURN,
|
|
162
|
+
role=role,
|
|
163
|
+
raw=record,
|
|
164
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
165
|
+
extra=common["extra"],
|
|
166
|
+
)
|
|
167
|
+
return
|
|
168
|
+
|
|
169
|
+
# No recognizable message content at all -- still preserve the line.
|
|
170
|
+
yield NormalizedEvent(
|
|
171
|
+
event_type=EventType.TURN,
|
|
172
|
+
role=role,
|
|
173
|
+
raw=record,
|
|
174
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
175
|
+
extra=common["extra"],
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
def _parse_assistant(self, record: dict, common: dict) -> Iterator[NormalizedEvent]:
|
|
179
|
+
message = record.get("message") or {}
|
|
180
|
+
role = message.get("role", "assistant")
|
|
181
|
+
model = message.get("model")
|
|
182
|
+
content = message.get("content")
|
|
183
|
+
stop_reason = message.get("stop_reason")
|
|
184
|
+
extra = {**common["extra"], "stop_reason": stop_reason}
|
|
185
|
+
|
|
186
|
+
if isinstance(content, list):
|
|
187
|
+
for block in content:
|
|
188
|
+
if not isinstance(block, dict):
|
|
189
|
+
continue
|
|
190
|
+
btype = block.get("type")
|
|
191
|
+
if btype == "text" and isinstance(block.get("text"), str):
|
|
192
|
+
yield NormalizedEvent(
|
|
193
|
+
event_type=EventType.TURN,
|
|
194
|
+
role=role,
|
|
195
|
+
model=model,
|
|
196
|
+
content=block["text"],
|
|
197
|
+
raw=record,
|
|
198
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
199
|
+
extra=extra,
|
|
200
|
+
)
|
|
201
|
+
elif btype == "thinking" and isinstance(block.get("thinking"), str):
|
|
202
|
+
yield NormalizedEvent(
|
|
203
|
+
event_type=EventType.TURN,
|
|
204
|
+
role=role,
|
|
205
|
+
model=model,
|
|
206
|
+
content=block["thinking"],
|
|
207
|
+
raw=record,
|
|
208
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
209
|
+
extra={**extra, "phase": "thinking"},
|
|
210
|
+
)
|
|
211
|
+
elif btype == "tool_use":
|
|
212
|
+
yield NormalizedEvent(
|
|
213
|
+
event_type=EventType.TOOL,
|
|
214
|
+
tool_phase="call",
|
|
215
|
+
tool_name=block.get("name"),
|
|
216
|
+
tool_call_id=block.get("id"),
|
|
217
|
+
tool_input=block.get("input"),
|
|
218
|
+
model=model,
|
|
219
|
+
raw=record,
|
|
220
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
221
|
+
extra=extra,
|
|
222
|
+
)
|
|
223
|
+
elif isinstance(content, str):
|
|
224
|
+
yield NormalizedEvent(
|
|
225
|
+
event_type=EventType.TURN,
|
|
226
|
+
role=role,
|
|
227
|
+
model=model,
|
|
228
|
+
content=content,
|
|
229
|
+
raw=record,
|
|
230
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
231
|
+
extra=extra,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
usage = message.get("usage")
|
|
235
|
+
if isinstance(usage, dict):
|
|
236
|
+
yield NormalizedEvent(
|
|
237
|
+
event_type=EventType.USAGE,
|
|
238
|
+
model=model,
|
|
239
|
+
usage=normalize_usage(
|
|
240
|
+
input_tokens=usage.get("input_tokens"),
|
|
241
|
+
output_tokens=usage.get("output_tokens"),
|
|
242
|
+
cache_read_input_tokens=usage.get("cache_read_input_tokens"),
|
|
243
|
+
cache_creation_input_tokens=usage.get("cache_creation_input_tokens"),
|
|
244
|
+
),
|
|
245
|
+
raw=record,
|
|
246
|
+
**{k: v for k, v in common.items() if k != "extra"},
|
|
247
|
+
extra={**extra, "service_tier": usage.get("service_tier")},
|
|
248
|
+
)
|