codex-transcript-viewer 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1204 @@
1
+ """Parse Codex CLI JSONL session transcripts into structured events."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ast
6
+ import json
7
+ import re
8
+ from collections import Counter
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+
13
+ def _as_text(value: Any) -> str:
14
+ """Normalize possibly-null payload fields to text."""
15
+ if value is None:
16
+ return ""
17
+ if isinstance(value, str):
18
+ return value
19
+ if isinstance(value, (dict, list)):
20
+ try:
21
+ return json.dumps(value, ensure_ascii=False)
22
+ except TypeError:
23
+ return ""
24
+ if isinstance(value, (int, float, bool)):
25
+ return str(value)
26
+ return ""
27
+
28
+
29
+ def parse_jsonl(path: str | Path) -> list[dict]:
30
+ """Read a JSONL file and return a list of parsed JSON objects."""
31
+ entries = []
32
+ with open(path, encoding="utf-8") as f:
33
+ for line in f:
34
+ line = line.strip()
35
+ if not line:
36
+ continue
37
+ try:
38
+ entries.append(json.loads(line))
39
+ except json.JSONDecodeError:
40
+ continue
41
+ return entries
42
+
43
+
44
+ def _has_positive_usage(total: dict[str, Any]) -> bool:
45
+ """Return True when total token usage contains any positive numeric value."""
46
+ return any(
47
+ isinstance(value, (int, float)) and value > 0
48
+ for value in total.values()
49
+ )
50
+
51
+
52
+ def extract_conversation(
53
+ entries: list[dict],
54
+ ) -> tuple[dict | None, list[dict]]:
55
+ """Extract session metadata and meaningful conversation events.
56
+
57
+ Returns (meta, events) where meta is the session_meta payload and events
58
+ is a flat list of typed dicts representing user messages, assistant
59
+ responses, tool calls, reasoning blocks, and system events.
60
+ """
61
+ raw_events: list[dict] = []
62
+ meta: dict | None = None
63
+ turn_seq = 0
64
+ inherited_turns: set[int] = set()
65
+
66
+ for entry in entries:
67
+ ts = entry.get("timestamp", "")
68
+ etype = entry.get("type", "")
69
+ payload = entry.get("payload") or {}
70
+
71
+ if etype == "session_meta":
72
+ # A forked subagent log also carries its parent's session_meta;
73
+ # the first record describes this session.
74
+ if meta is None:
75
+ meta = payload
76
+ continue
77
+
78
+ if etype == "event_msg":
79
+ if payload.get("type", "") == "task_started":
80
+ turn_seq += 1
81
+ if _is_inherited_turn(meta, payload):
82
+ inherited_turns.add(turn_seq)
83
+ _handle_event_msg(payload, ts, raw_events, turn_seq)
84
+ continue
85
+
86
+ if etype == "response_item":
87
+ _handle_response_item(payload, ts, raw_events, turn_seq)
88
+ continue
89
+
90
+ raw_events = _attach_model_input_images(raw_events)
91
+ raw_events = _apply_exec_status(raw_events)
92
+ raw_events = _drop_repeated_reasoning_summaries(raw_events)
93
+ raw_events = _drop_unchanged_goal_updates(raw_events)
94
+ reconciled = _mark_reviews_repeated_by_reply(_reconcile_events(raw_events))
95
+ for event in reconciled:
96
+ if event.get("_turn_seq") in inherited_turns:
97
+ event["inherited"] = True
98
+ cleaned = [_strip_internal_keys(event) for event in reconciled]
99
+ return meta, cleaned
100
+
101
+
102
+ _INHERITANCE_TOLERANCE_SECONDS = 3.0
103
+
104
+
105
+ def _uuid7_seconds(value: Any) -> float | None:
106
+ """Creation time embedded in a UUIDv7, in seconds, or None."""
107
+ if not isinstance(value, str):
108
+ return None
109
+ digits = value.replace("-", "")
110
+ if len(digits) != 32 or digits[12] != "7":
111
+ return None
112
+ try:
113
+ return int(digits[:12], 16) / 1000
114
+ except ValueError:
115
+ return None
116
+
117
+
118
+ def _is_inherited_turn(meta: dict | None, task_started: dict) -> bool:
119
+ """True for a turn copied from the parent into a forked subagent log.
120
+
121
+ Session and turn ids are UUIDv7, so a turn created before this subagent
122
+ session existed belongs to the parent's history.
123
+ """
124
+ if not isinstance(meta, dict):
125
+ return False
126
+ source = meta.get("source")
127
+ if not isinstance(source, dict) or "subagent" not in source:
128
+ return False
129
+ session_time = _uuid7_seconds(meta.get("id"))
130
+ turn_time = _uuid7_seconds(task_started.get("turn_id"))
131
+ if session_time is None or turn_time is None:
132
+ return False
133
+ return turn_time < session_time - _INHERITANCE_TOLERANCE_SECONDS
134
+
135
+
136
+ # Record kinds the parser reads, and kinds it skips on purpose because they
137
+ # duplicate other records or carry no transcript content. Anything outside both
138
+ # sets is reported by unrecognized_record_kinds() so format changes get noticed.
139
+ _HANDLED_EVENT_MSG = {
140
+ "user_message", "agent_message", "agent_reasoning", "task_complete",
141
+ "task_started", "turn_aborted", "token_count", "thread_rolled_back",
142
+ "item_completed", "exec_command_end", "patch_apply_end",
143
+ "thread_goal_updated", "entered_review_mode", "exited_review_mode", "error",
144
+ }
145
+ # guardian_assessment and collab_* come from one alpha build; the commands and
146
+ # spawn_agent calls they describe are already shown as tool calls.
147
+ _IGNORED_EVENT_MSG = {
148
+ "mcp_tool_call_end", "view_image_tool_call", "web_search_end",
149
+ "context_compacted", "thread_settings_applied", "dynamic_tool_call_request",
150
+ "dynamic_tool_call_response", "thread_name_updated", "image_generation_end",
151
+ "guardian_assessment", "collab_agent_spawn_end", "collab_waiting_end",
152
+ "collab_close_end", "undo_completed",
153
+ }
154
+ _HANDLED_ITEM_COMPLETED = {"UserMessage", "EnteredReviewMode", "ExitedReviewMode", "HookPrompt"}
155
+ _IGNORED_ITEM_COMPLETED = {
156
+ "AgentMessage", "CommandExecution", "Reasoning", "FileChange", "McpToolCall",
157
+ "WebSearch", "Extension", "Plan", "ContextCompaction", "SubAgentActivity",
158
+ "ImageView", "FunctionCallOutput", "CollabAgentToolCall", "DynamicToolCall",
159
+ }
160
+ _HANDLED_RESPONSE_ITEM = {
161
+ "function_call", "function_call_output", "custom_tool_call",
162
+ "custom_tool_call_output", "web_search_call", "tool_search_call",
163
+ "tool_search_output", "message", "reasoning", "image_generation_call",
164
+ }
165
+ _IGNORED_RESPONSE_ITEM = {"ghost_snapshot", "agent_message"}
166
+ _IGNORED_TOP_LEVEL = {
167
+ "token_usage_record", "turn_context", "compacted", "world_state",
168
+ "inter_agent_communication_metadata", "realtime_item",
169
+ }
170
+
171
+
172
+ def unrecognized_record_kinds(entries: list[dict]) -> Counter:
173
+ """Count record kinds that are neither parsed nor deliberately ignored."""
174
+ unknown: Counter = Counter()
175
+ for entry in entries:
176
+ if not isinstance(entry, dict):
177
+ continue
178
+ etype = entry.get("type")
179
+ payload = entry.get("payload")
180
+ subtype = payload.get("type") if isinstance(payload, dict) else None
181
+ if etype == "session_meta":
182
+ continue
183
+ if etype == "event_msg":
184
+ if subtype == "item_completed":
185
+ item = payload.get("item")
186
+ kind = item.get("type") if isinstance(item, dict) else None
187
+ if kind not in _HANDLED_ITEM_COMPLETED and kind not in _IGNORED_ITEM_COMPLETED:
188
+ unknown[f"item_completed/{kind}"] += 1
189
+ elif subtype not in _HANDLED_EVENT_MSG and subtype not in _IGNORED_EVENT_MSG:
190
+ unknown[f"event_msg/{subtype}"] += 1
191
+ elif etype == "response_item":
192
+ if subtype not in _HANDLED_RESPONSE_ITEM and subtype not in _IGNORED_RESPONSE_ITEM:
193
+ unknown[f"response_item/{subtype}"] += 1
194
+ elif etype not in _IGNORED_TOP_LEVEL:
195
+ unknown[str(etype)] += 1
196
+ return unknown
197
+
198
+
199
+ def _handle_event_msg(
200
+ payload: dict[str, Any],
201
+ ts: str,
202
+ events: list[dict],
203
+ turn_seq: int,
204
+ ) -> None:
205
+ msg_type = payload.get("type", "")
206
+
207
+ if msg_type == "user_message":
208
+ local_images = payload.get("local_images")
209
+ if not isinstance(local_images, list):
210
+ local_images = []
211
+ events.append(
212
+ {
213
+ "type": "user_message",
214
+ "ts": ts,
215
+ "text": _as_text(payload.get("message", "")),
216
+ "images": local_images,
217
+ "attachments": _legacy_image_attachments(local_images),
218
+ "_source": "event_msg",
219
+ "_source_kind": "user_message",
220
+ "_turn_seq": turn_seq,
221
+ }
222
+ )
223
+ elif msg_type == "item_completed":
224
+ _handle_item_completed(payload, ts, events, turn_seq)
225
+ elif msg_type in ("exec_command_end", "patch_apply_end"):
226
+ _handle_exec_end(payload, events)
227
+ elif msg_type == "agent_message":
228
+ events.append(
229
+ {
230
+ "type": "agent_commentary",
231
+ "ts": ts,
232
+ "text": _as_text(payload.get("message", "")),
233
+ "_source": "event_msg",
234
+ "_turn_seq": turn_seq,
235
+ }
236
+ )
237
+ elif msg_type == "agent_reasoning":
238
+ events.append(
239
+ {
240
+ "type": "reasoning",
241
+ "ts": ts,
242
+ "text": _as_text(payload.get("text", "")),
243
+ "_source": "event_msg",
244
+ "_turn_seq": turn_seq,
245
+ }
246
+ )
247
+ elif msg_type == "task_complete":
248
+ events.append(
249
+ {
250
+ "type": "task_complete",
251
+ "ts": ts,
252
+ "text": _as_text(payload.get("last_agent_message", "")),
253
+ "turn_id": _as_text(payload.get("turn_id", "")),
254
+ "_source": "event_msg",
255
+ "_turn_seq": turn_seq,
256
+ }
257
+ )
258
+ elif msg_type == "task_started":
259
+ events.append(
260
+ {
261
+ "type": "task_started",
262
+ "ts": ts,
263
+ "turn_id": _as_text(payload.get("turn_id", "")),
264
+ "model_context_window": payload.get("model_context_window", ""),
265
+ "_source": "event_msg",
266
+ "_turn_seq": turn_seq,
267
+ }
268
+ )
269
+ elif msg_type == "turn_aborted":
270
+ events.append(
271
+ {
272
+ "type": "turn_aborted",
273
+ "ts": ts,
274
+ "reason": _as_text(payload.get("reason", "")),
275
+ "_source": "event_msg",
276
+ "_turn_seq": turn_seq,
277
+ }
278
+ )
279
+ elif msg_type == "token_count":
280
+ info = payload.get("info")
281
+ total = info.get("total_token_usage") if isinstance(info, dict) else None
282
+ if isinstance(total, dict) and total and _has_positive_usage(total):
283
+ rate_limits = payload.get("rate_limits")
284
+ limit_id = (
285
+ _as_text(rate_limits.get("limit_id", ""))
286
+ if isinstance(rate_limits, dict)
287
+ else ""
288
+ )
289
+ events.append(
290
+ {
291
+ "type": "token_count",
292
+ "ts": ts,
293
+ "total": total,
294
+ "rate_limit_ids": [limit_id] if limit_id else [],
295
+ "rate_limits": [rate_limits] if isinstance(rate_limits, dict) else [],
296
+ "_source": "event_msg",
297
+ "_turn_seq": turn_seq,
298
+ }
299
+ )
300
+ elif msg_type == "thread_rolled_back":
301
+ events.append(
302
+ {
303
+ "type": "thread_rolled_back",
304
+ "ts": ts,
305
+ "num_turns": payload.get("num_turns", 0),
306
+ "_source": "event_msg",
307
+ "_turn_seq": turn_seq,
308
+ }
309
+ )
310
+ elif msg_type == "thread_goal_updated":
311
+ goal = payload.get("goal")
312
+ if isinstance(goal, dict):
313
+ events.append(_goal_event(goal, ts, turn_seq))
314
+ elif msg_type == "entered_review_mode":
315
+ events.append(_review_started_event(payload, ts, turn_seq))
316
+ elif msg_type == "exited_review_mode":
317
+ events.append(_review_finished_event(payload.get("review_output"), ts, turn_seq))
318
+ elif msg_type == "error":
319
+ events.append(
320
+ {
321
+ "type": "error",
322
+ "ts": ts,
323
+ "message": _as_text(payload.get("message")),
324
+ "_source": "event_msg",
325
+ "_turn_seq": turn_seq,
326
+ }
327
+ )
328
+
329
+
330
+ def _goal_event(goal: dict, ts: str, turn_seq: int) -> dict:
331
+ tokens = goal.get("tokensUsed")
332
+ seconds = goal.get("timeUsedSeconds")
333
+ return {
334
+ "type": "goal_updated",
335
+ "ts": ts,
336
+ "objective": _as_text(goal.get("objective")),
337
+ "status": _as_text(goal.get("status")),
338
+ "tokens_used": tokens if isinstance(tokens, int) else None,
339
+ "time_used_seconds": seconds if isinstance(seconds, (int, float)) else None,
340
+ "_source": "event_msg",
341
+ "_turn_seq": turn_seq,
342
+ }
343
+
344
+
345
+ def _drop_unchanged_goal_updates(events: list[dict]) -> list[dict]:
346
+ """Keep goal updates that set a goal or change its status.
347
+
348
+ Codex logs the goal again after almost every step to update its token and
349
+ time counters; one session has 2,477 updates and six real changes.
350
+ """
351
+ kept = []
352
+ last: tuple[str, str] | None = None
353
+ for event in events:
354
+ if event.get("type") == "goal_updated":
355
+ key = (event["objective"], event["status"])
356
+ if key == last:
357
+ continue
358
+ event["new_objective"] = last is None or last[0] != event["objective"]
359
+ last = key
360
+ kept.append(event)
361
+ return kept
362
+
363
+
364
+ def _review_started_event(payload: dict, ts: str, turn_seq: int) -> dict:
365
+ hint = _as_text(payload.get("user_facing_hint")) or _as_text(payload.get("prompt"))
366
+ return {
367
+ "type": "review_started",
368
+ "ts": ts,
369
+ "hint": hint,
370
+ "_source": "event_msg",
371
+ "_turn_seq": turn_seq,
372
+ }
373
+
374
+
375
+ def _review_finished_event(output: Any, ts: str, turn_seq: int) -> dict:
376
+ output = output if isinstance(output, dict) else {}
377
+ findings = []
378
+ raw_findings = output.get("findings")
379
+ for finding in raw_findings if isinstance(raw_findings, list) else []:
380
+ if not isinstance(finding, dict):
381
+ continue
382
+ location = finding.get("code_location")
383
+ where = ""
384
+ if isinstance(location, dict):
385
+ where = _as_text(location.get("absolute_file_path"))
386
+ lines = location.get("line_range")
387
+ if where and isinstance(lines, dict) and isinstance(lines.get("start"), int):
388
+ start, end = lines["start"], lines.get("end")
389
+ where += f":{start}" + (f"-{end}" if isinstance(end, int) and end != start else "")
390
+ findings.append(
391
+ {
392
+ "title": _as_text(finding.get("title")),
393
+ "body": _as_text(finding.get("body")),
394
+ "location": where,
395
+ }
396
+ )
397
+ return {
398
+ "type": "review_finished",
399
+ "ts": ts,
400
+ "verdict": _as_text(output.get("overall_correctness")),
401
+ "explanation": _as_text(output.get("overall_explanation")),
402
+ "findings": findings,
403
+ "_source": "event_msg",
404
+ "_turn_seq": turn_seq,
405
+ }
406
+
407
+
408
+ def _mark_reviews_repeated_by_reply(events: list[dict]) -> list[dict]:
409
+ """Flag reviews whose findings an assistant message right after them repeats.
410
+
411
+ Codex usually writes the review as an assistant message too; the earliest
412
+ review versions went straight back to the model, so the findings are only
413
+ in the review record.
414
+ """
415
+ for idx, event in enumerate(events):
416
+ if event.get("type") != "review_finished":
417
+ continue
418
+ for later in events[idx + 1:]:
419
+ if later.get("type") in ("user_message", "task_started"):
420
+ break
421
+ if later.get("type") in ("assistant_text", "agent_commentary"):
422
+ event["repeated_by_reply"] = True
423
+ break
424
+ return events
425
+
426
+
427
+ def _legacy_image_attachments(local_images: list) -> list[dict]:
428
+ attachments = []
429
+ for image in local_images:
430
+ path = image.get("path") if isinstance(image, dict) else image
431
+ if isinstance(path, str) and path:
432
+ attachments.append({"kind": "local_image", "path": path})
433
+ return attachments
434
+
435
+
436
+ def _data_url_bytes(url: str) -> int:
437
+ """Approximate decoded size of a base64 data URL."""
438
+ _, _, data = url.partition(",")
439
+ return len(data) * 3 // 4
440
+
441
+
442
+ def _user_message_attachment(block: dict) -> dict | None:
443
+ kind = block.get("type")
444
+ if kind == "local_image":
445
+ path = block.get("path")
446
+ return {"kind": "local_image", "path": path} if isinstance(path, str) else None
447
+ if kind == "image":
448
+ url = block.get("image_url")
449
+ if isinstance(url, dict):
450
+ url = url.get("url")
451
+ if not isinstance(url, str):
452
+ return None
453
+ return {"kind": "image", "bytes": _data_url_bytes(url), "data_url": url}
454
+ if kind in ("skill", "mention"):
455
+ name = block.get("name")
456
+ if not isinstance(name, str) or not name:
457
+ return None
458
+ return {"kind": kind, "name": name, "path": _as_text(block.get("path"))}
459
+ return None
460
+
461
+
462
+ _MODEL_INPUT_IMAGES = "_model_input_images"
463
+ _IMAGE_ATTACHMENT_KINDS = {"local_image", "image"}
464
+
465
+
466
+ def _input_image_urls(content: Any) -> list[str]:
467
+ if not isinstance(content, list):
468
+ return []
469
+ urls = []
470
+ for block in content:
471
+ if not isinstance(block, dict) or block.get("type") != "input_image":
472
+ continue
473
+ url = block.get("image_url")
474
+ if isinstance(url, dict):
475
+ url = url.get("url")
476
+ if isinstance(url, str) and url.startswith("data:image/"):
477
+ urls.append(url)
478
+ return urls
479
+
480
+
481
+ def _attach_model_input_images(events: list[dict]) -> list[dict]:
482
+ """Give prompt image attachments the bytes the model received.
483
+
484
+ Local image files are usually gone (temp paths), but the session keeps the
485
+ base64 copy sent to the model in the same turn, in the same order.
486
+ """
487
+ queues: dict[Any, list[str]] = {}
488
+ for event in events:
489
+ if event.get("type") == _MODEL_INPUT_IMAGES:
490
+ queues.setdefault(event.get("_turn_seq"), []).extend(event["urls"])
491
+
492
+ for event in events:
493
+ if event.get("type") != "user_message":
494
+ continue
495
+ queue = queues.get(event.get("_turn_seq"))
496
+ for attachment in event.get("attachments", []):
497
+ if attachment.get("kind") not in _IMAGE_ATTACHMENT_KINDS or not queue:
498
+ continue
499
+ url = queue.pop(0)
500
+ if "data_url" not in attachment:
501
+ attachment["data_url"] = url
502
+ attachment["bytes"] = _data_url_bytes(url)
503
+
504
+ return [event for event in events if event.get("type") != _MODEL_INPUT_IMAGES]
505
+
506
+
507
+ def _handle_item_completed(
508
+ payload: dict[str, Any],
509
+ ts: str,
510
+ events: list[dict],
511
+ turn_seq: int,
512
+ ) -> None:
513
+ """Read typed prompts from CLI 0.135+ sessions, plus review and hook items.
514
+
515
+ The other item kinds duplicate response_item records that are already parsed.
516
+ """
517
+ item = payload.get("item")
518
+ if not isinstance(item, dict):
519
+ return
520
+ kind = item.get("type")
521
+ if kind == "EnteredReviewMode":
522
+ events.append(_review_started_event(item, ts, turn_seq))
523
+ return
524
+ if kind == "ExitedReviewMode":
525
+ events.append(_review_finished_event(item.get("review_output"), ts, turn_seq))
526
+ return
527
+ if kind == "HookPrompt":
528
+ _handle_hook_prompt(item, ts, events, turn_seq)
529
+ return
530
+ if kind != "UserMessage":
531
+ return
532
+ content = item.get("content")
533
+ if not isinstance(content, list):
534
+ content = []
535
+
536
+ texts: list[str] = []
537
+ attachments: list[dict] = []
538
+ for block in content:
539
+ if not isinstance(block, dict):
540
+ continue
541
+ if block.get("type") == "text":
542
+ text = _as_text(block.get("text"))
543
+ if text:
544
+ texts.append(text)
545
+ continue
546
+ attachment = _user_message_attachment(block)
547
+ if attachment is not None:
548
+ attachments.append(attachment)
549
+
550
+ text = "\n\n".join(texts).strip()
551
+ if not text and not attachments:
552
+ return
553
+
554
+ event = {
555
+ "type": "user_message",
556
+ "ts": ts,
557
+ "text": text,
558
+ "images": [a["path"] for a in attachments if a["kind"] == "local_image"],
559
+ "attachments": attachments,
560
+ "turn_id": _as_text(payload.get("turn_id")),
561
+ "item_id": _as_text(item.get("id")),
562
+ "_source": "event_msg",
563
+ "_source_kind": "item_completed",
564
+ "_turn_seq": turn_seq,
565
+ }
566
+ client_id = item.get("client_id")
567
+ if isinstance(client_id, str) and client_id:
568
+ event["client_id"] = client_id
569
+ events.append(event)
570
+
571
+
572
+ def _handle_hook_prompt(item: dict, ts: str, events: list[dict], turn_seq: int) -> None:
573
+ """Text a hook sent to the model; otherwise it is only in the model's input."""
574
+ fragments = item.get("fragments")
575
+ if not isinstance(fragments, list):
576
+ return
577
+ texts, hooks = [], []
578
+ for fragment in fragments:
579
+ if not isinstance(fragment, dict):
580
+ continue
581
+ text = _as_text(fragment.get("text")).strip()
582
+ if text:
583
+ texts.append(text)
584
+ # hookRunId looks like "stop:1:/path/to/hooks.json".
585
+ hook = _as_text(fragment.get("hookRunId")).split(":", 1)[0]
586
+ if hook and hook not in hooks:
587
+ hooks.append(hook)
588
+ if texts:
589
+ events.append(
590
+ {
591
+ "type": "hook_prompt",
592
+ "ts": ts,
593
+ "text": "\n\n".join(texts),
594
+ "hook": ", ".join(hooks),
595
+ "_source": "event_msg",
596
+ "_turn_seq": turn_seq,
597
+ }
598
+ )
599
+
600
+
601
+ def _json_object(text: str) -> dict | None:
602
+ stripped = text.strip()
603
+ if not (stripped.startswith("{") and stripped.endswith("}")):
604
+ return None
605
+ try:
606
+ value = json.loads(stripped)
607
+ except json.JSONDecodeError:
608
+ return None
609
+ return value if isinstance(value, dict) else None
610
+
611
+
612
+ def _exit_code(value: Any) -> int | None:
613
+ return value if isinstance(value, int) and not isinstance(value, bool) else None
614
+
615
+
616
+ _TRUNCATION_NOTICE = "Warning: truncated output"
617
+
618
+
619
+ def normalize_tool_output(value: Any) -> dict:
620
+ """Split a tool output into display text, image attachments and exit status.
621
+
622
+ Outputs come as plain strings, JSON strings wrapping {output, metadata}
623
+ (apply_patch), or lists of content blocks (images, code-mode exec chunks).
624
+ Exit codes are read only from structured fields, never searched for in text.
625
+ """
626
+ texts: list[str] = []
627
+ attachments: list[dict] = []
628
+ exit_codes: list[int] = []
629
+ duration: float | None = None
630
+
631
+ def add_text(text: str) -> None:
632
+ nonlocal duration
633
+ if text.startswith(_TRUNCATION_NOTICE) and "\n{" in text:
634
+ # Code-mode prefixes an oversized chunk with a notice; the JSON
635
+ # chunk after it still carries the exit code when it is complete.
636
+ head, _, tail = text.partition("\n{")
637
+ if _json_object("{" + tail) is not None:
638
+ texts.append(head.rstrip())
639
+ add_text("{" + tail)
640
+ return
641
+ obj = _json_object(text)
642
+ if obj is not None and "output" in obj:
643
+ texts.append(_as_text(obj.get("output")))
644
+ code = _exit_code(obj.get("exit_code"))
645
+ metadata = obj.get("metadata")
646
+ if isinstance(metadata, dict):
647
+ code = code if code is not None else _exit_code(metadata.get("exit_code"))
648
+ seconds = metadata.get("duration_seconds")
649
+ if isinstance(seconds, (int, float)) and not isinstance(seconds, bool):
650
+ duration = float(seconds)
651
+ if code is not None:
652
+ exit_codes.append(code)
653
+ return
654
+ texts.append(text)
655
+
656
+ if isinstance(value, str):
657
+ add_text(value)
658
+ elif isinstance(value, list):
659
+ for block in value:
660
+ if not isinstance(block, dict):
661
+ texts.append(_as_text(block))
662
+ continue
663
+ kind = block.get("type")
664
+ if kind in ("input_text", "output_text", "text"):
665
+ add_text(_as_text(block.get("text")))
666
+ elif kind == "input_image":
667
+ url = block.get("image_url")
668
+ if isinstance(url, dict):
669
+ url = url.get("url")
670
+ if isinstance(url, str) and url.startswith("data:image/"):
671
+ attachments.append(
672
+ {"kind": "image", "bytes": _data_url_bytes(url), "data_url": url}
673
+ )
674
+ else:
675
+ attachments.append({"kind": "image", "bytes": 0})
676
+ else:
677
+ texts.append(_as_text(block))
678
+ else:
679
+ texts.append(_as_text(value))
680
+
681
+ return {
682
+ "output": "\n".join(t for t in texts if t),
683
+ "attachments": attachments,
684
+ "exit_codes": exit_codes,
685
+ "failed": None,
686
+ "duration": duration,
687
+ }
688
+
689
+
690
+ _EXEC_END = "_exec_end"
691
+ _EXIT_HEADER_RE = re.compile(r"^Process exited with code (-?\d+)\s*$")
692
+ _PATCH_FAILED_PREFIX = "apply_patch verification failed"
693
+
694
+
695
+ def _handle_exec_end(payload: dict[str, Any], events: list[dict]) -> None:
696
+ """Record structured exit status, applied to the matching output later."""
697
+ call_id = payload.get("call_id")
698
+ if not isinstance(call_id, str) or not call_id:
699
+ return
700
+ status: dict[str, Any] = {"type": _EXEC_END, "call_id": call_id}
701
+ code = _exit_code(payload.get("exit_code"))
702
+ if code is not None:
703
+ status["exit_code"] = code
704
+ success = payload.get("success")
705
+ if isinstance(success, bool):
706
+ status["success"] = success
707
+ duration = payload.get("duration")
708
+ if isinstance(duration, dict):
709
+ secs, nanos = duration.get("secs"), duration.get("nanos")
710
+ if isinstance(secs, (int, float)) and isinstance(nanos, (int, float)):
711
+ status["duration"] = secs + nanos / 1e9
712
+ events.append(status)
713
+
714
+
715
+ def _header_exit_code(output: str) -> int | None:
716
+ """Legacy exec outputs state the exit code in their first few header lines."""
717
+ for line in output.splitlines()[:4]:
718
+ match = _EXIT_HEADER_RE.match(line)
719
+ if match:
720
+ return int(match.group(1))
721
+ return None
722
+
723
+
724
+ def _apply_exec_status(events: list[dict]) -> list[dict]:
725
+ """Attach exit status to tool outputs: end events first, then output headers."""
726
+ statuses = {e["call_id"]: e for e in events if e.get("type") == _EXEC_END}
727
+ for event in events:
728
+ if event.get("type") != "tool_output":
729
+ continue
730
+ status = statuses.get(event.get("call_id"))
731
+ if status is not None:
732
+ if "exit_code" in status and not event.get("exit_codes"):
733
+ event["exit_codes"] = [status["exit_code"]]
734
+ if status.get("success") is False:
735
+ event["failed"] = True
736
+ elif status.get("success") is True and event.get("failed") is None:
737
+ event["failed"] = False
738
+ if event.get("duration") is None and "duration" in status:
739
+ event["duration"] = status["duration"]
740
+ output = event.get("output", "")
741
+ if not event.get("exit_codes") and event.get("failed") is None:
742
+ code = _header_exit_code(output)
743
+ if code is not None:
744
+ event["exit_codes"] = [code]
745
+ if output.startswith(_PATCH_FAILED_PREFIX):
746
+ event["failed"] = True
747
+ return [e for e in events if e.get("type") != _EXEC_END]
748
+
749
+
750
+ def _web_search_summary(action: Any) -> str:
751
+ if not isinstance(action, dict) or not action:
752
+ return "(no details recorded)"
753
+ kind = _as_text(action.get("type")) or "search"
754
+ if kind == "search":
755
+ query = _as_text(action.get("query"))
756
+ queries = action.get("queries")
757
+ extra = [
758
+ _as_text(q) for q in queries if _as_text(q) and _as_text(q) != query
759
+ ] if isinstance(queries, list) else []
760
+ lines = [f"search: {query}" if query else "search"]
761
+ lines += [f" also: {q}" for q in extra]
762
+ return "\n".join(lines)
763
+ if kind == "open_page":
764
+ return f"open_page: {_as_text(action.get('url'))}"
765
+ if kind == "find_in_page":
766
+ return f'find_in_page: "{_as_text(action.get("pattern"))}" in {_as_text(action.get("url"))}'
767
+ return f"{kind}: {_as_text(action)}"
768
+
769
+
770
+ def _tool_search_query(arguments: Any) -> str:
771
+ """tool_search arguments arrive as JSON or as a Python dict repr."""
772
+ parsed = arguments
773
+ if isinstance(arguments, str):
774
+ try:
775
+ parsed = json.loads(arguments)
776
+ except json.JSONDecodeError:
777
+ try:
778
+ parsed = ast.literal_eval(arguments)
779
+ except (ValueError, SyntaxError, MemoryError, RecursionError):
780
+ return arguments
781
+ if isinstance(parsed, dict) and "query" in parsed:
782
+ return _as_text(parsed.get("query"))
783
+ return _as_text(parsed)
784
+
785
+
786
+ def _tool_search_names(tools: Any, prefix: str = "") -> list[str]:
787
+ """Flatten returned tool namespaces into qualified tool names."""
788
+ names: list[str] = []
789
+ if not isinstance(tools, list):
790
+ return names
791
+ for tool in tools:
792
+ if not isinstance(tool, dict):
793
+ continue
794
+ name = _as_text(tool.get("name"))
795
+ qualified = f"{prefix}.{name}" if prefix and name else name or prefix
796
+ if isinstance(tool.get("tools"), list):
797
+ names.extend(_tool_search_names(tool["tools"], qualified))
798
+ elif qualified:
799
+ names.append(qualified)
800
+ return names
801
+
802
+
803
+ def _handle_response_item(
804
+ payload: dict[str, Any],
805
+ ts: str,
806
+ events: list[dict],
807
+ turn_seq: int,
808
+ ) -> None:
809
+ item_type = payload.get("type", "")
810
+ role = payload.get("role", "")
811
+
812
+ if item_type == "function_call":
813
+ events.append(
814
+ {
815
+ "type": "tool_call",
816
+ "ts": ts,
817
+ "name": _as_text(payload.get("name", "")),
818
+ "arguments": _as_text(payload.get("arguments", "")),
819
+ "call_id": _as_text(payload.get("call_id", "")),
820
+ "_source": "response_item",
821
+ "_turn_seq": turn_seq,
822
+ }
823
+ )
824
+ elif item_type == "custom_tool_call":
825
+ events.append(
826
+ {
827
+ "type": "tool_call",
828
+ "ts": ts,
829
+ "name": _as_text(payload.get("name", "")),
830
+ "arguments": _as_text(payload.get("input", "")),
831
+ "input_kind": "custom",
832
+ "call_id": _as_text(payload.get("call_id", "")),
833
+ "_source": "response_item",
834
+ "_turn_seq": turn_seq,
835
+ }
836
+ )
837
+ elif item_type == "web_search_call":
838
+ # Single record: no call_id and no separate output.
839
+ events.append(
840
+ {
841
+ "type": "tool_call",
842
+ "ts": ts,
843
+ "name": "web_search",
844
+ "arguments": _web_search_summary(payload.get("action")),
845
+ "input_kind": "web_search",
846
+ "call_id": "",
847
+ "_source": "response_item",
848
+ "_turn_seq": turn_seq,
849
+ }
850
+ )
851
+ elif item_type == "tool_search_call":
852
+ events.append(
853
+ {
854
+ "type": "tool_call",
855
+ "ts": ts,
856
+ "name": "tool_search",
857
+ "arguments": _tool_search_query(payload.get("arguments")),
858
+ "input_kind": "tool_search",
859
+ "call_id": _as_text(payload.get("call_id", "")),
860
+ "_source": "response_item",
861
+ "_turn_seq": turn_seq,
862
+ }
863
+ )
864
+ elif item_type == "tool_search_output":
865
+ names = _tool_search_names(payload.get("tools"))
866
+ events.append(
867
+ {
868
+ "type": "tool_output",
869
+ "ts": ts,
870
+ "call_id": _as_text(payload.get("call_id", "")),
871
+ "output": "\n".join(names) if names else "(no tools returned)",
872
+ "attachments": [],
873
+ "exit_codes": [],
874
+ "failed": None,
875
+ "duration": None,
876
+ "_source": "response_item",
877
+ "_turn_seq": turn_seq,
878
+ }
879
+ )
880
+ elif item_type == "custom_tool_call_output":
881
+ events.append(
882
+ {
883
+ "type": "tool_output",
884
+ "ts": ts,
885
+ "call_id": _as_text(payload.get("call_id", "")),
886
+ **normalize_tool_output(payload.get("output", "")),
887
+ "_source": "response_item",
888
+ "_turn_seq": turn_seq,
889
+ }
890
+ )
891
+ elif item_type == "function_call_output":
892
+ events.append(
893
+ {
894
+ "type": "tool_output",
895
+ "ts": ts,
896
+ "call_id": _as_text(payload.get("call_id", "")),
897
+ **normalize_tool_output(payload.get("output", "")),
898
+ "_source": "response_item",
899
+ "_turn_seq": turn_seq,
900
+ }
901
+ )
902
+ elif item_type == "message" and role == "user":
903
+ # Only the images are used; see the note below about the text.
904
+ urls = _input_image_urls(payload.get("content"))
905
+ if urls:
906
+ events.append(
907
+ {
908
+ "type": _MODEL_INPUT_IMAGES,
909
+ "urls": urls,
910
+ "_source": "response_item",
911
+ "_turn_seq": turn_seq,
912
+ }
913
+ )
914
+ # Messages with role "user" are the model's input, not what the user typed:
915
+ # they carry injected context (environment, AGENTS.md, skills) and, in
916
+ # forked subagents, the parent's history. Typed prompts come from
917
+ # event_msg user_message (CLI <= 0.125) or item_completed UserMessage.
918
+ elif item_type == "message" and role == "assistant":
919
+ content = payload.get("content")
920
+ phase = payload.get("phase", "")
921
+ for block in content if isinstance(content, list) else []:
922
+ if isinstance(block, dict) and block.get("type") == "output_text":
923
+ events.append(
924
+ {
925
+ "type": "assistant_text",
926
+ "ts": ts,
927
+ "text": _as_text(block.get("text", "")),
928
+ "phase": _as_text(phase),
929
+ "_source": "response_item",
930
+ "_turn_seq": turn_seq,
931
+ }
932
+ )
933
+ elif item_type == "reasoning":
934
+ summary = payload.get("summary")
935
+ texts = [
936
+ _as_text(s.get("text", ""))
937
+ for s in (summary if isinstance(summary, list) else [])
938
+ if isinstance(s, dict) and s.get("type") == "summary_text"
939
+ ]
940
+ snapshot = [_normalize_text(text) for text in texts]
941
+ for index, text in enumerate(texts):
942
+ events.append(
943
+ {
944
+ "type": "reasoning",
945
+ "ts": ts,
946
+ "text": text,
947
+ "_summary": snapshot,
948
+ "_summary_index": index,
949
+ "_source": "response_item",
950
+ "_turn_seq": turn_seq,
951
+ }
952
+ )
953
+ elif item_type == "image_generation_call":
954
+ _handle_image_generation(payload, ts, events, turn_seq)
955
+
956
+
957
+ _IMAGE_BASE64_TYPES = (("iVBORw0KGgo", "png"), ("/9j/", "jpeg"), ("UklGR", "webp"), ("R0lGOD", "gif"))
958
+
959
+
960
+ def _handle_image_generation(
961
+ payload: dict[str, Any], ts: str, events: list[dict], turn_seq: int
962
+ ) -> None:
963
+ """Show an image generation as a tool call whose output is the image."""
964
+ call_id = _as_text(payload.get("id")) or f"image-generation-{len(events)}"
965
+ events.append(
966
+ {
967
+ "type": "tool_call",
968
+ "ts": ts,
969
+ "name": "image_generation",
970
+ "arguments": _as_text(payload.get("revised_prompt")) or "(no prompt recorded)",
971
+ "input_kind": "image_generation",
972
+ "call_id": call_id,
973
+ "_source": "response_item",
974
+ "_turn_seq": turn_seq,
975
+ }
976
+ )
977
+ result = payload.get("result")
978
+ attachments = []
979
+ if isinstance(result, str) and result:
980
+ kind = next((k for prefix, k in _IMAGE_BASE64_TYPES if result.startswith(prefix)), None)
981
+ attachment = {"kind": "image", "bytes": _data_url_bytes("," + result), "label": "generated image"}
982
+ if kind:
983
+ attachment["data_url"] = f"data:image/{kind};base64,{result}"
984
+ attachments.append(attachment)
985
+ status = _as_text(payload.get("status"))
986
+ events.append(
987
+ {
988
+ "type": "tool_output",
989
+ "ts": ts,
990
+ "call_id": call_id,
991
+ "output": "" if attachments else f"(no image recorded; status {status or 'unknown'})",
992
+ "attachments": attachments,
993
+ "exit_codes": [],
994
+ "failed": False if attachments else (True if status == "failed" else None),
995
+ "duration": None,
996
+ "_source": "response_item",
997
+ "_turn_seq": turn_seq,
998
+ }
999
+ )
1000
+
1001
+
1002
+ _TOOL_EVENT_TYPES = {"tool_call", "tool_output"}
1003
+
1004
+
1005
+ def _drop_repeated_reasoning_summaries(events: list[dict]) -> list[dict]:
1006
+ """Show each reasoning summary heading once per turn.
1007
+
1008
+ Newer models often start a turn's next reasoning item with the whole
1009
+ summary logged so far, then add new parts: ["A"], ["A", "B"], ["A", "B"].
1010
+ When an item's summary starts with the previous item's full summary, only
1011
+ the parts after it are new. Summaries never carry over between turns.
1012
+ """
1013
+ kept: list[dict] = []
1014
+ previous: dict[Any, list[str]] = {}
1015
+ repeated = 0
1016
+ for event in events:
1017
+ snapshot = event.get("_summary")
1018
+ if snapshot is None:
1019
+ kept.append(event)
1020
+ continue
1021
+ turn = event.get("_turn_seq")
1022
+ if event.get("_summary_index") == 0:
1023
+ before = previous.get(turn) or []
1024
+ repeated = len(before) if before and snapshot[: len(before)] == before else 0
1025
+ previous[turn] = snapshot
1026
+ if event["_summary_index"] >= repeated:
1027
+ kept.append(event)
1028
+ return kept
1029
+
1030
+
1031
+ def _merge_adjacent_token_events(events: list[dict]) -> list[dict]:
1032
+ """Collapse repeated token totals within a turn.
1033
+
1034
+ Tool events between two token_count events do not break the run, so how
1035
+ many tool calls are visible never changes which token totals survive.
1036
+ """
1037
+ merged: list[dict] = []
1038
+ last_non_tool: dict | None = None
1039
+ for event in events:
1040
+ if event.get("type") in _TOOL_EVENT_TYPES:
1041
+ merged.append(event.copy())
1042
+ continue
1043
+ if (
1044
+ event.get("type") == "token_count"
1045
+ and last_non_tool is not None
1046
+ and last_non_tool.get("type") == "token_count"
1047
+ and last_non_tool.get("_turn_seq") == event.get("_turn_seq")
1048
+ and last_non_tool.get("total") == event.get("total")
1049
+ ):
1050
+ _merge_token_metadata(last_non_tool, event)
1051
+ continue
1052
+ copy = event.copy()
1053
+ merged.append(copy)
1054
+ last_non_tool = copy
1055
+ return merged
1056
+
1057
+
1058
+ def _merge_token_metadata(base_event: dict, event: dict) -> None:
1059
+ base_ids = list(base_event.get("rate_limit_ids", []))
1060
+ seen_ids = set(base_ids)
1061
+ for limit_id in event.get("rate_limit_ids", []):
1062
+ if limit_id not in seen_ids:
1063
+ base_ids.append(limit_id)
1064
+ seen_ids.add(limit_id)
1065
+ base_event["rate_limit_ids"] = base_ids
1066
+
1067
+ base_rate_limits = list(base_event.get("rate_limits", []))
1068
+ for rate_limit in event.get("rate_limits", []):
1069
+ if isinstance(rate_limit, dict) and rate_limit not in base_rate_limits:
1070
+ base_rate_limits.append(rate_limit)
1071
+ base_event["rate_limits"] = base_rate_limits
1072
+
1073
+
1074
+ def _normalize_text(value: Any) -> str:
1075
+ return " ".join(_as_text(value).split())
1076
+
1077
+
1078
+ _EVENT_COPY_OMITS_RE = re.compile(r"<(oai-mem-citation|proposed_plan)>(?:.*?</\1>|.*\Z)", re.S)
1079
+
1080
+
1081
+ def _event_copy_text(response_text: Any) -> str:
1082
+ """The text an event_msg copy of this response carries.
1083
+
1084
+ agent_message and task_complete copies leave out proposed-plan and
1085
+ memory-citation blocks, wherever they sit in the answer.
1086
+ """
1087
+ return _normalize_text(_EVENT_COPY_OMITS_RE.sub("", _as_text(response_text)))
1088
+
1089
+
1090
+ def _is_response_counterpart(candidate: dict, response_event: dict) -> bool:
1091
+ if response_event.get("_source") != "response_item":
1092
+ return False
1093
+ if candidate.get("_turn_seq") != response_event.get("_turn_seq"):
1094
+ return False
1095
+
1096
+ candidate_type = candidate.get("type")
1097
+ candidate_text = _normalize_text(candidate.get("text", ""))
1098
+ response_text = _normalize_text(response_event.get("text", ""))
1099
+ if candidate_type == "reasoning":
1100
+ return response_event.get("type") == "reasoning" and candidate_text == response_text
1101
+ if candidate_type not in ("agent_commentary", "task_complete"):
1102
+ return False
1103
+ if response_event.get("type") != "assistant_text":
1104
+ return False
1105
+ # Older sessions log each assistant message, final answers included, as an
1106
+ # agent_message; task_complete repeats the turn's last message.
1107
+ if candidate_text in (response_text, _event_copy_text(response_event.get("text", ""))):
1108
+ return True
1109
+ # task_complete may also cut a final answer short before a trailing block.
1110
+ return (
1111
+ candidate_type == "task_complete"
1112
+ and response_event.get("phase") == "final_answer"
1113
+ and bool(candidate_text)
1114
+ and response_text.startswith(candidate_text)
1115
+ )
1116
+
1117
+
1118
+ _MATCHABLE_EVENT_MSG_TYPES = {"agent_commentary", "reasoning", "task_complete"}
1119
+ _RESPONSE_COUNTERPART_TYPES = {"assistant_text", "reasoning"}
1120
+
1121
+
1122
+ def _find_matching_response_index(
1123
+ events: list[dict],
1124
+ idx: int,
1125
+ turn_pool: list[int],
1126
+ used_indices: set[int],
1127
+ ) -> int | None:
1128
+ """Return the nearest unused same-turn response_item duplicating events[idx].
1129
+
1130
+ The pool holds only response-side message and reasoning events, so tool
1131
+ events never push a counterpart out of reach. task_complete repeats a
1132
+ message that its agent_message copy may already have matched, so it may
1133
+ match a used response too.
1134
+ """
1135
+ candidate = events[idx]
1136
+ if candidate.get("_source") != "event_msg":
1137
+ return None
1138
+ if candidate.get("type") not in _MATCHABLE_EVENT_MSG_TYPES:
1139
+ return None
1140
+
1141
+ for j in sorted(turn_pool, key=lambda j: abs(j - idx)):
1142
+ if j in used_indices and candidate.get("type") != "task_complete":
1143
+ continue
1144
+ if _is_response_counterpart(candidate, events[j]):
1145
+ return j
1146
+ return None
1147
+
1148
+
1149
+ def _drop_overlapped_event_msg_events(events: list[dict]) -> list[dict]:
1150
+ pools: dict[Any, list[int]] = {}
1151
+ for idx, event in enumerate(events):
1152
+ if (
1153
+ event.get("_source") == "response_item"
1154
+ and event.get("type") in _RESPONSE_COUNTERPART_TYPES
1155
+ ):
1156
+ pools.setdefault(event.get("_turn_seq"), []).append(idx)
1157
+
1158
+ filtered: list[dict] = []
1159
+ used_response_indices: set[int] = set()
1160
+
1161
+ for idx, event in enumerate(events):
1162
+ if event.get("type") == "task_complete" and not _normalize_text(event.get("text", "")):
1163
+ continue
1164
+
1165
+ match_idx = _find_matching_response_index(
1166
+ events, idx, pools.get(event.get("_turn_seq"), []), used_response_indices
1167
+ )
1168
+ if match_idx is not None:
1169
+ if event.get("type") != "task_complete":
1170
+ used_response_indices.add(match_idx)
1171
+ continue
1172
+
1173
+ filtered.append(event)
1174
+
1175
+ return filtered
1176
+
1177
+
1178
+ def _drop_legacy_prompts_shadowed_by_items(events: list[dict]) -> list[dict]:
1179
+ """Keep one prompt source per turn if a session ever records both."""
1180
+ item_turns = {
1181
+ event.get("_turn_seq")
1182
+ for event in events
1183
+ if event.get("type") == "user_message"
1184
+ and event.get("_source_kind") == "item_completed"
1185
+ }
1186
+ return [
1187
+ event
1188
+ for event in events
1189
+ if not (
1190
+ event.get("type") == "user_message"
1191
+ and event.get("_source_kind") == "user_message"
1192
+ and event.get("_turn_seq") in item_turns
1193
+ )
1194
+ ]
1195
+
1196
+
1197
+ def _reconcile_events(events: list[dict]) -> list[dict]:
1198
+ merged = _merge_adjacent_token_events(events)
1199
+ deduped = _drop_overlapped_event_msg_events(merged)
1200
+ return _drop_legacy_prompts_shadowed_by_items(deduped)
1201
+
1202
+
1203
+ def _strip_internal_keys(event: dict) -> dict:
1204
+ return {k: v for k, v in event.items() if not k.startswith("_")}