cc-transcript 12.0.0__tar.gz → 12.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/PKG-INFO +1 -1
  2. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/cli.py +56 -2
  3. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/discovery.py +9 -0
  4. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/facts.py +1 -0
  5. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/judge/similar.py +3 -1
  6. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/__init__.py +3 -1
  7. cc_transcript-12.1.1/cc_transcript/mining/sampling.py +109 -0
  8. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/store.py +4 -2
  9. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/tools.py +2 -2
  10. cc_transcript-12.1.1/cc_transcript/watch.py +235 -0
  11. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/pyproject.toml +1 -1
  12. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/Cargo.lock +0 -0
  13. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/Cargo.toml +0 -0
  14. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/LICENSE +0 -0
  15. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/README.md +0 -0
  16. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/__init__.py +0 -0
  17. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/__main__.py +0 -0
  18. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/_parser_rs.pyi +0 -0
  19. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/activity.py +0 -0
  20. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/activity_probe.py +0 -0
  21. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/backend.py +0 -0
  22. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/builders.py +0 -0
  23. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/command.py +0 -0
  24. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/context.py +0 -0
  25. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/corrections.py +0 -0
  26. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/corrections_cli.py +0 -0
  27. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/cost.py +0 -0
  28. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/decisions.py +0 -0
  29. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/disktruth.py +0 -0
  30. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/evidence.py +0 -0
  31. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/extract/__init__.py +0 -0
  32. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/extract/correct.py +0 -0
  33. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/filterspec.py +0 -0
  34. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/ids.py +0 -0
  35. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/judge/__init__.py +0 -0
  36. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/judge/llm.py +0 -0
  37. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/judge/verdicts.py +0 -0
  38. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/ledger.py +0 -0
  39. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/candidates.py +0 -0
  40. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/confidence.py +0 -0
  41. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/engine.py +0 -0
  42. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/filterspec.py +0 -0
  43. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/formats.py +0 -0
  44. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/signals.py +0 -0
  45. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/sourcekind.py +0 -0
  46. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/mining/spec.py +0 -0
  47. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/models.py +0 -0
  48. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/notifications.py +0 -0
  49. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/parser.py +0 -0
  50. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/py.typed +0 -0
  51. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/query.py +0 -0
  52. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/render.py +0 -0
  53. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/rust.py +0 -0
  54. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/__init__.py +0 -0
  55. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/buckets.py +0 -0
  56. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/data/afinn-en-165.tsv +0 -0
  57. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/data/domain_overrides.tsv +0 -0
  58. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/engine.py +0 -0
  59. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/lexicon.py +0 -0
  60. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/sentiment/scorespec.py +0 -0
  61. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/cc_transcript/store.py +0 -0
  62. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/Cargo.toml +0 -0
  63. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/build.rs +0 -0
  64. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/data/command_prefix_pins.tsv +0 -0
  65. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/rustfmt.toml +0 -0
  66. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/activity.rs +0 -0
  67. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/command.rs +0 -0
  68. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/event.rs +0 -0
  69. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/filter.rs +0 -0
  70. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/generated/command.rs +0 -0
  71. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/generated/mining.rs +0 -0
  72. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/generated/mod.rs +0 -0
  73. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/generated/protocol.rs +0 -0
  74. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/generated/unicode.rs +0 -0
  75. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/lexicon.rs +0 -0
  76. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/lib.rs +0 -0
  77. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/mining.rs +0 -0
  78. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/model.rs +0 -0
  79. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/parse.rs +0 -0
  80. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/protocol.rs +0 -0
  81. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/python.rs +0 -0
  82. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/score.rs +0 -0
  83. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/types.rs +0 -0
  84. {cc_transcript-12.0.0 → cc_transcript-12.1.1}/rust/src/value.rs +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cc-transcript
3
- Version: 12.0.0
3
+ Version: 12.1.1
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: Environment :: Console
6
6
  Classifier: Intended Audience :: Developers
@@ -26,11 +26,14 @@ from cc_transcript.builders import (
26
26
  )
27
27
  from cc_transcript.corrections_cli import corrections
28
28
  from cc_transcript.discovery import CLAUDE_PROJECTS_DIR, TranscriptDiscovery, find_transcript_sync
29
- from cc_transcript.filterspec import ASSISTANTS, USERS, EventKind, event_kind, keep, tool_names
29
+ from cc_transcript.facts import command_prefix_counts, mcp_summary, tool_facts
30
+ from cc_transcript.filterspec import ASSISTANTS, USERS, EventKind, event_kind, event_meta, keep, tool_names
30
31
  from cc_transcript.ids import SessionId, tool_digest
31
32
  from cc_transcript.models import AssistantEvent, ToolResultBlock, ToolUseBlock, UserEvent
32
33
  from cc_transcript.parser import TranscriptParser
33
34
  from cc_transcript.render import (
35
+ BLANK_TIME,
36
+ TAGS,
34
37
  WHERE_ALL,
35
38
  Budget,
36
39
  collect_stats,
@@ -39,6 +42,7 @@ from cc_transcript.render import (
39
42
  denial_line,
40
43
  display_path,
41
44
  event_dict,
45
+ event_payload,
42
46
  fact_dict,
43
47
  fact_line,
44
48
  haystack,
@@ -49,9 +53,10 @@ from cc_transcript.render import (
49
53
  render_tool_call,
50
54
  stats_dict,
51
55
  transcript_header,
56
+ truncate,
52
57
  )
53
- from cc_transcript.facts import command_prefix_counts, mcp_summary, tool_facts
54
58
  from cc_transcript.tools import file_path_of, parse_tool_call, tool_name_matches
59
+ from cc_transcript.watch import watch
55
60
 
56
61
  if TYPE_CHECKING:
57
62
  from collections.abc import Iterable, Mapping, Sequence
@@ -61,6 +66,7 @@ if TYPE_CHECKING:
61
66
  from cc_transcript.facts import ToolFact
62
67
  from cc_transcript.filterspec import FilterSpec
63
68
  from cc_transcript.models import EntryMeta, ToolUseId, TranscriptEvent
69
+ from cc_transcript.watch import WatchEvent
64
70
 
65
71
  type Row = tuple[int, TranscriptEvent]
66
72
 
@@ -280,6 +286,29 @@ def slice_line(meta: EntryMeta, block: ToolUseBlock) -> dict[str, Any]:
280
286
  }
281
287
 
282
288
 
289
+
290
+ def watch_dict(item: WatchEvent) -> dict[str, Any]:
291
+ meta = event_meta(item.event)
292
+ kind = event_kind(item.event)
293
+ return {
294
+ "path": str(item.path),
295
+ "session_id": item.session_id,
296
+ "is_sidechain": item.is_sidechain,
297
+ "uuid": meta.uuid if meta is not None else None,
298
+ "kind": kind,
299
+ "role": kind if kind in ("user", "assistant") else None,
300
+ "preview": truncate(event_payload(item.event, names={}, width=120, thinking=False), 120),
301
+ }
302
+
303
+
304
+ def watch_line(item: WatchEvent) -> str:
305
+ meta = event_meta(item.event)
306
+ time = meta.timestamp.strftime("%H:%M:%S") if meta is not None else BLANK_TIME
307
+ tag = TAGS[event_kind(item.event)] + ("*" if item.is_sidechain else "")
308
+ payload = event_payload(item.event, names={}, width=100, thinking=False)
309
+ return f"{time} {item.session_id[:8]} {tag:<5} {payload}".rstrip()
310
+
311
+
283
312
  @click.group()
284
313
  @click.version_option(package_name="cc-transcript")
285
314
  def cli() -> None:
@@ -672,3 +701,28 @@ def digest(check: Path | None) -> None:
672
701
  ),
673
702
  )
674
703
  )
704
+ @cli.command("watch")
705
+ @click.option(
706
+ "--root",
707
+ "roots",
708
+ multiple=True,
709
+ type=click.Path(file_okay=False, path_type=Path),
710
+ help="Projects directory to tail; repeatable [default: ~/.claude/projects].",
711
+ )
712
+ @click.option("--poll", default=1.0, show_default=True, help="Seconds between filesystem polls.")
713
+ @click.option("--from-start", is_flag=True, help="Replay preexisting transcript content instead of tailing from EOF.")
714
+ @click.option("--json", "as_json", is_flag=True, help="Emit one NDJSON object per event.")
715
+ def watch_(roots: tuple[Path, ...], poll: float, from_start: bool, as_json: bool) -> None:
716
+ """Tail transcripts live, one line per newly appended event, until interrupted."""
717
+
718
+ async def run() -> None:
719
+ async for item in watch(roots or (CLAUDE_PROJECTS_DIR,), poll=poll, from_start=from_start):
720
+ click.echo(orjson.dumps(watch_dict(item)) if as_json else watch_line(item))
721
+
722
+ try:
723
+ anyio.run(run)
724
+ except KeyboardInterrupt:
725
+ return
726
+ except BrokenPipeError:
727
+ os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno())
728
+ raise SystemExit(0) from None
@@ -151,6 +151,15 @@ async def find_transcript(session_id: SessionId, *, root: Path | None = None) ->
151
151
  return await anyio.to_thread.run_sync(partial(find_transcript_sync, session_id, root=root))
152
152
 
153
153
 
154
+ def is_subagent_path(path: Path) -> bool:
155
+ """Whether ``path`` names a subagent sidechain transcript.
156
+
157
+ Matches the ``agent-<tool_use_id>.jsonl`` naming convention that
158
+ :func:`subagent_paths` discovers.
159
+ """
160
+ return path.suffix == ".jsonl" and path.name.startswith("agent-")
161
+
162
+
154
163
  def subagent_paths(path: Path) -> tuple[Path, ...]:
155
164
  """Sidechain transcript files spawned by the session transcript at ``path``.
156
165
 
@@ -95,6 +95,7 @@ def fact_of(use: ToolUse, session_id: SessionId, path: Path, prefixes: tuple[str
95
95
  call = use.call
96
96
  server, tool, access = mcp_split(call.name)
97
97
  denied, user_said = denial_fields(use.result)
98
+ assert use.ref.tool_use_id is not None, "ToolUse refs always carry the tool-use id"
98
99
  return ToolFact(
99
100
  ts=use.ts,
100
101
  session_id=session_id,
@@ -188,7 +188,9 @@ async def embed_evidence(store: FileStateStore, *, dedup_key: DedupKey, canonica
188
188
  await prepare_connection(store)
189
189
  async with store.lock:
190
190
  cursor = await store.conn.execute("SELECT text FROM feedback_events WHERE dedup_key = ?", (dedup_key,))
191
- text = (await cursor.fetchone())["text"]
191
+ row = await cursor.fetchone()
192
+ assert row is not None, "verdict dedup keys always resolve to a stored feedback event"
193
+ text = row["text"]
192
194
  embedder = await anyio.to_thread.run_sync(default_embedder)
193
195
  vector = await anyio.to_thread.run_sync(embedder, f"{text}\n{summary}")
194
196
  return Evidence(serialize_vector(vector), text, canonical_key)
@@ -11,7 +11,8 @@ disqualification rules, their review formats), capture each candidate's durable
11
11
  :class:`~cc_transcript.context.ContextWindow` via
12
12
  :func:`~cc_transcript.context.capture_window`, and persist them through
13
13
  :class:`FeedbackStore`. LLM verdict passes over the stored corpus live in
14
- :mod:`cc_transcript.judge`.
14
+ :mod:`cc_transcript.judge`; deterministic "did not steer here" negatives come
15
+ from :func:`sample_windows`.
15
16
 
16
17
  The :class:`MiningSpec` is the mining analogue of :class:`~cc_transcript.FilterSpec`
17
18
  and :class:`~cc_transcript.sentiment.ScoreSpec`: a frozen-dataclass tree with a JSON
@@ -65,6 +66,7 @@ from cc_transcript.mining.formats import (
65
66
  StructuredFormat,
66
67
  extract_structured,
67
68
  )
69
+ from cc_transcript.mining.sampling import sample_windows
68
70
  from cc_transcript.mining.signals import (
69
71
  MiningSignal,
70
72
  mine,
@@ -0,0 +1,109 @@
1
+ """Deterministic negative-window sampling over a session's completed turns.
2
+
3
+ :func:`sample_windows` draws "the user did not steer here" negatives: durable
4
+ :class:`~cc_transcript.context.ContextWindow` captures anchored on completed
5
+ turns, reproducible for a given seed and session, kept clear of known
6
+ positives by an exclusion radius. Negative windows carry no trigger — the
7
+ sampled turn folds into ``before`` — so they render byte-compatibly with
8
+ positive steering windows, whose user-steer trigger is likewise excluded from
9
+ model input: both shapes read as the turns up to the moment being judged.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import random
15
+ from dataclasses import replace
16
+ from typing import TYPE_CHECKING
17
+
18
+ from cc_transcript.context import capture_window
19
+ from cc_transcript.filterspec import event_meta
20
+ from cc_transcript.ids import EventRef
21
+
22
+ if TYPE_CHECKING:
23
+ from collections.abc import Iterable
24
+
25
+ from cc_transcript.activity import SessionActivity, Turn
26
+ from cc_transcript.context import ContextWindow
27
+
28
+
29
+ def sample_windows(
30
+ activity: SessionActivity,
31
+ *,
32
+ n: int,
33
+ exclude: Iterable[EventRef] = (),
34
+ exclusion_radius: int = 6,
35
+ seed: int = 0,
36
+ before: int = 6,
37
+ after: int = 2,
38
+ preview_chars: int = 200,
39
+ ) -> list[ContextWindow]:
40
+ """Sample up to ``n`` triggerless context windows as steering negatives.
41
+
42
+ Each window anchors on a completed turn — the agent acted and the user's
43
+ next prompt was not a steer we know about — and folds that turn into the
44
+ window's context: ``trigger`` is None, ``before`` ends at the sampled
45
+ turn and keeps at most ``before`` turns, and ``after`` is unchanged. The
46
+ ``anchor`` stays the sampled turn's first meta-bearing event, so
47
+ consumers key negatives exactly like positives.
48
+
49
+ Candidates are every turn carrying at least one event with resolvable
50
+ meta (the anchor), except the session's final turn, which may still be in
51
+ flight. Every candidate in the ``exclusion_radius`` turns leading up to an
52
+ ``exclude`` ref's turn is dropped; refs that no longer resolve are
53
+ ignored. Sampling is deterministic:
54
+ ``random.Random(f"{seed}:{session_id}")`` draws from the candidates in
55
+ turn order, so one seed always yields the same windows for a session.
56
+
57
+ Args:
58
+ activity: The lifted session to sample from.
59
+ n: The maximum number of windows to return.
60
+ exclude: Anchors of known positives to keep clear of.
61
+ exclusion_radius: How many turns before each excluded turn to drop
62
+ (the excluded turn itself included). Turns after an excluded turn
63
+ stay eligible — once the user has steered, letting the agent run
64
+ again is a genuine negative; only the pre-steer approach, which
65
+ positive-window rewinds occupy, is label-conflicted.
66
+ seed: The determinism seed, mixed with the session id.
67
+ before: How many turns each window's folded ``before`` keeps, ending
68
+ at the sampled turn.
69
+ after: How many turns after each sampled turn to capture.
70
+ preview_chars: The per-chunk preview budget persisted on each window.
71
+
72
+ Returns:
73
+ The sampled windows, sorted by sampled turn index.
74
+ """
75
+ excluded = {turn.index for ref in exclude if (turn := activity.turn_of(ref)) is not None}
76
+ candidates = [
77
+ (turn.index, anchor)
78
+ for turn in activity.turns[:-1]
79
+ if all(not (0 <= index - turn.index <= exclusion_radius) for index in excluded)
80
+ if (anchor := turn_anchor(turn)) is not None
81
+ ]
82
+ rng = random.Random(f"{seed}:{activity.session_id}")
83
+ chosen = rng.sample(candidates, min(n, len(candidates)))
84
+ return [
85
+ fold_trigger(
86
+ capture_window(activity, anchor, before=before, after=after, preview_chars=preview_chars),
87
+ keep=before,
88
+ )
89
+ for _, anchor in sorted(chosen, key=lambda pair: pair[0])
90
+ ]
91
+
92
+
93
+ def fold_trigger(window: ContextWindow, *, keep: int) -> ContextWindow:
94
+ """Fold ``window``'s trigger into ``before`` — the negative shape has none.
95
+
96
+ Returns:
97
+ The window with ``trigger`` None and ``before`` ending at the old
98
+ trigger, truncated to the last ``keep`` turns.
99
+ """
100
+ folded = (*window.before, *(() if window.trigger is None else (window.trigger,)))
101
+ return replace(window, before=folded[-keep:] if keep > 0 else (), trigger=None)
102
+
103
+
104
+ def turn_anchor(turn: Turn) -> EventRef | None:
105
+ """The reference to ``turn``'s first meta-bearing event, or None without one."""
106
+ return next(
107
+ (EventRef(meta.session_id, meta.uuid) for event in turn.events if (meta := event_meta(event)) is not None),
108
+ None,
109
+ )
@@ -150,9 +150,11 @@ class FeedbackStore:
150
150
  by_source_cur = await conn.execute(
151
151
  "SELECT source_kind, COUNT(*) AS n FROM feedback_events GROUP BY source_kind ORDER BY source_kind"
152
152
  )
153
+ total_row, files_row = await total_cur.fetchone(), await files_cur.fetchone()
154
+ assert total_row is not None and files_row is not None, "COUNT(*) always returns one row"
153
155
  return Stats(
154
- total=(await total_cur.fetchone())["n"],
155
- files=(await files_cur.fetchone())["n"],
156
+ total=total_row["n"],
157
+ files=files_row["n"],
156
158
  by_source={row["source_kind"]: row["n"] async for row in by_source_cur},
157
159
  )
158
160
 
@@ -90,7 +90,7 @@ class EditSpan:
90
90
  def from_raw(cls, span: object) -> Self:
91
91
  match span:
92
92
  case {"old_string": str() as old, "new_string": str() as new}:
93
- return cls(old, new, span.get("replace_all", False))
93
+ return cls(old, new, bool(span.get("replace_all", False)))
94
94
  case _:
95
95
  raise TypeError(f"edit span missing or malformed: {span!r}")
96
96
 
@@ -408,7 +408,7 @@ ToolCall = (
408
408
  | OtherCall
409
409
  )
410
410
 
411
- TOOL_TYPES: dict[str, type[ToolCallBase]] = {
411
+ TOOL_TYPES: dict[str, type[ToolCall]] = {
412
412
  "Bash": BashCall,
413
413
  "Edit": EditCall,
414
414
  "MultiEdit": MultiEditCall,
@@ -0,0 +1,235 @@
1
+ """Live transcript tailing: poll the projects tree, yield each appended event once.
2
+
3
+ A byte-offset tailer over the ``*.jsonl`` transcripts: every file gets a
4
+ :class:`TailCursor`, each poll reads only what appended since the last one
5
+ (open-read-close, never holding a descriptor), complete lines decode through
6
+ the parser's per-line decode, and a bounded per-file uuid set keeps compaction
7
+ rewrites and replays from double-firing. All progression lives in
8
+ :func:`tick` — one deterministic step over a :class:`TailState`, directly
9
+ drivable by tests and embedders — and :func:`watch` is the thin poll-forever
10
+ loop over it.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass, field
16
+ from pathlib import Path
17
+ from typing import TYPE_CHECKING
18
+
19
+ import anyio
20
+ import anyio.to_thread
21
+
22
+ from cc_transcript.discovery import CLAUDE_PROJECTS_DIR, is_subagent_path
23
+ from cc_transcript.filterspec import event_meta
24
+ from cc_transcript.ids import SessionId
25
+ from cc_transcript.models import ModeEvent
26
+ from cc_transcript.parser import decode_line
27
+
28
+ if TYPE_CHECKING:
29
+ import os
30
+ from collections.abc import AsyncIterator, Sequence
31
+
32
+ from cc_transcript.ids import EventUuid
33
+ from cc_transcript.models import TranscriptEvent
34
+
35
+ SEEN_LIMIT = 4096
36
+ """How many yielded event uuids each file's dedupe set retains."""
37
+
38
+
39
+ @dataclass(frozen=True, slots=True)
40
+ class WatchEvent:
41
+ """One transcript event freshly appended to a watched file.
42
+
43
+ Attributes:
44
+ path: The transcript file the event was read from.
45
+ session_id: The session the event belongs to — from the event's own
46
+ envelope when it carries one, else the file's last-known session,
47
+ else derived from the transcript path.
48
+ is_sidechain: Whether the file is a subagent sidechain transcript.
49
+ event: The parsed transcript event.
50
+ """
51
+
52
+ path: Path
53
+ session_id: SessionId
54
+ is_sidechain: bool
55
+ event: TranscriptEvent
56
+
57
+
58
+ @dataclass(slots=True)
59
+ class TailCursor:
60
+ """One watched file's tail progress.
61
+
62
+ Attributes:
63
+ offset: Bytes consumed so far — always the end of the last complete
64
+ line, so a partial trailing line stays unconsumed until its
65
+ newline arrives.
66
+ size: The file size at the last processed stat.
67
+ mtime: The file mtime at the last processed stat.
68
+ session_id: The session id cached from the first decoded event that
69
+ carried one.
70
+ seen: Event uuids already yielded, insertion-ordered and bounded at
71
+ :data:`SEEN_LIMIT`.
72
+ """
73
+
74
+ offset: int
75
+ size: int
76
+ mtime: float
77
+ session_id: SessionId | None = None
78
+ seen: dict[EventUuid, None] = field(default_factory=dict)
79
+
80
+
81
+ @dataclass(slots=True)
82
+ class TailState:
83
+ """The tailer's whole mutable state: one cursor per discovered file.
84
+
85
+ Attributes:
86
+ cursors: Tail progress per transcript file.
87
+ primed: Whether the initial discovery pass has run — files first seen
88
+ on that pass are history, files appearing later are new content.
89
+ """
90
+
91
+ cursors: dict[Path, TailCursor] = field(default_factory=dict)
92
+ primed: bool = False
93
+
94
+
95
+ async def watch(
96
+ roots: Sequence[Path] = (CLAUDE_PROJECTS_DIR,),
97
+ *,
98
+ poll: float = 1.0,
99
+ from_start: bool = False,
100
+ ) -> AsyncIterator[WatchEvent]:
101
+ """Tail every transcript under ``roots`` forever, yielding appended events.
102
+
103
+ An async generator that never returns on its own: each iteration drains
104
+ one :func:`tick` and sleeps ``poll`` seconds. Content predating the first
105
+ tick is skipped unless ``from_start``; everything appended afterwards is
106
+ yielded exactly once, with sidechain transcripts flagged on the event.
107
+ """
108
+ state = TailState()
109
+ while True:
110
+ for event in await tick(state, roots, from_start=from_start):
111
+ yield event
112
+ await anyio.sleep(poll)
113
+
114
+
115
+ async def tick(state: TailState, roots: Sequence[Path], *, from_start: bool = False) -> list[WatchEvent]:
116
+ """Run one poll step: discover changes under ``roots`` and drain them.
117
+
118
+ A file is re-read only when its size or mtime changed — both are compared,
119
+ because mtime granularity can hide rapid appends. A file first seen on the
120
+ priming pass starts at end-of-file unless ``from_start``, so daemon starts
121
+ never replay history; a file appearing on a later pass starts at byte 0,
122
+ since its whole content is new. A file whose size fell below the cursor
123
+ was rewritten (compaction): the cursor resets to 0 and its dedupe set
124
+ clears. The cursor only ever advances past the last complete line — a
125
+ partial trailing line waits, and lines the decoder rejects are skipped.
126
+
127
+ Returns:
128
+ The newly appended events, files in path order, lines in file order.
129
+ """
130
+ stats = await scan(roots)
131
+ priming = not state.primed
132
+ state.primed = True
133
+ events: list[WatchEvent] = []
134
+ for path in sorted(stats):
135
+ stat = stats[path]
136
+ if (cursor := state.cursors.get(path)) is None:
137
+ skip_history = priming and not from_start
138
+ cursor = state.cursors[path] = TailCursor(
139
+ offset=stat.st_size if skip_history else 0,
140
+ size=stat.st_size if skip_history else -1,
141
+ mtime=stat.st_mtime if skip_history else -1.0,
142
+ )
143
+ if stat.st_size < cursor.offset:
144
+ cursor.offset = 0
145
+ cursor.seen.clear()
146
+ elif stat.st_size == cursor.size and stat.st_mtime == cursor.mtime:
147
+ continue
148
+ try:
149
+ chunk = await anyio.to_thread.run_sync(read_from, path, cursor.offset)
150
+ except OSError:
151
+ continue
152
+ cursor.size, cursor.mtime = stat.st_size, stat.st_mtime
153
+ complete, _, partial = chunk.rpartition(b"\n")
154
+ cursor.offset += len(chunk) - len(partial)
155
+ for line in complete.split(b"\n"):
156
+ if (event := decode(line)) is None:
157
+ continue
158
+ if (meta := event_meta(event)) is not None:
159
+ if meta.uuid in cursor.seen:
160
+ continue
161
+ remember(cursor, meta.uuid)
162
+ events.append(
163
+ WatchEvent(
164
+ path=path,
165
+ session_id=session_of(cursor, path, event),
166
+ is_sidechain=is_subagent_path(path),
167
+ event=event,
168
+ )
169
+ )
170
+ return events
171
+
172
+
173
+ async def scan(roots: Sequence[Path]) -> dict[Path, os.stat_result]:
174
+ """Stat every transcript under ``roots``, skipping macOS resource forks."""
175
+ found: dict[Path, os.stat_result] = {}
176
+ for root in roots:
177
+ base = anyio.Path(root)
178
+ if not await base.exists():
179
+ continue
180
+ async for entry in base.rglob("*.jsonl"):
181
+ if entry.name.startswith("._"):
182
+ continue
183
+ try:
184
+ found[Path(entry)] = await entry.stat()
185
+ except OSError:
186
+ continue
187
+ return found
188
+
189
+
190
+ def read_from(path: Path, offset: int) -> bytes:
191
+ with path.open("rb") as handle:
192
+ handle.seek(offset)
193
+ return handle.read()
194
+
195
+
196
+ def decode(line: bytes) -> TranscriptEvent | None:
197
+ """Decode one complete line, treating any malformed payload as garbage."""
198
+ if not line.strip():
199
+ return None
200
+ try:
201
+ return decode_line(line)
202
+ except (KeyError, ValueError, TypeError):
203
+ return None
204
+
205
+
206
+ def remember(cursor: TailCursor, uuid: EventUuid) -> None:
207
+ cursor.seen[uuid] = None
208
+ while len(cursor.seen) > SEEN_LIMIT:
209
+ del cursor.seen[next(iter(cursor.seen))]
210
+
211
+
212
+ def session_of(cursor: TailCursor, path: Path, event: TranscriptEvent) -> SessionId:
213
+ """The session ``event`` belongs to, cached on the file's cursor.
214
+
215
+ Prefers the event's own envelope — meta-bearing events and
216
+ :class:`~cc_transcript.models.ModeEvent` carry the session id — then the
217
+ cursor's cached value, then the path convention: the
218
+ ``<session_id>.jsonl`` stem, or the ``<session_id>/subagents/`` parent
219
+ directory for sidechain files.
220
+ """
221
+ session = event_session(event) or cursor.session_id or path_session_id(path)
222
+ cursor.session_id = session
223
+ return session
224
+
225
+
226
+ def event_session(event: TranscriptEvent) -> SessionId | None:
227
+ match event:
228
+ case ModeEvent(session_id=session_id):
229
+ return session_id
230
+ case _:
231
+ return meta.session_id if (meta := event_meta(event)) is not None else None
232
+
233
+
234
+ def path_session_id(path: Path) -> SessionId:
235
+ return SessionId(path.parent.parent.name if is_subagent_path(path) else path.stem)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "cc-transcript"
3
- version = "12.0.0"
3
+ version = "12.1.1"
4
4
  description = "Grep every Claude Code session you've ever run."
5
5
  readme = "README.md"
6
6
  license = "PolyForm-Noncommercial-1.0.0"
File without changes
File without changes