cc-transcript 12.0.0__tar.gz → 12.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/PKG-INFO +1 -1
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/cli.py +56 -2
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/discovery.py +9 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/__init__.py +1 -0
- cc_transcript-12.1.0/cc_transcript/mining/sampling.py +109 -0
- cc_transcript-12.1.0/cc_transcript/watch.py +235 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/pyproject.toml +1 -1
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/Cargo.lock +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/Cargo.toml +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/LICENSE +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/README.md +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/__init__.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/__main__.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/_parser_rs.pyi +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/activity.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/activity_probe.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/backend.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/builders.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/command.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/context.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/corrections.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/corrections_cli.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/cost.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/decisions.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/disktruth.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/evidence.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/extract/__init__.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/extract/correct.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/facts.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/filterspec.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/ids.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/judge/__init__.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/judge/llm.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/judge/similar.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/judge/verdicts.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/ledger.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/candidates.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/confidence.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/engine.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/filterspec.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/formats.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/signals.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/sourcekind.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/spec.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/mining/store.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/models.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/notifications.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/parser.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/py.typed +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/query.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/render.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/rust.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/__init__.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/buckets.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/data/afinn-en-165.tsv +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/data/domain_overrides.tsv +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/engine.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/lexicon.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/scorespec.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/store.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/tools.py +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/Cargo.toml +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/build.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/data/command_prefix_pins.tsv +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/rustfmt.toml +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/activity.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/command.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/event.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/filter.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/generated/command.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/generated/mining.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/generated/mod.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/generated/protocol.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/generated/unicode.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/lexicon.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/lib.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/mining.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/model.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/parse.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/protocol.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/python.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/score.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/types.rs +0 -0
- {cc_transcript-12.0.0 → cc_transcript-12.1.0}/rust/src/value.rs +0 -0
|
@@ -26,11 +26,14 @@ from cc_transcript.builders import (
|
|
|
26
26
|
)
|
|
27
27
|
from cc_transcript.corrections_cli import corrections
|
|
28
28
|
from cc_transcript.discovery import CLAUDE_PROJECTS_DIR, TranscriptDiscovery, find_transcript_sync
|
|
29
|
-
from cc_transcript.
|
|
29
|
+
from cc_transcript.facts import command_prefix_counts, mcp_summary, tool_facts
|
|
30
|
+
from cc_transcript.filterspec import ASSISTANTS, USERS, EventKind, event_kind, event_meta, keep, tool_names
|
|
30
31
|
from cc_transcript.ids import SessionId, tool_digest
|
|
31
32
|
from cc_transcript.models import AssistantEvent, ToolResultBlock, ToolUseBlock, UserEvent
|
|
32
33
|
from cc_transcript.parser import TranscriptParser
|
|
33
34
|
from cc_transcript.render import (
|
|
35
|
+
BLANK_TIME,
|
|
36
|
+
TAGS,
|
|
34
37
|
WHERE_ALL,
|
|
35
38
|
Budget,
|
|
36
39
|
collect_stats,
|
|
@@ -39,6 +42,7 @@ from cc_transcript.render import (
|
|
|
39
42
|
denial_line,
|
|
40
43
|
display_path,
|
|
41
44
|
event_dict,
|
|
45
|
+
event_payload,
|
|
42
46
|
fact_dict,
|
|
43
47
|
fact_line,
|
|
44
48
|
haystack,
|
|
@@ -49,9 +53,10 @@ from cc_transcript.render import (
|
|
|
49
53
|
render_tool_call,
|
|
50
54
|
stats_dict,
|
|
51
55
|
transcript_header,
|
|
56
|
+
truncate,
|
|
52
57
|
)
|
|
53
|
-
from cc_transcript.facts import command_prefix_counts, mcp_summary, tool_facts
|
|
54
58
|
from cc_transcript.tools import file_path_of, parse_tool_call, tool_name_matches
|
|
59
|
+
from cc_transcript.watch import watch
|
|
55
60
|
|
|
56
61
|
if TYPE_CHECKING:
|
|
57
62
|
from collections.abc import Iterable, Mapping, Sequence
|
|
@@ -61,6 +66,7 @@ if TYPE_CHECKING:
|
|
|
61
66
|
from cc_transcript.facts import ToolFact
|
|
62
67
|
from cc_transcript.filterspec import FilterSpec
|
|
63
68
|
from cc_transcript.models import EntryMeta, ToolUseId, TranscriptEvent
|
|
69
|
+
from cc_transcript.watch import WatchEvent
|
|
64
70
|
|
|
65
71
|
type Row = tuple[int, TranscriptEvent]
|
|
66
72
|
|
|
@@ -280,6 +286,29 @@ def slice_line(meta: EntryMeta, block: ToolUseBlock) -> dict[str, Any]:
|
|
|
280
286
|
}
|
|
281
287
|
|
|
282
288
|
|
|
289
|
+
|
|
290
|
+
def watch_dict(item: WatchEvent) -> dict[str, Any]:
|
|
291
|
+
meta = event_meta(item.event)
|
|
292
|
+
kind = event_kind(item.event)
|
|
293
|
+
return {
|
|
294
|
+
"path": str(item.path),
|
|
295
|
+
"session_id": item.session_id,
|
|
296
|
+
"is_sidechain": item.is_sidechain,
|
|
297
|
+
"uuid": meta.uuid if meta is not None else None,
|
|
298
|
+
"kind": kind,
|
|
299
|
+
"role": kind if kind in ("user", "assistant") else None,
|
|
300
|
+
"preview": truncate(event_payload(item.event, names={}, width=120, thinking=False), 120),
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def watch_line(item: WatchEvent) -> str:
|
|
305
|
+
meta = event_meta(item.event)
|
|
306
|
+
time = meta.timestamp.strftime("%H:%M:%S") if meta is not None else BLANK_TIME
|
|
307
|
+
tag = TAGS[event_kind(item.event)] + ("*" if item.is_sidechain else "")
|
|
308
|
+
payload = event_payload(item.event, names={}, width=100, thinking=False)
|
|
309
|
+
return f"{time} {item.session_id[:8]} {tag:<5} {payload}".rstrip()
|
|
310
|
+
|
|
311
|
+
|
|
283
312
|
@click.group()
|
|
284
313
|
@click.version_option(package_name="cc-transcript")
|
|
285
314
|
def cli() -> None:
|
|
@@ -672,3 +701,28 @@ def digest(check: Path | None) -> None:
|
|
|
672
701
|
),
|
|
673
702
|
)
|
|
674
703
|
)
|
|
704
|
+
@cli.command("watch")
|
|
705
|
+
@click.option(
|
|
706
|
+
"--root",
|
|
707
|
+
"roots",
|
|
708
|
+
multiple=True,
|
|
709
|
+
type=click.Path(file_okay=False, path_type=Path),
|
|
710
|
+
help="Projects directory to tail; repeatable [default: ~/.claude/projects].",
|
|
711
|
+
)
|
|
712
|
+
@click.option("--poll", default=1.0, show_default=True, help="Seconds between filesystem polls.")
|
|
713
|
+
@click.option("--from-start", is_flag=True, help="Replay preexisting transcript content instead of tailing from EOF.")
|
|
714
|
+
@click.option("--json", "as_json", is_flag=True, help="Emit one NDJSON object per event.")
|
|
715
|
+
def watch_(roots: tuple[Path, ...], poll: float, from_start: bool, as_json: bool) -> None:
|
|
716
|
+
"""Tail transcripts live, one line per newly appended event, until interrupted."""
|
|
717
|
+
|
|
718
|
+
async def run() -> None:
|
|
719
|
+
async for item in watch(roots or (CLAUDE_PROJECTS_DIR,), poll=poll, from_start=from_start):
|
|
720
|
+
click.echo(orjson.dumps(watch_dict(item)) if as_json else watch_line(item))
|
|
721
|
+
|
|
722
|
+
try:
|
|
723
|
+
anyio.run(run)
|
|
724
|
+
except KeyboardInterrupt:
|
|
725
|
+
return
|
|
726
|
+
except BrokenPipeError:
|
|
727
|
+
os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno())
|
|
728
|
+
raise SystemExit(0) from None
|
|
@@ -151,6 +151,15 @@ async def find_transcript(session_id: SessionId, *, root: Path | None = None) ->
|
|
|
151
151
|
return await anyio.to_thread.run_sync(partial(find_transcript_sync, session_id, root=root))
|
|
152
152
|
|
|
153
153
|
|
|
154
|
+
def is_subagent_path(path: Path) -> bool:
|
|
155
|
+
"""Whether ``path`` names a subagent sidechain transcript.
|
|
156
|
+
|
|
157
|
+
Matches the ``agent-<tool_use_id>.jsonl`` naming convention that
|
|
158
|
+
:func:`subagent_paths` discovers.
|
|
159
|
+
"""
|
|
160
|
+
return path.suffix == ".jsonl" and path.name.startswith("agent-")
|
|
161
|
+
|
|
162
|
+
|
|
154
163
|
def subagent_paths(path: Path) -> tuple[Path, ...]:
|
|
155
164
|
"""Sidechain transcript files spawned by the session transcript at ``path``.
|
|
156
165
|
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Deterministic negative-window sampling over a session's completed turns.
|
|
2
|
+
|
|
3
|
+
:func:`sample_windows` draws "the user did not steer here" negatives: durable
|
|
4
|
+
:class:`~cc_transcript.context.ContextWindow` captures anchored on completed
|
|
5
|
+
turns, reproducible for a given seed and session, kept clear of known
|
|
6
|
+
positives by an exclusion radius. Negative windows carry no trigger — the
|
|
7
|
+
sampled turn folds into ``before`` — so they render byte-compatibly with
|
|
8
|
+
positive steering windows, whose user-steer trigger is likewise excluded from
|
|
9
|
+
model input: both shapes read as the turns up to the moment being judged.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import random
|
|
15
|
+
from dataclasses import replace
|
|
16
|
+
from typing import TYPE_CHECKING
|
|
17
|
+
|
|
18
|
+
from cc_transcript.context import capture_window
|
|
19
|
+
from cc_transcript.filterspec import event_meta
|
|
20
|
+
from cc_transcript.ids import EventRef
|
|
21
|
+
|
|
22
|
+
if TYPE_CHECKING:
|
|
23
|
+
from collections.abc import Iterable
|
|
24
|
+
|
|
25
|
+
from cc_transcript.activity import SessionActivity, Turn
|
|
26
|
+
from cc_transcript.context import ContextWindow
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def sample_windows(
|
|
30
|
+
activity: SessionActivity,
|
|
31
|
+
*,
|
|
32
|
+
n: int,
|
|
33
|
+
exclude: Iterable[EventRef] = (),
|
|
34
|
+
exclusion_radius: int = 6,
|
|
35
|
+
seed: int = 0,
|
|
36
|
+
before: int = 6,
|
|
37
|
+
after: int = 2,
|
|
38
|
+
preview_chars: int = 200,
|
|
39
|
+
) -> list[ContextWindow]:
|
|
40
|
+
"""Sample up to ``n`` triggerless context windows as steering negatives.
|
|
41
|
+
|
|
42
|
+
Each window anchors on a completed turn — the agent acted and the user's
|
|
43
|
+
next prompt was not a steer we know about — and folds that turn into the
|
|
44
|
+
window's context: ``trigger`` is None, ``before`` ends at the sampled
|
|
45
|
+
turn and keeps at most ``before`` turns, and ``after`` is unchanged. The
|
|
46
|
+
``anchor`` stays the sampled turn's first meta-bearing event, so
|
|
47
|
+
consumers key negatives exactly like positives.
|
|
48
|
+
|
|
49
|
+
Candidates are every turn carrying at least one event with resolvable
|
|
50
|
+
meta (the anchor), except the session's final turn, which may still be in
|
|
51
|
+
flight. Every candidate in the ``exclusion_radius`` turns leading up to an
|
|
52
|
+
``exclude`` ref's turn is dropped; refs that no longer resolve are
|
|
53
|
+
ignored. Sampling is deterministic:
|
|
54
|
+
``random.Random(f"{seed}:{session_id}")`` draws from the candidates in
|
|
55
|
+
turn order, so one seed always yields the same windows for a session.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
activity: The lifted session to sample from.
|
|
59
|
+
n: The maximum number of windows to return.
|
|
60
|
+
exclude: Anchors of known positives to keep clear of.
|
|
61
|
+
exclusion_radius: How many turns before each excluded turn to drop
|
|
62
|
+
(the excluded turn itself included). Turns after an excluded turn
|
|
63
|
+
stay eligible — once the user has steered, letting the agent run
|
|
64
|
+
again is a genuine negative; only the pre-steer approach, which
|
|
65
|
+
positive-window rewinds occupy, is label-conflicted.
|
|
66
|
+
seed: The determinism seed, mixed with the session id.
|
|
67
|
+
before: How many turns each window's folded ``before`` keeps, ending
|
|
68
|
+
at the sampled turn.
|
|
69
|
+
after: How many turns after each sampled turn to capture.
|
|
70
|
+
preview_chars: The per-chunk preview budget persisted on each window.
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
The sampled windows, sorted by sampled turn index.
|
|
74
|
+
"""
|
|
75
|
+
excluded = {turn.index for ref in exclude if (turn := activity.turn_of(ref)) is not None}
|
|
76
|
+
candidates = [
|
|
77
|
+
(turn.index, anchor)
|
|
78
|
+
for turn in activity.turns[:-1]
|
|
79
|
+
if all(not (0 <= index - turn.index <= exclusion_radius) for index in excluded)
|
|
80
|
+
if (anchor := turn_anchor(turn)) is not None
|
|
81
|
+
]
|
|
82
|
+
rng = random.Random(f"{seed}:{activity.session_id}")
|
|
83
|
+
chosen = rng.sample(candidates, min(n, len(candidates)))
|
|
84
|
+
return [
|
|
85
|
+
fold_trigger(
|
|
86
|
+
capture_window(activity, anchor, before=before, after=after, preview_chars=preview_chars),
|
|
87
|
+
keep=before,
|
|
88
|
+
)
|
|
89
|
+
for _, anchor in sorted(chosen, key=lambda pair: pair[0])
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def fold_trigger(window: ContextWindow, *, keep: int) -> ContextWindow:
|
|
94
|
+
"""Fold ``window``'s trigger into ``before`` — the negative shape has none.
|
|
95
|
+
|
|
96
|
+
Returns:
|
|
97
|
+
The window with ``trigger`` None and ``before`` ending at the old
|
|
98
|
+
trigger, truncated to the last ``keep`` turns.
|
|
99
|
+
"""
|
|
100
|
+
folded = (*window.before, *(() if window.trigger is None else (window.trigger,)))
|
|
101
|
+
return replace(window, before=folded[-keep:] if keep > 0 else (), trigger=None)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def turn_anchor(turn: Turn) -> EventRef | None:
|
|
105
|
+
"""The reference to ``turn``'s first meta-bearing event, or None without one."""
|
|
106
|
+
return next(
|
|
107
|
+
(EventRef(meta.session_id, meta.uuid) for event in turn.events if (meta := event_meta(event)) is not None),
|
|
108
|
+
None,
|
|
109
|
+
)
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
"""Live transcript tailing: poll the projects tree, yield each appended event once.
|
|
2
|
+
|
|
3
|
+
A byte-offset tailer over the ``*.jsonl`` transcripts: every file gets a
|
|
4
|
+
:class:`TailCursor`, each poll reads only what appended since the last one
|
|
5
|
+
(open-read-close, never holding a descriptor), complete lines decode through
|
|
6
|
+
the parser's per-line decode, and a bounded per-file uuid set keeps compaction
|
|
7
|
+
rewrites and replays from double-firing. All progression lives in
|
|
8
|
+
:func:`tick` — one deterministic step over a :class:`TailState`, directly
|
|
9
|
+
drivable by tests and embedders — and :func:`watch` is the thin poll-forever
|
|
10
|
+
loop over it.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import TYPE_CHECKING
|
|
18
|
+
|
|
19
|
+
import anyio
|
|
20
|
+
import anyio.to_thread
|
|
21
|
+
|
|
22
|
+
from cc_transcript.discovery import CLAUDE_PROJECTS_DIR, is_subagent_path
|
|
23
|
+
from cc_transcript.filterspec import event_meta
|
|
24
|
+
from cc_transcript.ids import SessionId
|
|
25
|
+
from cc_transcript.models import ModeEvent
|
|
26
|
+
from cc_transcript.parser import decode_line
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
import os
|
|
30
|
+
from collections.abc import AsyncIterator, Sequence
|
|
31
|
+
|
|
32
|
+
from cc_transcript.ids import EventUuid
|
|
33
|
+
from cc_transcript.models import TranscriptEvent
|
|
34
|
+
|
|
35
|
+
SEEN_LIMIT = 4096
|
|
36
|
+
"""How many yielded event uuids each file's dedupe set retains."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class WatchEvent:
|
|
41
|
+
"""One transcript event freshly appended to a watched file.
|
|
42
|
+
|
|
43
|
+
Attributes:
|
|
44
|
+
path: The transcript file the event was read from.
|
|
45
|
+
session_id: The session the event belongs to — from the event's own
|
|
46
|
+
envelope when it carries one, else the file's last-known session,
|
|
47
|
+
else derived from the transcript path.
|
|
48
|
+
is_sidechain: Whether the file is a subagent sidechain transcript.
|
|
49
|
+
event: The parsed transcript event.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
path: Path
|
|
53
|
+
session_id: SessionId
|
|
54
|
+
is_sidechain: bool
|
|
55
|
+
event: TranscriptEvent
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(slots=True)
|
|
59
|
+
class TailCursor:
|
|
60
|
+
"""One watched file's tail progress.
|
|
61
|
+
|
|
62
|
+
Attributes:
|
|
63
|
+
offset: Bytes consumed so far — always the end of the last complete
|
|
64
|
+
line, so a partial trailing line stays unconsumed until its
|
|
65
|
+
newline arrives.
|
|
66
|
+
size: The file size at the last processed stat.
|
|
67
|
+
mtime: The file mtime at the last processed stat.
|
|
68
|
+
session_id: The session id cached from the first decoded event that
|
|
69
|
+
carried one.
|
|
70
|
+
seen: Event uuids already yielded, insertion-ordered and bounded at
|
|
71
|
+
:data:`SEEN_LIMIT`.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
offset: int
|
|
75
|
+
size: int
|
|
76
|
+
mtime: float
|
|
77
|
+
session_id: SessionId | None = None
|
|
78
|
+
seen: dict[EventUuid, None] = field(default_factory=dict)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(slots=True)
|
|
82
|
+
class TailState:
|
|
83
|
+
"""The tailer's whole mutable state: one cursor per discovered file.
|
|
84
|
+
|
|
85
|
+
Attributes:
|
|
86
|
+
cursors: Tail progress per transcript file.
|
|
87
|
+
primed: Whether the initial discovery pass has run — files first seen
|
|
88
|
+
on that pass are history, files appearing later are new content.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
cursors: dict[Path, TailCursor] = field(default_factory=dict)
|
|
92
|
+
primed: bool = False
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
async def watch(
|
|
96
|
+
roots: Sequence[Path] = (CLAUDE_PROJECTS_DIR,),
|
|
97
|
+
*,
|
|
98
|
+
poll: float = 1.0,
|
|
99
|
+
from_start: bool = False,
|
|
100
|
+
) -> AsyncIterator[WatchEvent]:
|
|
101
|
+
"""Tail every transcript under ``roots`` forever, yielding appended events.
|
|
102
|
+
|
|
103
|
+
An async generator that never returns on its own: each iteration drains
|
|
104
|
+
one :func:`tick` and sleeps ``poll`` seconds. Content predating the first
|
|
105
|
+
tick is skipped unless ``from_start``; everything appended afterwards is
|
|
106
|
+
yielded exactly once, with sidechain transcripts flagged on the event.
|
|
107
|
+
"""
|
|
108
|
+
state = TailState()
|
|
109
|
+
while True:
|
|
110
|
+
for event in await tick(state, roots, from_start=from_start):
|
|
111
|
+
yield event
|
|
112
|
+
await anyio.sleep(poll)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
async def tick(state: TailState, roots: Sequence[Path], *, from_start: bool = False) -> list[WatchEvent]:
|
|
116
|
+
"""Run one poll step: discover changes under ``roots`` and drain them.
|
|
117
|
+
|
|
118
|
+
A file is re-read only when its size or mtime changed — both are compared,
|
|
119
|
+
because mtime granularity can hide rapid appends. A file first seen on the
|
|
120
|
+
priming pass starts at end-of-file unless ``from_start``, so daemon starts
|
|
121
|
+
never replay history; a file appearing on a later pass starts at byte 0,
|
|
122
|
+
since its whole content is new. A file whose size fell below the cursor
|
|
123
|
+
was rewritten (compaction): the cursor resets to 0 and its dedupe set
|
|
124
|
+
clears. The cursor only ever advances past the last complete line — a
|
|
125
|
+
partial trailing line waits, and lines the decoder rejects are skipped.
|
|
126
|
+
|
|
127
|
+
Returns:
|
|
128
|
+
The newly appended events, files in path order, lines in file order.
|
|
129
|
+
"""
|
|
130
|
+
stats = await scan(roots)
|
|
131
|
+
priming = not state.primed
|
|
132
|
+
state.primed = True
|
|
133
|
+
events: list[WatchEvent] = []
|
|
134
|
+
for path in sorted(stats):
|
|
135
|
+
stat = stats[path]
|
|
136
|
+
if (cursor := state.cursors.get(path)) is None:
|
|
137
|
+
skip_history = priming and not from_start
|
|
138
|
+
cursor = state.cursors[path] = TailCursor(
|
|
139
|
+
offset=stat.st_size if skip_history else 0,
|
|
140
|
+
size=stat.st_size if skip_history else -1,
|
|
141
|
+
mtime=stat.st_mtime if skip_history else -1.0,
|
|
142
|
+
)
|
|
143
|
+
if stat.st_size < cursor.offset:
|
|
144
|
+
cursor.offset = 0
|
|
145
|
+
cursor.seen.clear()
|
|
146
|
+
elif stat.st_size == cursor.size and stat.st_mtime == cursor.mtime:
|
|
147
|
+
continue
|
|
148
|
+
try:
|
|
149
|
+
chunk = await anyio.to_thread.run_sync(read_from, path, cursor.offset)
|
|
150
|
+
except OSError:
|
|
151
|
+
continue
|
|
152
|
+
cursor.size, cursor.mtime = stat.st_size, stat.st_mtime
|
|
153
|
+
complete, _, partial = chunk.rpartition(b"\n")
|
|
154
|
+
cursor.offset += len(chunk) - len(partial)
|
|
155
|
+
for line in complete.split(b"\n"):
|
|
156
|
+
if (event := decode(line)) is None:
|
|
157
|
+
continue
|
|
158
|
+
if (meta := event_meta(event)) is not None:
|
|
159
|
+
if meta.uuid in cursor.seen:
|
|
160
|
+
continue
|
|
161
|
+
remember(cursor, meta.uuid)
|
|
162
|
+
events.append(
|
|
163
|
+
WatchEvent(
|
|
164
|
+
path=path,
|
|
165
|
+
session_id=session_of(cursor, path, event),
|
|
166
|
+
is_sidechain=is_subagent_path(path),
|
|
167
|
+
event=event,
|
|
168
|
+
)
|
|
169
|
+
)
|
|
170
|
+
return events
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
async def scan(roots: Sequence[Path]) -> dict[Path, os.stat_result]:
|
|
174
|
+
"""Stat every transcript under ``roots``, skipping macOS resource forks."""
|
|
175
|
+
found: dict[Path, os.stat_result] = {}
|
|
176
|
+
for root in roots:
|
|
177
|
+
base = anyio.Path(root)
|
|
178
|
+
if not await base.exists():
|
|
179
|
+
continue
|
|
180
|
+
async for entry in base.rglob("*.jsonl"):
|
|
181
|
+
if entry.name.startswith("._"):
|
|
182
|
+
continue
|
|
183
|
+
try:
|
|
184
|
+
found[Path(entry)] = await entry.stat()
|
|
185
|
+
except OSError:
|
|
186
|
+
continue
|
|
187
|
+
return found
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def read_from(path: Path, offset: int) -> bytes:
|
|
191
|
+
with path.open("rb") as handle:
|
|
192
|
+
handle.seek(offset)
|
|
193
|
+
return handle.read()
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def decode(line: bytes) -> TranscriptEvent | None:
|
|
197
|
+
"""Decode one complete line, treating any malformed payload as garbage."""
|
|
198
|
+
if not line.strip():
|
|
199
|
+
return None
|
|
200
|
+
try:
|
|
201
|
+
return decode_line(line)
|
|
202
|
+
except (KeyError, ValueError, TypeError):
|
|
203
|
+
return None
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def remember(cursor: TailCursor, uuid: EventUuid) -> None:
|
|
207
|
+
cursor.seen[uuid] = None
|
|
208
|
+
while len(cursor.seen) > SEEN_LIMIT:
|
|
209
|
+
del cursor.seen[next(iter(cursor.seen))]
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def session_of(cursor: TailCursor, path: Path, event: TranscriptEvent) -> SessionId:
|
|
213
|
+
"""The session ``event`` belongs to, cached on the file's cursor.
|
|
214
|
+
|
|
215
|
+
Prefers the event's own envelope — meta-bearing events and
|
|
216
|
+
:class:`~cc_transcript.models.ModeEvent` carry the session id — then the
|
|
217
|
+
cursor's cached value, then the path convention: the
|
|
218
|
+
``<session_id>.jsonl`` stem, or the ``<session_id>/subagents/`` parent
|
|
219
|
+
directory for sidechain files.
|
|
220
|
+
"""
|
|
221
|
+
session = event_session(event) or cursor.session_id or path_session_id(path)
|
|
222
|
+
cursor.session_id = session
|
|
223
|
+
return session
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def event_session(event: TranscriptEvent) -> SessionId | None:
|
|
227
|
+
match event:
|
|
228
|
+
case ModeEvent(session_id=session_id):
|
|
229
|
+
return session_id
|
|
230
|
+
case _:
|
|
231
|
+
return meta.session_id if (meta := event_meta(event)) is not None else None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def path_session_id(path: Path) -> SessionId:
|
|
235
|
+
return SessionId(path.parent.parent.name if is_subagent_path(path) else path.stem)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cc_transcript-12.0.0 → cc_transcript-12.1.0}/cc_transcript/sentiment/data/domain_overrides.tsv
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|