cc-transcript 10.6.0__tar.gz → 10.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/PKG-INFO +1 -1
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/__init__.py +4 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/_parser_rs.pyi +3 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/corrections.py +47 -93
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/decisions.py +37 -82
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/discovery.py +20 -1
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/filterspec.py +29 -26
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/judge/verdicts.py +12 -2
- cc_transcript-10.8.0/cc_transcript/ledger.py +87 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/signals.py +13 -22
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/spec.py +2 -8
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/models.py +55 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/query.py +12 -3
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/sentiment/buckets.py +9 -2
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/pyproject.toml +1 -1
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/Cargo.toml +2 -1
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/build.rs +8 -3
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/command.rs +5 -15
- cc_transcript-10.8.0/rust/src/generated/command.rs +27 -0
- cc_transcript-10.8.0/rust/src/generated/mining.rs +19 -0
- cc_transcript-10.8.0/rust/src/generated/mod.rs +5 -0
- cc_transcript-10.8.0/rust/src/generated/protocol.rs +10 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/lib.rs +2 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/mining.rs +7 -45
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/protocol.rs +9 -18
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/python.rs +37 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/Cargo.lock +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/Cargo.toml +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/LICENSE +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/README.md +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/__main__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/activity.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/activity_probe.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/backend.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/builders.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/cli.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/command.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/context.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/corrections_cli.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/cost.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/disktruth.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/evidence.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/extract/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/extract/correct.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/facts.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/ids.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/judge/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/judge/llm.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/judge/similar.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/candidates.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/confidence.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/engine.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/filterspec.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/formats.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/sourcekind.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/mining/store.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/notifications.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/parser.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/py.typed +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/render.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/rust.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/sentiment/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/sentiment/engine.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/sentiment/lexicon.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/sentiment/scorespec.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/store.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/cc_transcript/tools.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/data/afinn-en-165.tsv +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/data/domain_overrides.tsv +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/activity.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/event.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/filter.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/lexicon.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/model.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/parse.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/score.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/types.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.8.0}/rust/src/value.rs +0 -0
|
@@ -83,6 +83,7 @@ EXPORTS: dict[str, str] = {
|
|
|
83
83
|
"Plugin",
|
|
84
84
|
"PrintMessage",
|
|
85
85
|
"PrintResult",
|
|
86
|
+
"Question",
|
|
86
87
|
"ServerToolUse",
|
|
87
88
|
"SystemEvent",
|
|
88
89
|
"TextBlock",
|
|
@@ -522,6 +523,9 @@ if TYPE_CHECKING:
|
|
|
522
523
|
from cc_transcript.models import (
|
|
523
524
|
PrintResult as PrintResult,
|
|
524
525
|
)
|
|
526
|
+
from cc_transcript.models import (
|
|
527
|
+
Question as Question,
|
|
528
|
+
)
|
|
525
529
|
from cc_transcript.models import (
|
|
526
530
|
ServerToolUse as ServerToolUse,
|
|
527
531
|
)
|
|
@@ -35,6 +35,9 @@ def lexicon_has_hit(text: str, floor: int, want_negative: bool, /) -> bool:
|
|
|
35
35
|
def lexicon_overrides() -> list[tuple[str, int]]:
|
|
36
36
|
"""The embedded domain-override entries (for the single-source drift guard)."""
|
|
37
37
|
|
|
38
|
+
def embedded_literals() -> dict[str, str | float | list[str]]:
|
|
39
|
+
"""The generated protocol, mining, and command literals keyed ``module.NAME`` (for the single-source drift guard)."""
|
|
40
|
+
|
|
38
41
|
def score_short_circuit(spec_json: str, buckets: list[list[str]], /) -> list[int | None]:
|
|
39
42
|
"""Per bucket, the first short-circuit stage's score over its user texts, else None."""
|
|
40
43
|
|
|
@@ -19,8 +19,9 @@ from __future__ import annotations
|
|
|
19
19
|
import json
|
|
20
20
|
import sqlite3
|
|
21
21
|
from dataclasses import dataclass, field
|
|
22
|
-
from
|
|
23
|
-
|
|
22
|
+
from typing import TYPE_CHECKING, Literal
|
|
23
|
+
|
|
24
|
+
from cc_transcript.ledger import SyncLedger
|
|
24
25
|
|
|
25
26
|
if TYPE_CHECKING:
|
|
26
27
|
from collections.abc import Mapping
|
|
@@ -57,12 +58,6 @@ CREATE INDEX IF NOT EXISTS idx_corrections_incorrect_digest ON corrections (inco
|
|
|
57
58
|
|
|
58
59
|
Origin = Literal["session", "git", "review"]
|
|
59
60
|
|
|
60
|
-
CORRECTION_COLUMNS = (
|
|
61
|
-
"ts_ms, session_id, source, anchor_uuid, incorrect_digest, incorrect_file, incorrect_old, "
|
|
62
|
-
"incorrect_new, correction_origin, correction_file, correction_old, correction_new, "
|
|
63
|
-
"correction_commit, correction_text, overlap, detail_json"
|
|
64
|
-
)
|
|
65
|
-
|
|
66
61
|
|
|
67
62
|
@dataclass(frozen=True, slots=True)
|
|
68
63
|
class Correction:
|
|
@@ -111,7 +106,7 @@ class Correction:
|
|
|
111
106
|
detail: Mapping[str, Any] = field(default_factory=dict)
|
|
112
107
|
|
|
113
108
|
|
|
114
|
-
class CorrectionLog:
|
|
109
|
+
class CorrectionLog(SyncLedger[Correction]):
|
|
115
110
|
"""The ``corrections`` ledger at ``~/.cc-transcript/corrections.db``.
|
|
116
111
|
|
|
117
112
|
Opened in WAL mode with a busy timeout because writers across the family
|
|
@@ -124,66 +119,46 @@ class CorrectionLog:
|
|
|
124
119
|
>>> log.by_digest(session_id, incorrect_digest=digest)
|
|
125
120
|
"""
|
|
126
121
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
""
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
""
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
return cls(conn)
|
|
149
|
-
|
|
150
|
-
def append(self, correction: Correction) -> None:
|
|
151
|
-
"""Appends ``correction`` as a single ``INSERT OR IGNORE``.
|
|
152
|
-
|
|
153
|
-
Idempotent on the UNIQUE key ``(session_id, anchor_uuid,
|
|
154
|
-
incorrect_digest)`` — re-harvesting the same edit, across reruns or
|
|
155
|
-
across the pairs of one event, writes one row.
|
|
156
|
-
"""
|
|
157
|
-
self.conn.execute(
|
|
158
|
-
f"INSERT OR IGNORE INTO corrections ({CORRECTION_COLUMNS}) "
|
|
159
|
-
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
160
|
-
(
|
|
161
|
-
correction.ts_ms,
|
|
162
|
-
correction.session_id,
|
|
163
|
-
correction.source,
|
|
164
|
-
correction.anchor_uuid,
|
|
165
|
-
correction.incorrect_digest,
|
|
166
|
-
correction.incorrect_file,
|
|
167
|
-
correction.incorrect_old,
|
|
168
|
-
correction.incorrect_new,
|
|
169
|
-
correction.correction_origin,
|
|
170
|
-
correction.correction_file,
|
|
171
|
-
correction.correction_old,
|
|
172
|
-
correction.correction_new,
|
|
173
|
-
correction.correction_commit,
|
|
174
|
-
correction.correction_text,
|
|
175
|
-
correction.overlap,
|
|
176
|
-
json.dumps(dict(correction.detail)),
|
|
177
|
-
),
|
|
178
|
-
)
|
|
122
|
+
DDL = CORRECTIONS_DDL
|
|
123
|
+
FILENAME = "corrections.db"
|
|
124
|
+
TABLE = "corrections"
|
|
125
|
+
COLUMNS = (
|
|
126
|
+
"ts_ms",
|
|
127
|
+
"session_id",
|
|
128
|
+
"source",
|
|
129
|
+
"anchor_uuid",
|
|
130
|
+
"incorrect_digest",
|
|
131
|
+
"incorrect_file",
|
|
132
|
+
"incorrect_old",
|
|
133
|
+
"incorrect_new",
|
|
134
|
+
"correction_origin",
|
|
135
|
+
"correction_file",
|
|
136
|
+
"correction_old",
|
|
137
|
+
"correction_new",
|
|
138
|
+
"correction_commit",
|
|
139
|
+
"correction_text",
|
|
140
|
+
"overlap",
|
|
141
|
+
"detail_json",
|
|
142
|
+
)
|
|
179
143
|
|
|
180
|
-
def
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
144
|
+
def row_to_record(self, row: sqlite3.Row) -> Correction:
|
|
145
|
+
return Correction(
|
|
146
|
+
ts_ms=row["ts_ms"],
|
|
147
|
+
session_id=row["session_id"],
|
|
148
|
+
source=row["source"],
|
|
149
|
+
anchor_uuid=row["anchor_uuid"],
|
|
150
|
+
incorrect_digest=row["incorrect_digest"],
|
|
151
|
+
incorrect_file=row["incorrect_file"],
|
|
152
|
+
incorrect_old=row["incorrect_old"],
|
|
153
|
+
incorrect_new=row["incorrect_new"],
|
|
154
|
+
correction_origin=row["correction_origin"],
|
|
155
|
+
correction_file=row["correction_file"],
|
|
156
|
+
correction_old=row["correction_old"],
|
|
157
|
+
correction_new=row["correction_new"],
|
|
158
|
+
correction_commit=row["correction_commit"],
|
|
159
|
+
correction_text=row["correction_text"],
|
|
160
|
+
overlap=row["overlap"],
|
|
161
|
+
detail=json.loads(row["detail_json"]),
|
|
187
162
|
)
|
|
188
163
|
|
|
189
164
|
def for_repo(self, repo: str) -> tuple[Correction, ...]:
|
|
@@ -193,7 +168,7 @@ class CorrectionLog:
|
|
|
193
168
|
captain-hook reviewer) can pull every correction for its repo at once.
|
|
194
169
|
"""
|
|
195
170
|
return tuple(
|
|
196
|
-
|
|
171
|
+
self.row_to_record(row)
|
|
197
172
|
for row in self.conn.execute(
|
|
198
173
|
"SELECT * FROM corrections WHERE json_extract(detail_json, '$.repo') = ? ORDER BY ts_ms, id",
|
|
199
174
|
(repo,),
|
|
@@ -212,12 +187,12 @@ class CorrectionLog:
|
|
|
212
187
|
rows = self.conn.execute(
|
|
213
188
|
"SELECT * FROM corrections WHERE ts_ms > ? AND source = ? ORDER BY ts_ms, id", (ts_ms, source)
|
|
214
189
|
)
|
|
215
|
-
return tuple(
|
|
190
|
+
return tuple(self.row_to_record(row) for row in rows)
|
|
216
191
|
|
|
217
192
|
def for_anchor(self, session_id: SessionId, anchor_uuid: EventUuid) -> tuple[Correction, ...]:
|
|
218
193
|
"""The corrections harvested around one feedback ``anchor_uuid``."""
|
|
219
194
|
return tuple(
|
|
220
|
-
|
|
195
|
+
self.row_to_record(row)
|
|
221
196
|
for row in self.conn.execute(
|
|
222
197
|
"SELECT * FROM corrections WHERE session_id = ? AND anchor_uuid = ? ORDER BY ts_ms, id",
|
|
223
198
|
(session_id, anchor_uuid),
|
|
@@ -232,30 +207,9 @@ class CorrectionLog:
|
|
|
232
207
|
corrected.
|
|
233
208
|
"""
|
|
234
209
|
return tuple(
|
|
235
|
-
|
|
210
|
+
self.row_to_record(row)
|
|
236
211
|
for row in self.conn.execute(
|
|
237
212
|
"SELECT * FROM corrections WHERE session_id = ? AND incorrect_digest = ? ORDER BY ts_ms, id",
|
|
238
213
|
(session_id, incorrect_digest),
|
|
239
214
|
)
|
|
240
215
|
)
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
def correction_of(row: sqlite3.Row) -> Correction:
|
|
244
|
-
return Correction(
|
|
245
|
-
ts_ms=row["ts_ms"],
|
|
246
|
-
session_id=row["session_id"],
|
|
247
|
-
source=row["source"],
|
|
248
|
-
anchor_uuid=row["anchor_uuid"],
|
|
249
|
-
incorrect_digest=row["incorrect_digest"],
|
|
250
|
-
incorrect_file=row["incorrect_file"],
|
|
251
|
-
incorrect_old=row["incorrect_old"],
|
|
252
|
-
incorrect_new=row["incorrect_new"],
|
|
253
|
-
correction_origin=row["correction_origin"],
|
|
254
|
-
correction_file=row["correction_file"],
|
|
255
|
-
correction_old=row["correction_old"],
|
|
256
|
-
correction_new=row["correction_new"],
|
|
257
|
-
correction_commit=row["correction_commit"],
|
|
258
|
-
correction_text=row["correction_text"],
|
|
259
|
-
overlap=row["overlap"],
|
|
260
|
-
detail=json.loads(row["detail_json"]),
|
|
261
|
-
)
|
|
@@ -12,8 +12,9 @@ from __future__ import annotations
|
|
|
12
12
|
import json
|
|
13
13
|
import sqlite3
|
|
14
14
|
from dataclasses import dataclass, field
|
|
15
|
-
from
|
|
16
|
-
|
|
15
|
+
from typing import TYPE_CHECKING, Literal
|
|
16
|
+
|
|
17
|
+
from cc_transcript.ledger import SyncLedger
|
|
17
18
|
|
|
18
19
|
if TYPE_CHECKING:
|
|
19
20
|
from collections.abc import Mapping
|
|
@@ -48,11 +49,6 @@ CREATE INDEX IF NOT EXISTS idx_decisions_source_file ON decisions (source_file);
|
|
|
48
49
|
|
|
49
50
|
Action = Literal["allow", "block", "warn", "nudge", "note"]
|
|
50
51
|
|
|
51
|
-
DECISION_COLUMNS = (
|
|
52
|
-
"ts_ms, session_id, source, kind, source_file, event, action, "
|
|
53
|
-
"tool_name, tool_digest, event_uuid, message, detail_json"
|
|
54
|
-
)
|
|
55
|
-
|
|
56
52
|
|
|
57
53
|
@dataclass(frozen=True, slots=True)
|
|
58
54
|
class Decision:
|
|
@@ -91,7 +87,7 @@ class Decision:
|
|
|
91
87
|
detail: Mapping[str, Any] = field(default_factory=dict)
|
|
92
88
|
|
|
93
89
|
|
|
94
|
-
class DecisionLog:
|
|
90
|
+
class DecisionLog(SyncLedger[Decision]):
|
|
95
91
|
"""The ``decisions`` ledger at ``~/.cc-transcript/decisions.db``.
|
|
96
92
|
|
|
97
93
|
Opened in WAL mode with a busy timeout because cc-review's Go daemon
|
|
@@ -104,62 +100,38 @@ class DecisionLog:
|
|
|
104
100
|
>>> log.attribute_tool(session_id, tool_digest=digest, near_ts_ms=ts)
|
|
105
101
|
"""
|
|
106
102
|
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
""
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
""
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
conn.row_factory = sqlite3.Row
|
|
125
|
-
conn.execute("PRAGMA journal_mode = WAL")
|
|
126
|
-
conn.execute("PRAGMA busy_timeout = 2000")
|
|
127
|
-
conn.executescript(DECISIONS_DDL)
|
|
128
|
-
return cls(conn)
|
|
129
|
-
|
|
130
|
-
def append(self, decision: Decision) -> None:
|
|
131
|
-
"""Appends ``decision`` as a single ``INSERT OR IGNORE``.
|
|
132
|
-
|
|
133
|
-
Idempotent on the UNIQUE key ``(session_id, ts_ms, source, kind,
|
|
134
|
-
tool_digest)`` when ``tool_digest`` is present; SQLite treats NULL
|
|
135
|
-
digests as distinct, so digestless rows rely on the writer not
|
|
136
|
-
re-running the same integer-ms timestamp.
|
|
137
|
-
"""
|
|
138
|
-
self.conn.execute(
|
|
139
|
-
f"INSERT OR IGNORE INTO decisions ({DECISION_COLUMNS}) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
140
|
-
(
|
|
141
|
-
decision.ts_ms,
|
|
142
|
-
decision.session_id,
|
|
143
|
-
decision.source,
|
|
144
|
-
decision.kind,
|
|
145
|
-
decision.source_file,
|
|
146
|
-
decision.event,
|
|
147
|
-
decision.action,
|
|
148
|
-
decision.tool_name,
|
|
149
|
-
decision.tool_digest,
|
|
150
|
-
decision.event_uuid,
|
|
151
|
-
decision.message,
|
|
152
|
-
json.dumps(dict(decision.detail)),
|
|
153
|
-
),
|
|
154
|
-
)
|
|
103
|
+
DDL = DECISIONS_DDL
|
|
104
|
+
FILENAME = "decisions.db"
|
|
105
|
+
TABLE = "decisions"
|
|
106
|
+
COLUMNS = (
|
|
107
|
+
"ts_ms",
|
|
108
|
+
"session_id",
|
|
109
|
+
"source",
|
|
110
|
+
"kind",
|
|
111
|
+
"source_file",
|
|
112
|
+
"event",
|
|
113
|
+
"action",
|
|
114
|
+
"tool_name",
|
|
115
|
+
"tool_digest",
|
|
116
|
+
"event_uuid",
|
|
117
|
+
"message",
|
|
118
|
+
"detail_json",
|
|
119
|
+
)
|
|
155
120
|
|
|
156
|
-
def
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
121
|
+
def row_to_record(self, row: sqlite3.Row) -> Decision:
|
|
122
|
+
return Decision(
|
|
123
|
+
ts_ms=row["ts_ms"],
|
|
124
|
+
session_id=row["session_id"],
|
|
125
|
+
source=row["source"],
|
|
126
|
+
kind=row["kind"],
|
|
127
|
+
source_file=row["source_file"],
|
|
128
|
+
event=row["event"],
|
|
129
|
+
action=row["action"],
|
|
130
|
+
tool_name=row["tool_name"],
|
|
131
|
+
tool_digest=row["tool_digest"],
|
|
132
|
+
event_uuid=row["event_uuid"],
|
|
133
|
+
message=row["message"],
|
|
134
|
+
detail=json.loads(row["detail_json"]),
|
|
163
135
|
)
|
|
164
136
|
|
|
165
137
|
def attribute_tool(
|
|
@@ -181,7 +153,7 @@ class DecisionLog:
|
|
|
181
153
|
" ORDER BY ts_ms DESC, id DESC LIMIT 1",
|
|
182
154
|
(session_id, tool_digest, near_ts_ms - window_ms, near_ts_ms),
|
|
183
155
|
).fetchone()
|
|
184
|
-
return
|
|
156
|
+
return self.row_to_record(row) if row else None
|
|
185
157
|
|
|
186
158
|
def attribute_nearest(
|
|
187
159
|
self,
|
|
@@ -209,21 +181,4 @@ class DecisionLog:
|
|
|
209
181
|
" ORDER BY ABS(ts_ms - ?), id DESC LIMIT 1",
|
|
210
182
|
(session_id, event, *kind_params, near_ts_ms - window_ms, near_ts_ms + window_ms, near_ts_ms),
|
|
211
183
|
).fetchone()
|
|
212
|
-
return
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
def decision_of(row: sqlite3.Row) -> Decision:
|
|
216
|
-
return Decision(
|
|
217
|
-
ts_ms=row["ts_ms"],
|
|
218
|
-
session_id=row["session_id"],
|
|
219
|
-
source=row["source"],
|
|
220
|
-
kind=row["kind"],
|
|
221
|
-
source_file=row["source_file"],
|
|
222
|
-
event=row["event"],
|
|
223
|
-
action=row["action"],
|
|
224
|
-
tool_name=row["tool_name"],
|
|
225
|
-
tool_digest=row["tool_digest"],
|
|
226
|
-
event_uuid=row["event_uuid"],
|
|
227
|
-
message=row["message"],
|
|
228
|
-
detail=json.loads(row["detail_json"]),
|
|
229
|
-
)
|
|
184
|
+
return self.row_to_record(row) if row else None
|
|
@@ -14,6 +14,15 @@ if TYPE_CHECKING:
|
|
|
14
14
|
|
|
15
15
|
CLAUDE_PROJECTS_DIR = Path.home() / ".claude" / "projects"
|
|
16
16
|
|
|
17
|
+
TRANSCRIPT_MEMO: dict[tuple[SessionId, Path], Path] = {}
|
|
18
|
+
"""Positive-hit memo for :func:`find_transcript_sync`, keyed by ``(session_id, resolved root)``.
|
|
19
|
+
|
|
20
|
+
Only successful lookups are cached: a miss may become a hit once Claude Code
|
|
21
|
+
writes the file, so ``None`` is never stored. A cached hit is revalidated with
|
|
22
|
+
``Path.exists`` before it is reused — a transcript file can be pruned from disk —
|
|
23
|
+
and a stale entry falls through to a fresh scan.
|
|
24
|
+
"""
|
|
25
|
+
|
|
17
26
|
|
|
18
27
|
class TranscriptExpiredError(RuntimeError):
|
|
19
28
|
"""A session's transcript file is gone from disk.
|
|
@@ -106,10 +115,18 @@ def find_transcript_sync(session_id: SessionId, *, root: Path | None = None) ->
|
|
|
106
115
|
:data:`CLAUDE_PROJECTS_DIR`), resolving symlinks — cc-pool gives one
|
|
107
116
|
transcript several path spellings — and deduping by resolved real path.
|
|
108
117
|
|
|
118
|
+
A successful lookup is memoized in ``TRANSCRIPT_MEMO`` under
|
|
119
|
+
``(session_id, resolved root)`` and revalidated on the next hit, so a
|
|
120
|
+
repeated probe skips the recursive glob while a pruned transcript still
|
|
121
|
+
forces a fresh scan.
|
|
122
|
+
|
|
109
123
|
Returns:
|
|
110
124
|
The newest-mtime real path, or None when no transcript exists.
|
|
111
125
|
"""
|
|
112
126
|
base = root or CLAUDE_PROJECTS_DIR
|
|
127
|
+
key = (session_id, base.resolve())
|
|
128
|
+
if (cached := TRANSCRIPT_MEMO.get(key)) is not None and cached.exists():
|
|
129
|
+
return cached
|
|
113
130
|
if not base.exists():
|
|
114
131
|
return None
|
|
115
132
|
candidates: dict[Path, float] = {}
|
|
@@ -120,7 +137,9 @@ def find_transcript_sync(session_id: SessionId, *, root: Path | None = None) ->
|
|
|
120
137
|
candidates[real] = real.stat().st_mtime
|
|
121
138
|
except OSError:
|
|
122
139
|
continue
|
|
123
|
-
|
|
140
|
+
if (hit := max(candidates, key=candidates.__getitem__, default=None)) is not None:
|
|
141
|
+
TRANSCRIPT_MEMO[key] = hit
|
|
142
|
+
return hit
|
|
124
143
|
|
|
125
144
|
|
|
126
145
|
async def find_transcript(session_id: SessionId, *, root: Path | None = None) -> Path | None:
|
|
@@ -43,7 +43,7 @@ TRAILING_PUNCT = ".!?…,;:"
|
|
|
43
43
|
STRUCTURAL_TAG_GROUP: tuple[str, str] = (
|
|
44
44
|
"xml_tags",
|
|
45
45
|
r"<(?:system[_-](?:instruction|reminder)"
|
|
46
|
-
r"|local-command-(?:stdout|caveat)"
|
|
46
|
+
r"|local-command-(?:stdout|stderr|caveat)"
|
|
47
47
|
r"|command-(?:name|message|args)"
|
|
48
48
|
r"|task-notification"
|
|
49
49
|
r"|persisted-output"
|
|
@@ -59,33 +59,31 @@ STRUCTURAL_GROUPS: tuple[tuple[str, str], ...] = (
|
|
|
59
59
|
("hash_banner", r"(?:###\s+[\w][\w \-]{0,30}\s+){3,}###"),
|
|
60
60
|
)
|
|
61
61
|
|
|
62
|
-
#
|
|
63
|
-
#
|
|
64
|
-
# ASCII-pinned via (?-i:[Ii]) so Rust regex and Python re stay in parity across combining marks
|
|
65
|
-
# and dotted/dotless-I folding. Mirror any change in rust/src/protocol.rs AGENT_INJECTION_PATTERN.
|
|
62
|
+
# Start-anchored so a mid-text mention is not a banner.
|
|
63
|
+
# Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
|
|
66
64
|
AGENT_INJECTION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
67
65
|
("xml_tags_extra", r"\A\s*<(?:teammate-message|scheduled-task)(?:[\s/>]|$)"),
|
|
68
66
|
("augment_agent", r"\A\s*# Augment Agent(?:\s|$)"),
|
|
69
67
|
("role_reminder", r"\A\s*\[Role Rem(?-i:[Ii])nder(?:[\s:\]]|$)"),
|
|
70
68
|
)
|
|
71
69
|
|
|
72
|
-
|
|
70
|
+
# The leading "i" of "interrupted" is ASCII-pinned (?-i:[Ii]) so Python re.IGNORECASE
|
|
71
|
+
# doesn't fold the dotted/dotless-I forms (U+0130/U+0131) that Rust regex never matches.
|
|
72
|
+
# Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
|
|
73
|
+
INTERRUPT_MARKER_GROUPS: tuple[tuple[str, str], ...] = (
|
|
74
|
+
("interrupt", r"^\s*\[Request (?-i:[Ii])nterrupted by user"),
|
|
75
|
+
)
|
|
73
76
|
STOP_HOOK_GROUPS: tuple[tuple[str, str], ...] = (("stop_hook", r"Stop hook feedback:"),)
|
|
74
77
|
|
|
75
|
-
# Raw CC-injected protocol strings carried in tool-result content
|
|
76
|
-
# and the markers that wrap the user's verbatim instruction in a rejected tool use,
|
|
77
|
-
# and the banner pair that wraps an answered AskUserQuestion round.
|
|
78
|
+
# Raw CC-injected protocol strings carried in tool-result content, not user-authored text.
|
|
78
79
|
DENIAL_PREFIX = "The user doesn't want to proceed with this tool use. The tool use was rejected"
|
|
79
80
|
USER_SAID_MARKER = "To tell you how to proceed, the user said:\n"
|
|
80
81
|
USER_SAID_TRAILER = "Note: The user's next message"
|
|
81
82
|
ANSWERED_PREFIX = "Your questions have been answered: "
|
|
82
83
|
ANSWERED_TRAILER = ". You can now continue with these answers in mind."
|
|
83
84
|
|
|
84
|
-
# Approve-and-advance directives
|
|
85
|
-
#
|
|
86
|
-
# correcting it — the opposite of pushback — so a pushback consumer drops them. The
|
|
87
|
-
# approve-and-advance arm is start-anchored so a mid-sentence "commit"/"push" inside
|
|
88
|
-
# a real correction never matches; only the resume arm searches anywhere.
|
|
85
|
+
# Approve-and-advance directives advance the prior assistant turn rather than correct it — the opposite of
|
|
86
|
+
# pushback — so pushback consumers drop them; start-anchoring keeps a mid-correction "commit"/"push" from matching.
|
|
89
87
|
CONTINUATION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
90
88
|
(
|
|
91
89
|
"continuation",
|
|
@@ -98,12 +96,12 @@ CONTINUATION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
|
98
96
|
)
|
|
99
97
|
|
|
100
98
|
# Bash-mode (`!`) command echoes recorded as user turns — the command line and its
|
|
101
|
-
# captured stdout/stderr, not authored feedback.
|
|
102
|
-
|
|
99
|
+
# captured stdout/stderr, not authored feedback. Start-anchored (\A\s*) so a mid-text
|
|
100
|
+
# mention of a bash tag is authored prose, not an echo — a real echo begins with the tag.
|
|
101
|
+
# Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
|
|
102
|
+
COMMAND_ECHO_GROUPS: tuple[tuple[str, str], ...] = (("command_echo", r"\A\s*<bash-(?:input|stdout|stderr)\b"),)
|
|
103
103
|
|
|
104
|
-
#
|
|
105
|
-
# stop-hook are kept separate because they carry pushback and must never be folded
|
|
106
|
-
# into the structural-noise default.
|
|
104
|
+
# Interrupt and stop-hook stay separate categories: they carry pushback and must never fold into structural noise.
|
|
107
105
|
JUNK_CATEGORIES: dict[str, tuple[tuple[str, str], ...]] = {
|
|
108
106
|
"structural": STRUCTURAL_GROUPS,
|
|
109
107
|
"agent_injection": AGENT_INJECTION_GROUPS,
|
|
@@ -113,13 +111,13 @@ JUNK_CATEGORIES: dict[str, tuple[tuple[str, str], ...]] = {
|
|
|
113
111
|
"command_echo": COMMAND_ECHO_GROUPS,
|
|
114
112
|
}
|
|
115
113
|
|
|
116
|
-
#
|
|
117
|
-
# interrupt/stop-hook — those carry pushback and must never live here.
|
|
114
|
+
# Structural ∪ agent-injection, WITHOUT interrupt/stop-hook — those carry pushback and must never live here.
|
|
118
115
|
STRUCTURAL_NOISE_GROUPS: tuple[tuple[str, str], ...] = STRUCTURAL_GROUPS + AGENT_INJECTION_GROUPS
|
|
119
116
|
|
|
120
|
-
# The
|
|
121
|
-
|
|
122
|
-
|
|
117
|
+
# The historical monolithic JUNK_USER_MESSAGE_RE plus bash-mode command echoes.
|
|
118
|
+
SENTIMENT_JUNK_GROUPS: tuple[tuple[str, str], ...] = (
|
|
119
|
+
STRUCTURAL_GROUPS + INTERRUPT_MARKER_GROUPS + STOP_HOOK_GROUPS + COMMAND_ECHO_GROUPS
|
|
120
|
+
)
|
|
123
121
|
|
|
124
122
|
FRUSTRATION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
125
123
|
(
|
|
@@ -351,9 +349,14 @@ class FilterSpec:
|
|
|
351
349
|
clauses: tuple[Clause, ...]
|
|
352
350
|
|
|
353
351
|
|
|
352
|
+
def group_pattern(groups: tuple[tuple[str, str], ...]) -> str:
|
|
353
|
+
"""Composes named regex groups into one non-capturing alternation."""
|
|
354
|
+
return "|".join(f"(?:{pattern})" for _, pattern in groups)
|
|
355
|
+
|
|
356
|
+
|
|
354
357
|
@cache
|
|
355
|
-
def compile_groups(groups: tuple[tuple[str, str], ...], ignore_case: bool) -> re.Pattern[str]:
|
|
356
|
-
return re.compile(
|
|
358
|
+
def compile_groups(groups: tuple[tuple[str, str], ...], ignore_case: bool, *, multiline: bool = False) -> re.Pattern[str]:
|
|
359
|
+
return re.compile(group_pattern(groups), (re.IGNORECASE if ignore_case else 0) | (re.MULTILINE if multiline else 0))
|
|
357
360
|
|
|
358
361
|
|
|
359
362
|
def event_kind(event: TranscriptEvent) -> EventKind:
|
|
@@ -282,6 +282,7 @@ class VerdictStoreMixin:
|
|
|
282
282
|
prompt_version: int,
|
|
283
283
|
limit: int | None = None,
|
|
284
284
|
refresh_summary: bool = False,
|
|
285
|
+
probe_hydration: bool = True,
|
|
285
286
|
) -> list[dict[str, object]]:
|
|
286
287
|
"""Returns events lacking a verdict for ``(role, prompt_version)``, unjudged first.
|
|
287
288
|
|
|
@@ -298,7 +299,15 @@ class VerdictStoreMixin:
|
|
|
298
299
|
full fidelity once their windows hydrate again. A summary row
|
|
299
300
|
whose context window no longer hydrates — its transcript expired
|
|
300
301
|
or a ref was compacted away — is dropped, so a dead transcript
|
|
301
|
-
stops burning the cap.
|
|
302
|
+
stops burning the cap (unless ``probe_hydration`` is False).
|
|
303
|
+
probe_hydration: When True (default), each summary-refresh row is
|
|
304
|
+
probed with a transcript-discovery hydration check and dropped
|
|
305
|
+
when its window no longer hydrates. When False, that per-row
|
|
306
|
+
probe — and the recursive transcript scan behind it — is skipped
|
|
307
|
+
entirely: every summary-refresh row is returned unprobed, so some
|
|
308
|
+
may name a transcript that is no longer discoverable. Only the
|
|
309
|
+
``refresh_summary`` path probes, so this flag is a no-op
|
|
310
|
+
otherwise. The default preserves the probing behavior exactly.
|
|
302
311
|
|
|
303
312
|
Returns:
|
|
304
313
|
One dict per event with the columns needed to build its prompt.
|
|
@@ -327,7 +336,8 @@ class VerdictStoreMixin:
|
|
|
327
336
|
) as cur:
|
|
328
337
|
async for raw in cur:
|
|
329
338
|
row = dict(raw)
|
|
330
|
-
|
|
339
|
+
fresh = row.pop("verdict_id") is None
|
|
340
|
+
if not probe_hydration or fresh or await hydratable(str(row["context_json"])):
|
|
331
341
|
kept.append(row)
|
|
332
342
|
if limit is not None and len(kept) >= limit:
|
|
333
343
|
break
|