cc-transcript 10.6.0__tar.gz → 10.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/PKG-INFO +1 -1
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/__init__.py +4 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/_parser_rs.pyi +3 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/corrections.py +47 -93
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/decisions.py +37 -82
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/filterspec.py +29 -26
- cc_transcript-10.7.0/cc_transcript/ledger.py +87 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/signals.py +13 -22
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/spec.py +2 -8
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/models.py +55 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/query.py +12 -3
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/buckets.py +9 -2
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/pyproject.toml +1 -1
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/Cargo.toml +2 -1
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/build.rs +8 -3
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/command.rs +5 -15
- cc_transcript-10.7.0/rust/src/generated/command.rs +27 -0
- cc_transcript-10.7.0/rust/src/generated/mining.rs +19 -0
- cc_transcript-10.7.0/rust/src/generated/mod.rs +5 -0
- cc_transcript-10.7.0/rust/src/generated/protocol.rs +10 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/lib.rs +2 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/mining.rs +7 -45
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/protocol.rs +9 -18
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/python.rs +37 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/Cargo.lock +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/Cargo.toml +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/LICENSE +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/README.md +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/__main__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/activity.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/activity_probe.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/backend.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/builders.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/cli.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/command.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/context.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/corrections_cli.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/cost.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/discovery.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/disktruth.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/evidence.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/extract/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/extract/correct.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/facts.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/ids.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/llm.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/similar.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/verdicts.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/candidates.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/confidence.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/engine.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/filterspec.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/formats.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/sourcekind.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/store.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/notifications.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/parser.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/py.typed +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/render.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/rust.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/__init__.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/engine.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/lexicon.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/scorespec.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/store.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/tools.py +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/data/afinn-en-165.tsv +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/data/domain_overrides.tsv +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/activity.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/event.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/filter.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/lexicon.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/model.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/parse.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/score.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/types.rs +0 -0
- {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/value.rs +0 -0
|
@@ -83,6 +83,7 @@ EXPORTS: dict[str, str] = {
|
|
|
83
83
|
"Plugin",
|
|
84
84
|
"PrintMessage",
|
|
85
85
|
"PrintResult",
|
|
86
|
+
"Question",
|
|
86
87
|
"ServerToolUse",
|
|
87
88
|
"SystemEvent",
|
|
88
89
|
"TextBlock",
|
|
@@ -522,6 +523,9 @@ if TYPE_CHECKING:
|
|
|
522
523
|
from cc_transcript.models import (
|
|
523
524
|
PrintResult as PrintResult,
|
|
524
525
|
)
|
|
526
|
+
from cc_transcript.models import (
|
|
527
|
+
Question as Question,
|
|
528
|
+
)
|
|
525
529
|
from cc_transcript.models import (
|
|
526
530
|
ServerToolUse as ServerToolUse,
|
|
527
531
|
)
|
|
@@ -35,6 +35,9 @@ def lexicon_has_hit(text: str, floor: int, want_negative: bool, /) -> bool:
|
|
|
35
35
|
def lexicon_overrides() -> list[tuple[str, int]]:
|
|
36
36
|
"""The embedded domain-override entries (for the single-source drift guard)."""
|
|
37
37
|
|
|
38
|
+
def embedded_literals() -> dict[str, str | float | list[str]]:
|
|
39
|
+
"""The generated protocol, mining, and command literals keyed ``module.NAME`` (for the single-source drift guard)."""
|
|
40
|
+
|
|
38
41
|
def score_short_circuit(spec_json: str, buckets: list[list[str]], /) -> list[int | None]:
|
|
39
42
|
"""Per bucket, the first short-circuit stage's score over its user texts, else None."""
|
|
40
43
|
|
|
@@ -19,8 +19,9 @@ from __future__ import annotations
|
|
|
19
19
|
import json
|
|
20
20
|
import sqlite3
|
|
21
21
|
from dataclasses import dataclass, field
|
|
22
|
-
from
|
|
23
|
-
|
|
22
|
+
from typing import TYPE_CHECKING, Literal
|
|
23
|
+
|
|
24
|
+
from cc_transcript.ledger import SyncLedger
|
|
24
25
|
|
|
25
26
|
if TYPE_CHECKING:
|
|
26
27
|
from collections.abc import Mapping
|
|
@@ -57,12 +58,6 @@ CREATE INDEX IF NOT EXISTS idx_corrections_incorrect_digest ON corrections (inco
|
|
|
57
58
|
|
|
58
59
|
Origin = Literal["session", "git", "review"]
|
|
59
60
|
|
|
60
|
-
CORRECTION_COLUMNS = (
|
|
61
|
-
"ts_ms, session_id, source, anchor_uuid, incorrect_digest, incorrect_file, incorrect_old, "
|
|
62
|
-
"incorrect_new, correction_origin, correction_file, correction_old, correction_new, "
|
|
63
|
-
"correction_commit, correction_text, overlap, detail_json"
|
|
64
|
-
)
|
|
65
|
-
|
|
66
61
|
|
|
67
62
|
@dataclass(frozen=True, slots=True)
|
|
68
63
|
class Correction:
|
|
@@ -111,7 +106,7 @@ class Correction:
|
|
|
111
106
|
detail: Mapping[str, Any] = field(default_factory=dict)
|
|
112
107
|
|
|
113
108
|
|
|
114
|
-
class CorrectionLog:
|
|
109
|
+
class CorrectionLog(SyncLedger[Correction]):
|
|
115
110
|
"""The ``corrections`` ledger at ``~/.cc-transcript/corrections.db``.
|
|
116
111
|
|
|
117
112
|
Opened in WAL mode with a busy timeout because writers across the family
|
|
@@ -124,66 +119,46 @@ class CorrectionLog:
|
|
|
124
119
|
>>> log.by_digest(session_id, incorrect_digest=digest)
|
|
125
120
|
"""
|
|
126
121
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
""
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
""
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
return cls(conn)
|
|
149
|
-
|
|
150
|
-
def append(self, correction: Correction) -> None:
|
|
151
|
-
"""Appends ``correction`` as a single ``INSERT OR IGNORE``.
|
|
152
|
-
|
|
153
|
-
Idempotent on the UNIQUE key ``(session_id, anchor_uuid,
|
|
154
|
-
incorrect_digest)`` — re-harvesting the same edit, across reruns or
|
|
155
|
-
across the pairs of one event, writes one row.
|
|
156
|
-
"""
|
|
157
|
-
self.conn.execute(
|
|
158
|
-
f"INSERT OR IGNORE INTO corrections ({CORRECTION_COLUMNS}) "
|
|
159
|
-
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
160
|
-
(
|
|
161
|
-
correction.ts_ms,
|
|
162
|
-
correction.session_id,
|
|
163
|
-
correction.source,
|
|
164
|
-
correction.anchor_uuid,
|
|
165
|
-
correction.incorrect_digest,
|
|
166
|
-
correction.incorrect_file,
|
|
167
|
-
correction.incorrect_old,
|
|
168
|
-
correction.incorrect_new,
|
|
169
|
-
correction.correction_origin,
|
|
170
|
-
correction.correction_file,
|
|
171
|
-
correction.correction_old,
|
|
172
|
-
correction.correction_new,
|
|
173
|
-
correction.correction_commit,
|
|
174
|
-
correction.correction_text,
|
|
175
|
-
correction.overlap,
|
|
176
|
-
json.dumps(dict(correction.detail)),
|
|
177
|
-
),
|
|
178
|
-
)
|
|
122
|
+
DDL = CORRECTIONS_DDL
|
|
123
|
+
FILENAME = "corrections.db"
|
|
124
|
+
TABLE = "corrections"
|
|
125
|
+
COLUMNS = (
|
|
126
|
+
"ts_ms",
|
|
127
|
+
"session_id",
|
|
128
|
+
"source",
|
|
129
|
+
"anchor_uuid",
|
|
130
|
+
"incorrect_digest",
|
|
131
|
+
"incorrect_file",
|
|
132
|
+
"incorrect_old",
|
|
133
|
+
"incorrect_new",
|
|
134
|
+
"correction_origin",
|
|
135
|
+
"correction_file",
|
|
136
|
+
"correction_old",
|
|
137
|
+
"correction_new",
|
|
138
|
+
"correction_commit",
|
|
139
|
+
"correction_text",
|
|
140
|
+
"overlap",
|
|
141
|
+
"detail_json",
|
|
142
|
+
)
|
|
179
143
|
|
|
180
|
-
def
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
144
|
+
def row_to_record(self, row: sqlite3.Row) -> Correction:
|
|
145
|
+
return Correction(
|
|
146
|
+
ts_ms=row["ts_ms"],
|
|
147
|
+
session_id=row["session_id"],
|
|
148
|
+
source=row["source"],
|
|
149
|
+
anchor_uuid=row["anchor_uuid"],
|
|
150
|
+
incorrect_digest=row["incorrect_digest"],
|
|
151
|
+
incorrect_file=row["incorrect_file"],
|
|
152
|
+
incorrect_old=row["incorrect_old"],
|
|
153
|
+
incorrect_new=row["incorrect_new"],
|
|
154
|
+
correction_origin=row["correction_origin"],
|
|
155
|
+
correction_file=row["correction_file"],
|
|
156
|
+
correction_old=row["correction_old"],
|
|
157
|
+
correction_new=row["correction_new"],
|
|
158
|
+
correction_commit=row["correction_commit"],
|
|
159
|
+
correction_text=row["correction_text"],
|
|
160
|
+
overlap=row["overlap"],
|
|
161
|
+
detail=json.loads(row["detail_json"]),
|
|
187
162
|
)
|
|
188
163
|
|
|
189
164
|
def for_repo(self, repo: str) -> tuple[Correction, ...]:
|
|
@@ -193,7 +168,7 @@ class CorrectionLog:
|
|
|
193
168
|
captain-hook reviewer) can pull every correction for its repo at once.
|
|
194
169
|
"""
|
|
195
170
|
return tuple(
|
|
196
|
-
|
|
171
|
+
self.row_to_record(row)
|
|
197
172
|
for row in self.conn.execute(
|
|
198
173
|
"SELECT * FROM corrections WHERE json_extract(detail_json, '$.repo') = ? ORDER BY ts_ms, id",
|
|
199
174
|
(repo,),
|
|
@@ -212,12 +187,12 @@ class CorrectionLog:
|
|
|
212
187
|
rows = self.conn.execute(
|
|
213
188
|
"SELECT * FROM corrections WHERE ts_ms > ? AND source = ? ORDER BY ts_ms, id", (ts_ms, source)
|
|
214
189
|
)
|
|
215
|
-
return tuple(
|
|
190
|
+
return tuple(self.row_to_record(row) for row in rows)
|
|
216
191
|
|
|
217
192
|
def for_anchor(self, session_id: SessionId, anchor_uuid: EventUuid) -> tuple[Correction, ...]:
|
|
218
193
|
"""The corrections harvested around one feedback ``anchor_uuid``."""
|
|
219
194
|
return tuple(
|
|
220
|
-
|
|
195
|
+
self.row_to_record(row)
|
|
221
196
|
for row in self.conn.execute(
|
|
222
197
|
"SELECT * FROM corrections WHERE session_id = ? AND anchor_uuid = ? ORDER BY ts_ms, id",
|
|
223
198
|
(session_id, anchor_uuid),
|
|
@@ -232,30 +207,9 @@ class CorrectionLog:
|
|
|
232
207
|
corrected.
|
|
233
208
|
"""
|
|
234
209
|
return tuple(
|
|
235
|
-
|
|
210
|
+
self.row_to_record(row)
|
|
236
211
|
for row in self.conn.execute(
|
|
237
212
|
"SELECT * FROM corrections WHERE session_id = ? AND incorrect_digest = ? ORDER BY ts_ms, id",
|
|
238
213
|
(session_id, incorrect_digest),
|
|
239
214
|
)
|
|
240
215
|
)
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
def correction_of(row: sqlite3.Row) -> Correction:
|
|
244
|
-
return Correction(
|
|
245
|
-
ts_ms=row["ts_ms"],
|
|
246
|
-
session_id=row["session_id"],
|
|
247
|
-
source=row["source"],
|
|
248
|
-
anchor_uuid=row["anchor_uuid"],
|
|
249
|
-
incorrect_digest=row["incorrect_digest"],
|
|
250
|
-
incorrect_file=row["incorrect_file"],
|
|
251
|
-
incorrect_old=row["incorrect_old"],
|
|
252
|
-
incorrect_new=row["incorrect_new"],
|
|
253
|
-
correction_origin=row["correction_origin"],
|
|
254
|
-
correction_file=row["correction_file"],
|
|
255
|
-
correction_old=row["correction_old"],
|
|
256
|
-
correction_new=row["correction_new"],
|
|
257
|
-
correction_commit=row["correction_commit"],
|
|
258
|
-
correction_text=row["correction_text"],
|
|
259
|
-
overlap=row["overlap"],
|
|
260
|
-
detail=json.loads(row["detail_json"]),
|
|
261
|
-
)
|
|
@@ -12,8 +12,9 @@ from __future__ import annotations
|
|
|
12
12
|
import json
|
|
13
13
|
import sqlite3
|
|
14
14
|
from dataclasses import dataclass, field
|
|
15
|
-
from
|
|
16
|
-
|
|
15
|
+
from typing import TYPE_CHECKING, Literal
|
|
16
|
+
|
|
17
|
+
from cc_transcript.ledger import SyncLedger
|
|
17
18
|
|
|
18
19
|
if TYPE_CHECKING:
|
|
19
20
|
from collections.abc import Mapping
|
|
@@ -48,11 +49,6 @@ CREATE INDEX IF NOT EXISTS idx_decisions_source_file ON decisions (source_file);
|
|
|
48
49
|
|
|
49
50
|
Action = Literal["allow", "block", "warn", "nudge", "note"]
|
|
50
51
|
|
|
51
|
-
DECISION_COLUMNS = (
|
|
52
|
-
"ts_ms, session_id, source, kind, source_file, event, action, "
|
|
53
|
-
"tool_name, tool_digest, event_uuid, message, detail_json"
|
|
54
|
-
)
|
|
55
|
-
|
|
56
52
|
|
|
57
53
|
@dataclass(frozen=True, slots=True)
|
|
58
54
|
class Decision:
|
|
@@ -91,7 +87,7 @@ class Decision:
|
|
|
91
87
|
detail: Mapping[str, Any] = field(default_factory=dict)
|
|
92
88
|
|
|
93
89
|
|
|
94
|
-
class DecisionLog:
|
|
90
|
+
class DecisionLog(SyncLedger[Decision]):
|
|
95
91
|
"""The ``decisions`` ledger at ``~/.cc-transcript/decisions.db``.
|
|
96
92
|
|
|
97
93
|
Opened in WAL mode with a busy timeout because cc-review's Go daemon
|
|
@@ -104,62 +100,38 @@ class DecisionLog:
|
|
|
104
100
|
>>> log.attribute_tool(session_id, tool_digest=digest, near_ts_ms=ts)
|
|
105
101
|
"""
|
|
106
102
|
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
""
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
""
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
conn.row_factory = sqlite3.Row
|
|
125
|
-
conn.execute("PRAGMA journal_mode = WAL")
|
|
126
|
-
conn.execute("PRAGMA busy_timeout = 2000")
|
|
127
|
-
conn.executescript(DECISIONS_DDL)
|
|
128
|
-
return cls(conn)
|
|
129
|
-
|
|
130
|
-
def append(self, decision: Decision) -> None:
|
|
131
|
-
"""Appends ``decision`` as a single ``INSERT OR IGNORE``.
|
|
132
|
-
|
|
133
|
-
Idempotent on the UNIQUE key ``(session_id, ts_ms, source, kind,
|
|
134
|
-
tool_digest)`` when ``tool_digest`` is present; SQLite treats NULL
|
|
135
|
-
digests as distinct, so digestless rows rely on the writer not
|
|
136
|
-
re-running the same integer-ms timestamp.
|
|
137
|
-
"""
|
|
138
|
-
self.conn.execute(
|
|
139
|
-
f"INSERT OR IGNORE INTO decisions ({DECISION_COLUMNS}) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
140
|
-
(
|
|
141
|
-
decision.ts_ms,
|
|
142
|
-
decision.session_id,
|
|
143
|
-
decision.source,
|
|
144
|
-
decision.kind,
|
|
145
|
-
decision.source_file,
|
|
146
|
-
decision.event,
|
|
147
|
-
decision.action,
|
|
148
|
-
decision.tool_name,
|
|
149
|
-
decision.tool_digest,
|
|
150
|
-
decision.event_uuid,
|
|
151
|
-
decision.message,
|
|
152
|
-
json.dumps(dict(decision.detail)),
|
|
153
|
-
),
|
|
154
|
-
)
|
|
103
|
+
DDL = DECISIONS_DDL
|
|
104
|
+
FILENAME = "decisions.db"
|
|
105
|
+
TABLE = "decisions"
|
|
106
|
+
COLUMNS = (
|
|
107
|
+
"ts_ms",
|
|
108
|
+
"session_id",
|
|
109
|
+
"source",
|
|
110
|
+
"kind",
|
|
111
|
+
"source_file",
|
|
112
|
+
"event",
|
|
113
|
+
"action",
|
|
114
|
+
"tool_name",
|
|
115
|
+
"tool_digest",
|
|
116
|
+
"event_uuid",
|
|
117
|
+
"message",
|
|
118
|
+
"detail_json",
|
|
119
|
+
)
|
|
155
120
|
|
|
156
|
-
def
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
121
|
+
def row_to_record(self, row: sqlite3.Row) -> Decision:
|
|
122
|
+
return Decision(
|
|
123
|
+
ts_ms=row["ts_ms"],
|
|
124
|
+
session_id=row["session_id"],
|
|
125
|
+
source=row["source"],
|
|
126
|
+
kind=row["kind"],
|
|
127
|
+
source_file=row["source_file"],
|
|
128
|
+
event=row["event"],
|
|
129
|
+
action=row["action"],
|
|
130
|
+
tool_name=row["tool_name"],
|
|
131
|
+
tool_digest=row["tool_digest"],
|
|
132
|
+
event_uuid=row["event_uuid"],
|
|
133
|
+
message=row["message"],
|
|
134
|
+
detail=json.loads(row["detail_json"]),
|
|
163
135
|
)
|
|
164
136
|
|
|
165
137
|
def attribute_tool(
|
|
@@ -181,7 +153,7 @@ class DecisionLog:
|
|
|
181
153
|
" ORDER BY ts_ms DESC, id DESC LIMIT 1",
|
|
182
154
|
(session_id, tool_digest, near_ts_ms - window_ms, near_ts_ms),
|
|
183
155
|
).fetchone()
|
|
184
|
-
return
|
|
156
|
+
return self.row_to_record(row) if row else None
|
|
185
157
|
|
|
186
158
|
def attribute_nearest(
|
|
187
159
|
self,
|
|
@@ -209,21 +181,4 @@ class DecisionLog:
|
|
|
209
181
|
" ORDER BY ABS(ts_ms - ?), id DESC LIMIT 1",
|
|
210
182
|
(session_id, event, *kind_params, near_ts_ms - window_ms, near_ts_ms + window_ms, near_ts_ms),
|
|
211
183
|
).fetchone()
|
|
212
|
-
return
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
def decision_of(row: sqlite3.Row) -> Decision:
|
|
216
|
-
return Decision(
|
|
217
|
-
ts_ms=row["ts_ms"],
|
|
218
|
-
session_id=row["session_id"],
|
|
219
|
-
source=row["source"],
|
|
220
|
-
kind=row["kind"],
|
|
221
|
-
source_file=row["source_file"],
|
|
222
|
-
event=row["event"],
|
|
223
|
-
action=row["action"],
|
|
224
|
-
tool_name=row["tool_name"],
|
|
225
|
-
tool_digest=row["tool_digest"],
|
|
226
|
-
event_uuid=row["event_uuid"],
|
|
227
|
-
message=row["message"],
|
|
228
|
-
detail=json.loads(row["detail_json"]),
|
|
229
|
-
)
|
|
184
|
+
return self.row_to_record(row) if row else None
|
|
@@ -43,7 +43,7 @@ TRAILING_PUNCT = ".!?…,;:"
|
|
|
43
43
|
STRUCTURAL_TAG_GROUP: tuple[str, str] = (
|
|
44
44
|
"xml_tags",
|
|
45
45
|
r"<(?:system[_-](?:instruction|reminder)"
|
|
46
|
-
r"|local-command-(?:stdout|caveat)"
|
|
46
|
+
r"|local-command-(?:stdout|stderr|caveat)"
|
|
47
47
|
r"|command-(?:name|message|args)"
|
|
48
48
|
r"|task-notification"
|
|
49
49
|
r"|persisted-output"
|
|
@@ -59,33 +59,31 @@ STRUCTURAL_GROUPS: tuple[tuple[str, str], ...] = (
|
|
|
59
59
|
("hash_banner", r"(?:###\s+[\w][\w \-]{0,30}\s+){3,}###"),
|
|
60
60
|
)
|
|
61
61
|
|
|
62
|
-
#
|
|
63
|
-
#
|
|
64
|
-
# ASCII-pinned via (?-i:[Ii]) so Rust regex and Python re stay in parity across combining marks
|
|
65
|
-
# and dotted/dotless-I folding. Mirror any change in rust/src/protocol.rs AGENT_INJECTION_PATTERN.
|
|
62
|
+
# Start-anchored so a mid-text mention is not a banner.
|
|
63
|
+
# Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
|
|
66
64
|
AGENT_INJECTION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
67
65
|
("xml_tags_extra", r"\A\s*<(?:teammate-message|scheduled-task)(?:[\s/>]|$)"),
|
|
68
66
|
("augment_agent", r"\A\s*# Augment Agent(?:\s|$)"),
|
|
69
67
|
("role_reminder", r"\A\s*\[Role Rem(?-i:[Ii])nder(?:[\s:\]]|$)"),
|
|
70
68
|
)
|
|
71
69
|
|
|
72
|
-
|
|
70
|
+
# The leading "i" of "interrupted" is ASCII-pinned (?-i:[Ii]) so Python re.IGNORECASE
|
|
71
|
+
# doesn't fold the dotted/dotless-I forms (U+0130/U+0131) that Rust regex never matches.
|
|
72
|
+
# Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
|
|
73
|
+
INTERRUPT_MARKER_GROUPS: tuple[tuple[str, str], ...] = (
|
|
74
|
+
("interrupt", r"^\s*\[Request (?-i:[Ii])nterrupted by user"),
|
|
75
|
+
)
|
|
73
76
|
STOP_HOOK_GROUPS: tuple[tuple[str, str], ...] = (("stop_hook", r"Stop hook feedback:"),)
|
|
74
77
|
|
|
75
|
-
# Raw CC-injected protocol strings carried in tool-result content
|
|
76
|
-
# and the markers that wrap the user's verbatim instruction in a rejected tool use,
|
|
77
|
-
# and the banner pair that wraps an answered AskUserQuestion round.
|
|
78
|
+
# Raw CC-injected protocol strings carried in tool-result content, not user-authored text.
|
|
78
79
|
DENIAL_PREFIX = "The user doesn't want to proceed with this tool use. The tool use was rejected"
|
|
79
80
|
USER_SAID_MARKER = "To tell you how to proceed, the user said:\n"
|
|
80
81
|
USER_SAID_TRAILER = "Note: The user's next message"
|
|
81
82
|
ANSWERED_PREFIX = "Your questions have been answered: "
|
|
82
83
|
ANSWERED_TRAILER = ". You can now continue with these answers in mind."
|
|
83
84
|
|
|
84
|
-
# Approve-and-advance directives
|
|
85
|
-
#
|
|
86
|
-
# correcting it — the opposite of pushback — so a pushback consumer drops them. The
|
|
87
|
-
# approve-and-advance arm is start-anchored so a mid-sentence "commit"/"push" inside
|
|
88
|
-
# a real correction never matches; only the resume arm searches anywhere.
|
|
85
|
+
# Approve-and-advance directives advance the prior assistant turn rather than correct it — the opposite of
|
|
86
|
+
# pushback — so pushback consumers drop them; start-anchoring keeps a mid-correction "commit"/"push" from matching.
|
|
89
87
|
CONTINUATION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
90
88
|
(
|
|
91
89
|
"continuation",
|
|
@@ -98,12 +96,12 @@ CONTINUATION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
|
98
96
|
)
|
|
99
97
|
|
|
100
98
|
# Bash-mode (`!`) command echoes recorded as user turns — the command line and its
|
|
101
|
-
# captured stdout/stderr, not authored feedback.
|
|
102
|
-
|
|
99
|
+
# captured stdout/stderr, not authored feedback. Start-anchored (\A\s*) so a mid-text
|
|
100
|
+
# mention of a bash tag is authored prose, not an echo — a real echo begins with the tag.
|
|
101
|
+
# Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
|
|
102
|
+
COMMAND_ECHO_GROUPS: tuple[tuple[str, str], ...] = (("command_echo", r"\A\s*<bash-(?:input|stdout|stderr)\b"),)
|
|
103
103
|
|
|
104
|
-
#
|
|
105
|
-
# stop-hook are kept separate because they carry pushback and must never be folded
|
|
106
|
-
# into the structural-noise default.
|
|
104
|
+
# Interrupt and stop-hook stay separate categories: they carry pushback and must never fold into structural noise.
|
|
107
105
|
JUNK_CATEGORIES: dict[str, tuple[tuple[str, str], ...]] = {
|
|
108
106
|
"structural": STRUCTURAL_GROUPS,
|
|
109
107
|
"agent_injection": AGENT_INJECTION_GROUPS,
|
|
@@ -113,13 +111,13 @@ JUNK_CATEGORIES: dict[str, tuple[tuple[str, str], ...]] = {
|
|
|
113
111
|
"command_echo": COMMAND_ECHO_GROUPS,
|
|
114
112
|
}
|
|
115
113
|
|
|
116
|
-
#
|
|
117
|
-
# interrupt/stop-hook — those carry pushback and must never live here.
|
|
114
|
+
# Structural ∪ agent-injection, WITHOUT interrupt/stop-hook — those carry pushback and must never live here.
|
|
118
115
|
STRUCTURAL_NOISE_GROUPS: tuple[tuple[str, str], ...] = STRUCTURAL_GROUPS + AGENT_INJECTION_GROUPS
|
|
119
116
|
|
|
120
|
-
# The
|
|
121
|
-
|
|
122
|
-
|
|
117
|
+
# The historical monolithic JUNK_USER_MESSAGE_RE plus bash-mode command echoes.
|
|
118
|
+
SENTIMENT_JUNK_GROUPS: tuple[tuple[str, str], ...] = (
|
|
119
|
+
STRUCTURAL_GROUPS + INTERRUPT_MARKER_GROUPS + STOP_HOOK_GROUPS + COMMAND_ECHO_GROUPS
|
|
120
|
+
)
|
|
123
121
|
|
|
124
122
|
FRUSTRATION_GROUPS: tuple[tuple[str, str], ...] = (
|
|
125
123
|
(
|
|
@@ -351,9 +349,14 @@ class FilterSpec:
|
|
|
351
349
|
clauses: tuple[Clause, ...]
|
|
352
350
|
|
|
353
351
|
|
|
352
|
+
def group_pattern(groups: tuple[tuple[str, str], ...]) -> str:
|
|
353
|
+
"""Composes named regex groups into one non-capturing alternation."""
|
|
354
|
+
return "|".join(f"(?:{pattern})" for _, pattern in groups)
|
|
355
|
+
|
|
356
|
+
|
|
354
357
|
@cache
|
|
355
|
-
def compile_groups(groups: tuple[tuple[str, str], ...], ignore_case: bool) -> re.Pattern[str]:
|
|
356
|
-
return re.compile(
|
|
358
|
+
def compile_groups(groups: tuple[tuple[str, str], ...], ignore_case: bool, *, multiline: bool = False) -> re.Pattern[str]:
|
|
359
|
+
return re.compile(group_pattern(groups), (re.IGNORECASE if ignore_case else 0) | (re.MULTILINE if multiline else 0))
|
|
357
360
|
|
|
358
361
|
|
|
359
362
|
def event_kind(event: TranscriptEvent) -> EventKind:
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""The append-only SQLite base shared by the correction and decision ledgers.
|
|
2
|
+
|
|
3
|
+
Both family ledgers are one durable, WAL-mode, ``INSERT OR IGNORE`` table behind
|
|
4
|
+
a fixed schema, so :class:`SyncLedger` owns the connection, the ``open`` plumbing,
|
|
5
|
+
and the schema-driven append and read — each ledger supplies only its DDL,
|
|
6
|
+
filename, table, columns, and row mapper.
|
|
7
|
+
|
|
8
|
+
Import-light by contract, like :mod:`cc_transcript.ids`: the standard library
|
|
9
|
+
plus identity primitives only, so a hook reading a ledger pays nothing for the
|
|
10
|
+
parser.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import sqlite3
|
|
17
|
+
from abc import ABC, abstractmethod
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import TYPE_CHECKING, ClassVar, Protocol, Self
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING:
|
|
22
|
+
from collections.abc import Mapping
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from cc_transcript.ids import SessionId
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class LedgerRecord(Protocol):
|
|
29
|
+
detail: Mapping[str, Any]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class SyncLedger[R: LedgerRecord](ABC):
|
|
33
|
+
DDL: ClassVar[str]
|
|
34
|
+
FILENAME: ClassVar[str]
|
|
35
|
+
TABLE: ClassVar[str]
|
|
36
|
+
COLUMNS: ClassVar[tuple[str, ...]]
|
|
37
|
+
|
|
38
|
+
def __init__(self, conn: sqlite3.Connection) -> None:
|
|
39
|
+
self.conn = conn
|
|
40
|
+
|
|
41
|
+
@abstractmethod
|
|
42
|
+
def row_to_record(self, row: sqlite3.Row) -> R: ...
|
|
43
|
+
|
|
44
|
+
@classmethod
|
|
45
|
+
def open(cls, path: Path | None = None) -> Self:
|
|
46
|
+
"""Opens (creating if needed) the ledger at ``path``.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
path: The database file path; its parents are created if absent.
|
|
50
|
+
Defaults to the ledger's file under ``~/.cc-transcript``.
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
The opened log.
|
|
54
|
+
"""
|
|
55
|
+
path = path or Path.home() / ".cc-transcript" / cls.FILENAME
|
|
56
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
57
|
+
conn = sqlite3.connect(path, autocommit=True)
|
|
58
|
+
conn.row_factory = sqlite3.Row
|
|
59
|
+
conn.execute("PRAGMA journal_mode = WAL")
|
|
60
|
+
conn.execute("PRAGMA busy_timeout = 2000")
|
|
61
|
+
conn.executescript(cls.DDL)
|
|
62
|
+
return cls(conn)
|
|
63
|
+
|
|
64
|
+
def append(self, record: R) -> None:
|
|
65
|
+
"""Appends ``record`` as a single ``INSERT OR IGNORE``.
|
|
66
|
+
|
|
67
|
+
Idempotent on the table's UNIQUE key, so re-running a writer writes one
|
|
68
|
+
row; SQLite treats NULL key columns as distinct, so rows whose key
|
|
69
|
+
carries a NULL rely on the writer not repeating the same values.
|
|
70
|
+
"""
|
|
71
|
+
self.conn.execute(
|
|
72
|
+
f"INSERT OR IGNORE INTO {self.TABLE} ({', '.join(self.COLUMNS)}) "
|
|
73
|
+
f"VALUES ({', '.join(['?'] * len(self.COLUMNS))})",
|
|
74
|
+
tuple(
|
|
75
|
+
json.dumps(dict(record.detail)) if column == "detail_json" else getattr(record, column)
|
|
76
|
+
for column in self.COLUMNS
|
|
77
|
+
),
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
def for_session(self, session_id: SessionId) -> tuple[R, ...]:
|
|
81
|
+
"""All records for ``session_id``, ordered by timestamp."""
|
|
82
|
+
return tuple(
|
|
83
|
+
self.row_to_record(row)
|
|
84
|
+
for row in self.conn.execute(
|
|
85
|
+
f"SELECT * FROM {self.TABLE} WHERE session_id = ? ORDER BY ts_ms, id", (session_id,)
|
|
86
|
+
)
|
|
87
|
+
)
|