cc-transcript 10.6.0__tar.gz → 10.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/PKG-INFO +1 -1
  2. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/__init__.py +4 -0
  3. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/_parser_rs.pyi +3 -0
  4. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/corrections.py +47 -93
  5. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/decisions.py +37 -82
  6. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/filterspec.py +29 -26
  7. cc_transcript-10.7.0/cc_transcript/ledger.py +87 -0
  8. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/signals.py +13 -22
  9. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/spec.py +2 -8
  10. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/models.py +55 -0
  11. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/query.py +12 -3
  12. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/buckets.py +9 -2
  13. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/pyproject.toml +1 -1
  14. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/Cargo.toml +2 -1
  15. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/build.rs +8 -3
  16. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/command.rs +5 -15
  17. cc_transcript-10.7.0/rust/src/generated/command.rs +27 -0
  18. cc_transcript-10.7.0/rust/src/generated/mining.rs +19 -0
  19. cc_transcript-10.7.0/rust/src/generated/mod.rs +5 -0
  20. cc_transcript-10.7.0/rust/src/generated/protocol.rs +10 -0
  21. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/lib.rs +2 -0
  22. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/mining.rs +7 -45
  23. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/protocol.rs +9 -18
  24. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/python.rs +37 -0
  25. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/Cargo.lock +0 -0
  26. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/Cargo.toml +0 -0
  27. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/LICENSE +0 -0
  28. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/README.md +0 -0
  29. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/__main__.py +0 -0
  30. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/activity.py +0 -0
  31. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/activity_probe.py +0 -0
  32. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/backend.py +0 -0
  33. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/builders.py +0 -0
  34. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/cli.py +0 -0
  35. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/command.py +0 -0
  36. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/context.py +0 -0
  37. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/corrections_cli.py +0 -0
  38. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/cost.py +0 -0
  39. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/discovery.py +0 -0
  40. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/disktruth.py +0 -0
  41. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/evidence.py +0 -0
  42. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/extract/__init__.py +0 -0
  43. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/extract/correct.py +0 -0
  44. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/facts.py +0 -0
  45. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/ids.py +0 -0
  46. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/__init__.py +0 -0
  47. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/llm.py +0 -0
  48. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/similar.py +0 -0
  49. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/judge/verdicts.py +0 -0
  50. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/__init__.py +0 -0
  51. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/candidates.py +0 -0
  52. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/confidence.py +0 -0
  53. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/engine.py +0 -0
  54. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/filterspec.py +0 -0
  55. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/formats.py +0 -0
  56. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/sourcekind.py +0 -0
  57. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/mining/store.py +0 -0
  58. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/notifications.py +0 -0
  59. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/parser.py +0 -0
  60. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/py.typed +0 -0
  61. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/render.py +0 -0
  62. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/rust.py +0 -0
  63. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/__init__.py +0 -0
  64. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/engine.py +0 -0
  65. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/lexicon.py +0 -0
  66. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/sentiment/scorespec.py +0 -0
  67. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/store.py +0 -0
  68. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/cc_transcript/tools.py +0 -0
  69. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/data/afinn-en-165.tsv +0 -0
  70. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/data/domain_overrides.tsv +0 -0
  71. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/activity.rs +0 -0
  72. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/event.rs +0 -0
  73. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/filter.rs +0 -0
  74. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/lexicon.rs +0 -0
  75. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/model.rs +0 -0
  76. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/parse.rs +0 -0
  77. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/score.rs +0 -0
  78. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/types.rs +0 -0
  79. {cc_transcript-10.6.0 → cc_transcript-10.7.0}/rust/src/value.rs +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cc-transcript
3
- Version: 10.6.0
3
+ Version: 10.7.0
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: Environment :: Console
6
6
  Classifier: Intended Audience :: Developers
@@ -83,6 +83,7 @@ EXPORTS: dict[str, str] = {
83
83
  "Plugin",
84
84
  "PrintMessage",
85
85
  "PrintResult",
86
+ "Question",
86
87
  "ServerToolUse",
87
88
  "SystemEvent",
88
89
  "TextBlock",
@@ -522,6 +523,9 @@ if TYPE_CHECKING:
522
523
  from cc_transcript.models import (
523
524
  PrintResult as PrintResult,
524
525
  )
526
+ from cc_transcript.models import (
527
+ Question as Question,
528
+ )
525
529
  from cc_transcript.models import (
526
530
  ServerToolUse as ServerToolUse,
527
531
  )
@@ -35,6 +35,9 @@ def lexicon_has_hit(text: str, floor: int, want_negative: bool, /) -> bool:
35
35
  def lexicon_overrides() -> list[tuple[str, int]]:
36
36
  """The embedded domain-override entries (for the single-source drift guard)."""
37
37
 
38
+ def embedded_literals() -> dict[str, str | float | list[str]]:
39
+ """The generated protocol, mining, and command literals keyed ``module.NAME`` (for the single-source drift guard)."""
40
+
38
41
  def score_short_circuit(spec_json: str, buckets: list[list[str]], /) -> list[int | None]:
39
42
  """Per bucket, the first short-circuit stage's score over its user texts, else None."""
40
43
 
@@ -19,8 +19,9 @@ from __future__ import annotations
19
19
  import json
20
20
  import sqlite3
21
21
  from dataclasses import dataclass, field
22
- from pathlib import Path
23
- from typing import TYPE_CHECKING, Literal, Self
22
+ from typing import TYPE_CHECKING, Literal
23
+
24
+ from cc_transcript.ledger import SyncLedger
24
25
 
25
26
  if TYPE_CHECKING:
26
27
  from collections.abc import Mapping
@@ -57,12 +58,6 @@ CREATE INDEX IF NOT EXISTS idx_corrections_incorrect_digest ON corrections (inco
57
58
 
58
59
  Origin = Literal["session", "git", "review"]
59
60
 
60
- CORRECTION_COLUMNS = (
61
- "ts_ms, session_id, source, anchor_uuid, incorrect_digest, incorrect_file, incorrect_old, "
62
- "incorrect_new, correction_origin, correction_file, correction_old, correction_new, "
63
- "correction_commit, correction_text, overlap, detail_json"
64
- )
65
-
66
61
 
67
62
  @dataclass(frozen=True, slots=True)
68
63
  class Correction:
@@ -111,7 +106,7 @@ class Correction:
111
106
  detail: Mapping[str, Any] = field(default_factory=dict)
112
107
 
113
108
 
114
- class CorrectionLog:
109
+ class CorrectionLog(SyncLedger[Correction]):
115
110
  """The ``corrections`` ledger at ``~/.cc-transcript/corrections.db``.
116
111
 
117
112
  Opened in WAL mode with a busy timeout because writers across the family
@@ -124,66 +119,46 @@ class CorrectionLog:
124
119
  >>> log.by_digest(session_id, incorrect_digest=digest)
125
120
  """
126
121
 
127
- def __init__(self, conn: sqlite3.Connection) -> None:
128
- self.conn = conn
129
-
130
- @classmethod
131
- def open(cls, path: Path | None = None) -> Self:
132
- """Opens (creating if needed) the ledger at ``path``.
133
-
134
- Args:
135
- path: The database file path; its parents are created if absent.
136
- Defaults to ``~/.cc-transcript/corrections.db``.
137
-
138
- Returns:
139
- The opened log.
140
- """
141
- path = path or Path.home() / ".cc-transcript" / "corrections.db"
142
- path.parent.mkdir(parents=True, exist_ok=True)
143
- conn = sqlite3.connect(path, autocommit=True)
144
- conn.row_factory = sqlite3.Row
145
- conn.execute("PRAGMA journal_mode = WAL")
146
- conn.execute("PRAGMA busy_timeout = 2000")
147
- conn.executescript(CORRECTIONS_DDL)
148
- return cls(conn)
149
-
150
- def append(self, correction: Correction) -> None:
151
- """Appends ``correction`` as a single ``INSERT OR IGNORE``.
152
-
153
- Idempotent on the UNIQUE key ``(session_id, anchor_uuid,
154
- incorrect_digest)`` — re-harvesting the same edit, across reruns or
155
- across the pairs of one event, writes one row.
156
- """
157
- self.conn.execute(
158
- f"INSERT OR IGNORE INTO corrections ({CORRECTION_COLUMNS}) "
159
- "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
160
- (
161
- correction.ts_ms,
162
- correction.session_id,
163
- correction.source,
164
- correction.anchor_uuid,
165
- correction.incorrect_digest,
166
- correction.incorrect_file,
167
- correction.incorrect_old,
168
- correction.incorrect_new,
169
- correction.correction_origin,
170
- correction.correction_file,
171
- correction.correction_old,
172
- correction.correction_new,
173
- correction.correction_commit,
174
- correction.correction_text,
175
- correction.overlap,
176
- json.dumps(dict(correction.detail)),
177
- ),
178
- )
122
+ DDL = CORRECTIONS_DDL
123
+ FILENAME = "corrections.db"
124
+ TABLE = "corrections"
125
+ COLUMNS = (
126
+ "ts_ms",
127
+ "session_id",
128
+ "source",
129
+ "anchor_uuid",
130
+ "incorrect_digest",
131
+ "incorrect_file",
132
+ "incorrect_old",
133
+ "incorrect_new",
134
+ "correction_origin",
135
+ "correction_file",
136
+ "correction_old",
137
+ "correction_new",
138
+ "correction_commit",
139
+ "correction_text",
140
+ "overlap",
141
+ "detail_json",
142
+ )
179
143
 
180
- def for_session(self, session_id: SessionId) -> tuple[Correction, ...]:
181
- """All corrections for ``session_id``, ordered by timestamp."""
182
- return tuple(
183
- correction_of(row)
184
- for row in self.conn.execute(
185
- "SELECT * FROM corrections WHERE session_id = ? ORDER BY ts_ms, id", (session_id,)
186
- )
144
+ def row_to_record(self, row: sqlite3.Row) -> Correction:
145
+ return Correction(
146
+ ts_ms=row["ts_ms"],
147
+ session_id=row["session_id"],
148
+ source=row["source"],
149
+ anchor_uuid=row["anchor_uuid"],
150
+ incorrect_digest=row["incorrect_digest"],
151
+ incorrect_file=row["incorrect_file"],
152
+ incorrect_old=row["incorrect_old"],
153
+ incorrect_new=row["incorrect_new"],
154
+ correction_origin=row["correction_origin"],
155
+ correction_file=row["correction_file"],
156
+ correction_old=row["correction_old"],
157
+ correction_new=row["correction_new"],
158
+ correction_commit=row["correction_commit"],
159
+ correction_text=row["correction_text"],
160
+ overlap=row["overlap"],
161
+ detail=json.loads(row["detail_json"]),
187
162
  )
188
163
 
189
164
  def for_repo(self, repo: str) -> tuple[Correction, ...]:
@@ -193,7 +168,7 @@ class CorrectionLog:
193
168
  captain-hook reviewer) can pull every correction for its repo at once.
194
169
  """
195
170
  return tuple(
196
- correction_of(row)
171
+ self.row_to_record(row)
197
172
  for row in self.conn.execute(
198
173
  "SELECT * FROM corrections WHERE json_extract(detail_json, '$.repo') = ? ORDER BY ts_ms, id",
199
174
  (repo,),
@@ -212,12 +187,12 @@ class CorrectionLog:
212
187
  rows = self.conn.execute(
213
188
  "SELECT * FROM corrections WHERE ts_ms > ? AND source = ? ORDER BY ts_ms, id", (ts_ms, source)
214
189
  )
215
- return tuple(correction_of(row) for row in rows)
190
+ return tuple(self.row_to_record(row) for row in rows)
216
191
 
217
192
  def for_anchor(self, session_id: SessionId, anchor_uuid: EventUuid) -> tuple[Correction, ...]:
218
193
  """The corrections harvested around one feedback ``anchor_uuid``."""
219
194
  return tuple(
220
- correction_of(row)
195
+ self.row_to_record(row)
221
196
  for row in self.conn.execute(
222
197
  "SELECT * FROM corrections WHERE session_id = ? AND anchor_uuid = ? ORDER BY ts_ms, id",
223
198
  (session_id, anchor_uuid),
@@ -232,30 +207,9 @@ class CorrectionLog:
232
207
  corrected.
233
208
  """
234
209
  return tuple(
235
- correction_of(row)
210
+ self.row_to_record(row)
236
211
  for row in self.conn.execute(
237
212
  "SELECT * FROM corrections WHERE session_id = ? AND incorrect_digest = ? ORDER BY ts_ms, id",
238
213
  (session_id, incorrect_digest),
239
214
  )
240
215
  )
241
-
242
-
243
- def correction_of(row: sqlite3.Row) -> Correction:
244
- return Correction(
245
- ts_ms=row["ts_ms"],
246
- session_id=row["session_id"],
247
- source=row["source"],
248
- anchor_uuid=row["anchor_uuid"],
249
- incorrect_digest=row["incorrect_digest"],
250
- incorrect_file=row["incorrect_file"],
251
- incorrect_old=row["incorrect_old"],
252
- incorrect_new=row["incorrect_new"],
253
- correction_origin=row["correction_origin"],
254
- correction_file=row["correction_file"],
255
- correction_old=row["correction_old"],
256
- correction_new=row["correction_new"],
257
- correction_commit=row["correction_commit"],
258
- correction_text=row["correction_text"],
259
- overlap=row["overlap"],
260
- detail=json.loads(row["detail_json"]),
261
- )
@@ -12,8 +12,9 @@ from __future__ import annotations
12
12
  import json
13
13
  import sqlite3
14
14
  from dataclasses import dataclass, field
15
- from pathlib import Path
16
- from typing import TYPE_CHECKING, Literal, Self
15
+ from typing import TYPE_CHECKING, Literal
16
+
17
+ from cc_transcript.ledger import SyncLedger
17
18
 
18
19
  if TYPE_CHECKING:
19
20
  from collections.abc import Mapping
@@ -48,11 +49,6 @@ CREATE INDEX IF NOT EXISTS idx_decisions_source_file ON decisions (source_file);
48
49
 
49
50
  Action = Literal["allow", "block", "warn", "nudge", "note"]
50
51
 
51
- DECISION_COLUMNS = (
52
- "ts_ms, session_id, source, kind, source_file, event, action, "
53
- "tool_name, tool_digest, event_uuid, message, detail_json"
54
- )
55
-
56
52
 
57
53
  @dataclass(frozen=True, slots=True)
58
54
  class Decision:
@@ -91,7 +87,7 @@ class Decision:
91
87
  detail: Mapping[str, Any] = field(default_factory=dict)
92
88
 
93
89
 
94
- class DecisionLog:
90
+ class DecisionLog(SyncLedger[Decision]):
95
91
  """The ``decisions`` ledger at ``~/.cc-transcript/decisions.db``.
96
92
 
97
93
  Opened in WAL mode with a busy timeout because cc-review's Go daemon
@@ -104,62 +100,38 @@ class DecisionLog:
104
100
  >>> log.attribute_tool(session_id, tool_digest=digest, near_ts_ms=ts)
105
101
  """
106
102
 
107
- def __init__(self, conn: sqlite3.Connection) -> None:
108
- self.conn = conn
109
-
110
- @classmethod
111
- def open(cls, path: Path | None = None) -> Self:
112
- """Opens (creating if needed) the ledger at ``path``.
113
-
114
- Args:
115
- path: The database file path; its parents are created if absent.
116
- Defaults to ``~/.cc-transcript/decisions.db``.
117
-
118
- Returns:
119
- The opened log.
120
- """
121
- path = path or Path.home() / ".cc-transcript" / "decisions.db"
122
- path.parent.mkdir(parents=True, exist_ok=True)
123
- conn = sqlite3.connect(path, autocommit=True)
124
- conn.row_factory = sqlite3.Row
125
- conn.execute("PRAGMA journal_mode = WAL")
126
- conn.execute("PRAGMA busy_timeout = 2000")
127
- conn.executescript(DECISIONS_DDL)
128
- return cls(conn)
129
-
130
- def append(self, decision: Decision) -> None:
131
- """Appends ``decision`` as a single ``INSERT OR IGNORE``.
132
-
133
- Idempotent on the UNIQUE key ``(session_id, ts_ms, source, kind,
134
- tool_digest)`` when ``tool_digest`` is present; SQLite treats NULL
135
- digests as distinct, so digestless rows rely on the writer not
136
- re-running the same integer-ms timestamp.
137
- """
138
- self.conn.execute(
139
- f"INSERT OR IGNORE INTO decisions ({DECISION_COLUMNS}) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
140
- (
141
- decision.ts_ms,
142
- decision.session_id,
143
- decision.source,
144
- decision.kind,
145
- decision.source_file,
146
- decision.event,
147
- decision.action,
148
- decision.tool_name,
149
- decision.tool_digest,
150
- decision.event_uuid,
151
- decision.message,
152
- json.dumps(dict(decision.detail)),
153
- ),
154
- )
103
+ DDL = DECISIONS_DDL
104
+ FILENAME = "decisions.db"
105
+ TABLE = "decisions"
106
+ COLUMNS = (
107
+ "ts_ms",
108
+ "session_id",
109
+ "source",
110
+ "kind",
111
+ "source_file",
112
+ "event",
113
+ "action",
114
+ "tool_name",
115
+ "tool_digest",
116
+ "event_uuid",
117
+ "message",
118
+ "detail_json",
119
+ )
155
120
 
156
- def for_session(self, session_id: SessionId) -> tuple[Decision, ...]:
157
- """All decisions for ``session_id``, ordered by timestamp."""
158
- return tuple(
159
- decision_of(row)
160
- for row in self.conn.execute(
161
- "SELECT * FROM decisions WHERE session_id = ? ORDER BY ts_ms, id", (session_id,)
162
- )
121
+ def row_to_record(self, row: sqlite3.Row) -> Decision:
122
+ return Decision(
123
+ ts_ms=row["ts_ms"],
124
+ session_id=row["session_id"],
125
+ source=row["source"],
126
+ kind=row["kind"],
127
+ source_file=row["source_file"],
128
+ event=row["event"],
129
+ action=row["action"],
130
+ tool_name=row["tool_name"],
131
+ tool_digest=row["tool_digest"],
132
+ event_uuid=row["event_uuid"],
133
+ message=row["message"],
134
+ detail=json.loads(row["detail_json"]),
163
135
  )
164
136
 
165
137
  def attribute_tool(
@@ -181,7 +153,7 @@ class DecisionLog:
181
153
  " ORDER BY ts_ms DESC, id DESC LIMIT 1",
182
154
  (session_id, tool_digest, near_ts_ms - window_ms, near_ts_ms),
183
155
  ).fetchone()
184
- return decision_of(row) if row else None
156
+ return self.row_to_record(row) if row else None
185
157
 
186
158
  def attribute_nearest(
187
159
  self,
@@ -209,21 +181,4 @@ class DecisionLog:
209
181
  " ORDER BY ABS(ts_ms - ?), id DESC LIMIT 1",
210
182
  (session_id, event, *kind_params, near_ts_ms - window_ms, near_ts_ms + window_ms, near_ts_ms),
211
183
  ).fetchone()
212
- return decision_of(row) if row else None
213
-
214
-
215
- def decision_of(row: sqlite3.Row) -> Decision:
216
- return Decision(
217
- ts_ms=row["ts_ms"],
218
- session_id=row["session_id"],
219
- source=row["source"],
220
- kind=row["kind"],
221
- source_file=row["source_file"],
222
- event=row["event"],
223
- action=row["action"],
224
- tool_name=row["tool_name"],
225
- tool_digest=row["tool_digest"],
226
- event_uuid=row["event_uuid"],
227
- message=row["message"],
228
- detail=json.loads(row["detail_json"]),
229
- )
184
+ return self.row_to_record(row) if row else None
@@ -43,7 +43,7 @@ TRAILING_PUNCT = ".!?…,;:"
43
43
  STRUCTURAL_TAG_GROUP: tuple[str, str] = (
44
44
  "xml_tags",
45
45
  r"<(?:system[_-](?:instruction|reminder)"
46
- r"|local-command-(?:stdout|caveat)"
46
+ r"|local-command-(?:stdout|stderr|caveat)"
47
47
  r"|command-(?:name|message|args)"
48
48
  r"|task-notification"
49
49
  r"|persisted-output"
@@ -59,33 +59,31 @@ STRUCTURAL_GROUPS: tuple[tuple[str, str], ...] = (
59
59
  ("hash_banner", r"(?:###\s+[\w][\w \-]{0,30}\s+){3,}###"),
60
60
  )
61
61
 
62
- # Agent-injected banners (teammate messages, scheduled tasks, foreign agents). Start-anchored
63
- # so a mid-text mention is not a banner; the follower class replaces \b and Reminder's "i" is
64
- # ASCII-pinned via (?-i:[Ii]) so Rust regex and Python re stay in parity across combining marks
65
- # and dotted/dotless-I folding. Mirror any change in rust/src/protocol.rs AGENT_INJECTION_PATTERN.
62
+ # Start-anchored so a mid-text mention is not a banner.
63
+ # Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
66
64
  AGENT_INJECTION_GROUPS: tuple[tuple[str, str], ...] = (
67
65
  ("xml_tags_extra", r"\A\s*<(?:teammate-message|scheduled-task)(?:[\s/>]|$)"),
68
66
  ("augment_agent", r"\A\s*# Augment Agent(?:\s|$)"),
69
67
  ("role_reminder", r"\A\s*\[Role Rem(?-i:[Ii])nder(?:[\s:\]]|$)"),
70
68
  )
71
69
 
72
- INTERRUPT_MARKER_GROUPS: tuple[tuple[str, str], ...] = (("interrupt", r"^\s*\[Request interrupted by user"),)
70
+ # The leading "i" of "interrupted" is ASCII-pinned (?-i:[Ii]) so Python re.IGNORECASE
71
+ # doesn't fold the dotted/dotless-I forms (U+0130/U+0131) that Rust regex never matches.
72
+ # Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
73
+ INTERRUPT_MARKER_GROUPS: tuple[tuple[str, str], ...] = (
74
+ ("interrupt", r"^\s*\[Request (?-i:[Ii])nterrupted by user"),
75
+ )
73
76
  STOP_HOOK_GROUPS: tuple[tuple[str, str], ...] = (("stop_hook", r"Stop hook feedback:"),)
74
77
 
75
- # Raw CC-injected protocol strings carried in tool-result content: the denial banner
76
- # and the markers that wrap the user's verbatim instruction in a rejected tool use,
77
- # and the banner pair that wraps an answered AskUserQuestion round.
78
+ # Raw CC-injected protocol strings carried in tool-result content, not user-authored text.
78
79
  DENIAL_PREFIX = "The user doesn't want to proceed with this tool use. The tool use was rejected"
79
80
  USER_SAID_MARKER = "To tell you how to proceed, the user said:\n"
80
81
  USER_SAID_TRAILER = "Note: The user's next message"
81
82
  ANSWERED_PREFIX = "Your questions have been answered: "
82
83
  ANSWERED_TRAILER = ". You can now continue with these answers in mind."
83
84
 
84
- # Approve-and-advance directives: a user telling the agent to proceed/commit/push or
85
- # to resume killed work. They follow an assistant turn but advance it rather than
86
- # correcting it — the opposite of pushback — so a pushback consumer drops them. The
87
- # approve-and-advance arm is start-anchored so a mid-sentence "commit"/"push" inside
88
- # a real correction never matches; only the resume arm searches anywhere.
85
+ # Approve-and-advance directives advance the prior assistant turn rather than correct it — the opposite of
86
+ # pushback — so pushback consumers drop them; start-anchoring keeps a mid-correction "commit"/"push" from matching.
89
87
  CONTINUATION_GROUPS: tuple[tuple[str, str], ...] = (
90
88
  (
91
89
  "continuation",
@@ -98,12 +96,12 @@ CONTINUATION_GROUPS: tuple[tuple[str, str], ...] = (
98
96
  )
99
97
 
100
98
  # Bash-mode (`!`) command echoes recorded as user turns — the command line and its
101
- # captured stdout/stderr, not authored feedback.
102
- COMMAND_ECHO_GROUPS: tuple[tuple[str, str], ...] = (("command_echo", r"<bash-(?:input|stdout|stderr)\b"),)
99
+ # captured stdout/stderr, not authored feedback. Start-anchored (\A\s*) so a mid-text
100
+ # mention of a bash tag is authored prose, not an echo — a real echo begins with the tag.
101
+ # Dialect-hardened for Rust-regex parity — see cc-notes doc "rust-regex-dialect-parity".
102
+ COMMAND_ECHO_GROUPS: tuple[tuple[str, str], ...] = (("command_echo", r"\A\s*<bash-(?:input|stdout|stderr)\b"),)
103
103
 
104
- # Named junk categories a consumer composes via ``drop_junk(...)``. Interrupt and
105
- # stop-hook are kept separate because they carry pushback and must never be folded
106
- # into the structural-noise default.
104
+ # Interrupt and stop-hook stay separate categories: they carry pushback and must never fold into structural noise.
107
105
  JUNK_CATEGORIES: dict[str, tuple[tuple[str, str], ...]] = {
108
106
  "structural": STRUCTURAL_GROUPS,
109
107
  "agent_injection": AGENT_INJECTION_GROUPS,
@@ -113,13 +111,13 @@ JUNK_CATEGORIES: dict[str, tuple[tuple[str, str], ...]] = {
113
111
  "command_echo": COMMAND_ECHO_GROUPS,
114
112
  }
115
113
 
116
- # The superset of structural noise (structural ∪ agent-injection), WITHOUT
117
- # interrupt/stop-hook — those carry pushback and must never live here.
114
+ # Structural ∪ agent-injection, WITHOUT interrupt/stop-hook — those carry pushback and must never live here.
118
115
  STRUCTURAL_NOISE_GROUPS: tuple[tuple[str, str], ...] = STRUCTURAL_GROUPS + AGENT_INJECTION_GROUPS
119
116
 
120
- # The sentiment junk set: structural (narrow) + interrupt + stop-hook. This
121
- # reproduces the historical monolithic JUNK_USER_MESSAGE_RE behavior.
122
- SENTIMENT_JUNK_GROUPS: tuple[tuple[str, str], ...] = STRUCTURAL_GROUPS + INTERRUPT_MARKER_GROUPS + STOP_HOOK_GROUPS
117
+ # The historical monolithic JUNK_USER_MESSAGE_RE plus bash-mode command echoes.
118
+ SENTIMENT_JUNK_GROUPS: tuple[tuple[str, str], ...] = (
119
+ STRUCTURAL_GROUPS + INTERRUPT_MARKER_GROUPS + STOP_HOOK_GROUPS + COMMAND_ECHO_GROUPS
120
+ )
123
121
 
124
122
  FRUSTRATION_GROUPS: tuple[tuple[str, str], ...] = (
125
123
  (
@@ -351,9 +349,14 @@ class FilterSpec:
351
349
  clauses: tuple[Clause, ...]
352
350
 
353
351
 
352
+ def group_pattern(groups: tuple[tuple[str, str], ...]) -> str:
353
+ """Composes named regex groups into one non-capturing alternation."""
354
+ return "|".join(f"(?:{pattern})" for _, pattern in groups)
355
+
356
+
354
357
  @cache
355
- def compile_groups(groups: tuple[tuple[str, str], ...], ignore_case: bool) -> re.Pattern[str]:
356
- return re.compile("|".join(f"(?:{pattern})" for _, pattern in groups), re.IGNORECASE if ignore_case else 0)
358
+ def compile_groups(groups: tuple[tuple[str, str], ...], ignore_case: bool, *, multiline: bool = False) -> re.Pattern[str]:
359
+ return re.compile(group_pattern(groups), (re.IGNORECASE if ignore_case else 0) | (re.MULTILINE if multiline else 0))
357
360
 
358
361
 
359
362
  def event_kind(event: TranscriptEvent) -> EventKind:
@@ -0,0 +1,87 @@
1
+ """The append-only SQLite base shared by the correction and decision ledgers.
2
+
3
+ Both family ledgers are one durable, WAL-mode, ``INSERT OR IGNORE`` table behind
4
+ a fixed schema, so :class:`SyncLedger` owns the connection, the ``open`` plumbing,
5
+ and the schema-driven append and read — each ledger supplies only its DDL,
6
+ filename, table, columns, and row mapper.
7
+
8
+ Import-light by contract, like :mod:`cc_transcript.ids`: the standard library
9
+ plus identity primitives only, so a hook reading a ledger pays nothing for the
10
+ parser.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import sqlite3
17
+ from abc import ABC, abstractmethod
18
+ from pathlib import Path
19
+ from typing import TYPE_CHECKING, ClassVar, Protocol, Self
20
+
21
+ if TYPE_CHECKING:
22
+ from collections.abc import Mapping
23
+ from typing import Any
24
+
25
+ from cc_transcript.ids import SessionId
26
+
27
+
28
+ class LedgerRecord(Protocol):
29
+ detail: Mapping[str, Any]
30
+
31
+
32
+ class SyncLedger[R: LedgerRecord](ABC):
33
+ DDL: ClassVar[str]
34
+ FILENAME: ClassVar[str]
35
+ TABLE: ClassVar[str]
36
+ COLUMNS: ClassVar[tuple[str, ...]]
37
+
38
+ def __init__(self, conn: sqlite3.Connection) -> None:
39
+ self.conn = conn
40
+
41
+ @abstractmethod
42
+ def row_to_record(self, row: sqlite3.Row) -> R: ...
43
+
44
+ @classmethod
45
+ def open(cls, path: Path | None = None) -> Self:
46
+ """Opens (creating if needed) the ledger at ``path``.
47
+
48
+ Args:
49
+ path: The database file path; its parents are created if absent.
50
+ Defaults to the ledger's file under ``~/.cc-transcript``.
51
+
52
+ Returns:
53
+ The opened log.
54
+ """
55
+ path = path or Path.home() / ".cc-transcript" / cls.FILENAME
56
+ path.parent.mkdir(parents=True, exist_ok=True)
57
+ conn = sqlite3.connect(path, autocommit=True)
58
+ conn.row_factory = sqlite3.Row
59
+ conn.execute("PRAGMA journal_mode = WAL")
60
+ conn.execute("PRAGMA busy_timeout = 2000")
61
+ conn.executescript(cls.DDL)
62
+ return cls(conn)
63
+
64
+ def append(self, record: R) -> None:
65
+ """Appends ``record`` as a single ``INSERT OR IGNORE``.
66
+
67
+ Idempotent on the table's UNIQUE key, so re-running a writer writes one
68
+ row; SQLite treats NULL key columns as distinct, so rows whose key
69
+ carries a NULL rely on the writer not repeating the same values.
70
+ """
71
+ self.conn.execute(
72
+ f"INSERT OR IGNORE INTO {self.TABLE} ({', '.join(self.COLUMNS)}) "
73
+ f"VALUES ({', '.join(['?'] * len(self.COLUMNS))})",
74
+ tuple(
75
+ json.dumps(dict(record.detail)) if column == "detail_json" else getattr(record, column)
76
+ for column in self.COLUMNS
77
+ ),
78
+ )
79
+
80
+ def for_session(self, session_id: SessionId) -> tuple[R, ...]:
81
+ """All records for ``session_id``, ordered by timestamp."""
82
+ return tuple(
83
+ self.row_to_record(row)
84
+ for row in self.conn.execute(
85
+ f"SELECT * FROM {self.TABLE} WHERE session_id = ? ORDER BY ts_ms, id", (session_id,)
86
+ )
87
+ )