ktalk-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ktalk_cli/registry.py ADDED
@@ -0,0 +1,562 @@
1
+ """SQLite operational store for the KTalk recordings registry."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ import sqlite3
8
+ from datetime import date, timedelta
9
+ from pathlib import Path
10
+
11
+ SCHEMA_VERSION = "1"
12
+
13
+ _VALID_STATUSES = {"new", "processing", "done", "skipped", "partial"}
14
+
15
+ _SCHEMA = """
16
+ CREATE TABLE IF NOT EXISTS recordings (
17
+ recording_id TEXT PRIMARY KEY,
18
+ name TEXT NOT NULL,
19
+ date TEXT NOT NULL,
20
+ duration_min INTEGER,
21
+ status TEXT NOT NULL,
22
+ meeting_type TEXT,
23
+ transcript_path TEXT,
24
+ protocol_path TEXT,
25
+ processed_at TEXT,
26
+ created_at TEXT NOT NULL,
27
+ updated_at TEXT NOT NULL,
28
+ raw_json TEXT
29
+ );
30
+
31
+ CREATE TABLE IF NOT EXISTS participants (
32
+ recording_id TEXT NOT NULL REFERENCES recordings(recording_id) ON DELETE CASCADE,
33
+ ktalk_id TEXT NOT NULL,
34
+ name TEXT NOT NULL,
35
+ vault_id TEXT,
36
+ PRIMARY KEY (recording_id, ktalk_id)
37
+ );
38
+
39
+ CREATE TABLE IF NOT EXISTS meta (
40
+ key TEXT PRIMARY KEY,
41
+ value TEXT
42
+ );
43
+
44
+ CREATE INDEX IF NOT EXISTS idx_recordings_status ON recordings(status);
45
+ CREATE INDEX IF NOT EXISTS idx_recordings_date ON recordings(date);
46
+ """
47
+
48
+
49
+ def today_str() -> str:
50
+ """Today's date as YYYY-MM-DD."""
51
+ return date.today().isoformat()
52
+
53
+
54
+ def _api_user_name(user_info: dict) -> str:
55
+ surname = user_info.get("surname")
56
+ firstname = user_info.get("firstname")
57
+ if surname and firstname:
58
+ return f"{surname} {firstname}"
59
+ if surname:
60
+ return surname
61
+ if firstname:
62
+ return firstname
63
+ return user_info.get("login") or "Неизвестный"
64
+
65
+
66
+ def recording_fields_from_api(rec: dict) -> dict:
67
+ """Map a KTalk /api/recordings entity to registry row fields."""
68
+ rid = rec.get("id") or rec.get("key") or ""
69
+ created = (rec.get("createdDate") or "")[:10]
70
+ duration = rec.get("duration", 0) or 0
71
+ return {
72
+ "recording_id": rid,
73
+ "name": rec.get("title", "Без названия"),
74
+ "date": created,
75
+ "duration_min": round(duration / 60),
76
+ "raw_json": json.dumps(rec, ensure_ascii=False),
77
+ }
78
+
79
+
80
+ def participants_from_api(rec: dict) -> list[dict]:
81
+ """Extract deduped participants with resolvable ktalk_id from an API entity."""
82
+ seen: set[str] = set()
83
+ out: list[dict] = []
84
+ for p in rec.get("participants", []) or []:
85
+ info = p.get("userInfo")
86
+ if not info:
87
+ continue
88
+ ktalk_id = info.get("key") or info.get("login")
89
+ if not ktalk_id or ktalk_id in seen:
90
+ continue
91
+ seen.add(ktalk_id)
92
+ out.append({"ktalk_id": str(ktalk_id), "name": _api_user_name(info)})
93
+ return out
94
+
95
+
96
+ _KTALK_RE = re.compile(r"\(ktalk:([^)]+)\)")
97
+
98
+
99
+ def parse_participants_field(raw: str) -> list[dict]:
100
+ """Parse 'Имя Фамилия (ktalk:ID), ...' into participant dicts."""
101
+ out: list[dict] = []
102
+ for part in raw.split(","):
103
+ part = part.strip()
104
+ m = _KTALK_RE.search(part)
105
+ if not m:
106
+ continue
107
+ name = _KTALK_RE.sub("", part).strip()
108
+ out.append({"ktalk_id": m.group(1).strip(), "name": name})
109
+ return out
110
+
111
+
112
+ def parse_duration_field(raw: str) -> int | None:
113
+ """Parse '61 мин' / '1 ч 5 мин' into minutes; '—'/'' -> None."""
114
+ raw = raw.strip()
115
+ if not raw or raw == "—":
116
+ return None
117
+ hours = re.search(r"(\d+)\s*ч", raw)
118
+ minutes = re.search(r"(\d+)\s*мин", raw)
119
+ total = 0
120
+ if hours:
121
+ total += int(hours.group(1)) * 60
122
+ if minutes:
123
+ total += int(minutes.group(1))
124
+ return total or None
125
+
126
+
127
+ def _table_rows(text: str) -> list[list[str]]:
128
+ rows: list[list[str]] = []
129
+ for line in text.splitlines():
130
+ line = line.strip()
131
+ if not line.startswith("|"):
132
+ continue
133
+ cells = [c.strip() for c in line.strip("|").split("|")]
134
+ if not cells or cells[0] == "recording_id":
135
+ continue
136
+ if all(set(c) <= {"-", ":"} for c in cells): # separator row
137
+ continue
138
+ rows.append(cells)
139
+ return rows
140
+
141
+
142
+ def _nullable_path(value: str) -> str | None:
143
+ value = value.strip()
144
+ return None if value in ("", "—") else value
145
+
146
+
147
+ def parse_unprocessed_table(text: str) -> list[dict]:
148
+ """Parse the 6-column 'Необработанные записи' table."""
149
+ out: list[dict] = []
150
+ for cells in _table_rows(text):
151
+ if len(cells) < 6:
152
+ continue
153
+ out.append(
154
+ {
155
+ "recording_id": cells[0],
156
+ "name": cells[1],
157
+ "participants": parse_participants_field(cells[2]),
158
+ "date": cells[3],
159
+ "duration_min": parse_duration_field(cells[4]),
160
+ "status": cells[5],
161
+ }
162
+ )
163
+ return out
164
+
165
+
166
+ _DATE_RE = re.compile(r"^\d{4}-\d{2}-\d{2}$")
167
+
168
+
169
+ def _clean_archive_name(fragments: list[str]) -> str:
170
+ """Join name fragments split by a literal '|' in the cell, dropping escape slashes."""
171
+ parts = []
172
+ for frag in fragments:
173
+ frag = frag.strip().rstrip("\\").strip()
174
+ if frag:
175
+ parts.append(frag)
176
+ return " | ".join(parts)
177
+
178
+
179
+ def parse_archive_table(text: str) -> list[dict]:
180
+ """Parse archive tables, tolerant of column-count variation and stray pipes.
181
+
182
+ Anchors recording_id from the left and the four status/path columns from the
183
+ right; detects date and participants by content in the middle. This survives
184
+ the 7-column and 8-column ('Участники') layouts as well as rows with an
185
+ escaped/unescaped '|' inside the title or a duplicated id cell.
186
+ """
187
+ out: list[dict] = []
188
+ for cells in _table_rows(text):
189
+ if len(cells) < 7:
190
+ continue
191
+ rid = cells[0]
192
+ protocol = _nullable_path(cells[-1])
193
+ transcript = _nullable_path(cells[-2])
194
+ processed_at = _nullable_path(cells[-3])
195
+ status = cells[-4].strip()
196
+ middle = cells[1:-4] # name [| name...], date, [participants]
197
+ date = ""
198
+ participants_raw = ""
199
+ name_fragments: list[str] = []
200
+ for cell in middle:
201
+ value = cell.strip()
202
+ if not date and _DATE_RE.match(value):
203
+ date = value
204
+ elif "ktalk:" in value:
205
+ participants_raw = value
206
+ else:
207
+ name_fragments.append(cell)
208
+ out.append(
209
+ {
210
+ "recording_id": rid,
211
+ "name": _clean_archive_name(name_fragments),
212
+ "date": date,
213
+ "status": status,
214
+ "processed_at": processed_at,
215
+ "transcript_path": transcript,
216
+ "protocol_path": protocol,
217
+ "participants": parse_participants_field(participants_raw),
218
+ }
219
+ )
220
+ return out
221
+
222
+
223
+ def migrate_from_vault(
224
+ registry: Registry,
225
+ vault_path: str | Path,
226
+ *,
227
+ dry_run: bool = False,
228
+ now: str | None = None,
229
+ ) -> dict:
230
+ """Import existing markdown registries into the SQLite store. Idempotent."""
231
+ now = now or today_str()
232
+ base = Path(vault_path) / "95_TRANSCRIPTS"
233
+ by_status: dict[str, int] = {}
234
+ recordings = 0
235
+ participants = 0
236
+
237
+ def _count(status: str) -> None:
238
+ by_status[status] = by_status.get(status, 0) + 1
239
+
240
+ reg_md = base / "registry.md"
241
+ if reg_md.exists():
242
+ for row in parse_unprocessed_table(reg_md.read_text(encoding="utf-8")):
243
+ recordings += 1
244
+ participants += len(row["participants"])
245
+ _count(row["status"])
246
+ if not dry_run:
247
+ registry.upsert_recording(
248
+ {
249
+ "recording_id": row["recording_id"],
250
+ "name": row["name"],
251
+ "date": row["date"],
252
+ "duration_min": row["duration_min"],
253
+ "status": row["status"],
254
+ },
255
+ participants=row["participants"],
256
+ now=now,
257
+ )
258
+ if row["status"] != "new":
259
+ registry.set_status(row["recording_id"], row["status"], now=now)
260
+
261
+ for archive in sorted(base.glob("registry-archive-*.md")):
262
+ for row in parse_archive_table(archive.read_text(encoding="utf-8")):
263
+ recordings += 1
264
+ participants += len(row["participants"])
265
+ _count(row["status"])
266
+ if not dry_run:
267
+ registry.upsert_recording(
268
+ {
269
+ "recording_id": row["recording_id"],
270
+ "name": row["name"],
271
+ "date": row["date"],
272
+ "status": row["status"],
273
+ },
274
+ participants=row["participants"] or None,
275
+ now=now,
276
+ )
277
+ registry.set_status(
278
+ row["recording_id"],
279
+ row["status"],
280
+ now=now,
281
+ transcript_path=row["transcript_path"],
282
+ protocol_path=row["protocol_path"],
283
+ processed_at=row["processed_at"],
284
+ )
285
+
286
+ return {
287
+ "recordings": recordings,
288
+ "participants": participants,
289
+ "by_status": by_status,
290
+ }
291
+
292
+
293
+ _MIRROR_HEADER = "<!-- GENERATED by ktalk export — НЕ редактировать вручную -->"
294
+
295
+
296
+ def _recent_months(now: str) -> set[str]:
297
+ d = date.fromisoformat(now)
298
+ cur = f"{d.year:04d}-{d.month:02d}"
299
+ if d.month == 1:
300
+ prev = f"{d.year - 1:04d}-12"
301
+ else:
302
+ prev = f"{d.year:04d}-{d.month - 1:02d}"
303
+ return {cur, prev}
304
+
305
+
306
+ def render_markdown_mirror(
307
+ registry: Registry, *, full: bool = False, now: str | None = None
308
+ ) -> str:
309
+ """Render a read-only markdown mirror of the registry for git visibility."""
310
+ now = now or today_str()
311
+ recs = registry.list_recordings()
312
+ unprocessed = [r for r in recs if r["status"] in ("new", "processing", "partial")]
313
+ processed = [r for r in recs if r["status"] in ("done", "skipped")]
314
+ if not full:
315
+ months = _recent_months(now)
316
+ processed = [r for r in processed if r["date"][:7] in months]
317
+
318
+ lines = [
319
+ _MIRROR_HEADER,
320
+ "",
321
+ "# Реестр записей Kontur Talk",
322
+ "",
323
+ "## Необработанные записи",
324
+ "",
325
+ "| recording_id | Название | Участники | Дата | Длительность | Статус |",
326
+ "|---|---|---|---|---|---|",
327
+ ]
328
+ for r in unprocessed:
329
+ parts = registry.get_participants(r["recording_id"])
330
+ parts_str = ", ".join(f"{p['name']} (ktalk:{p['ktalk_id']})" for p in parts)
331
+ dur = f"{r['duration_min']} мин" if r["duration_min"] else "—"
332
+ lines.append(
333
+ f"| {r['recording_id']} | {r['name']} | {parts_str} | {r['date']} | "
334
+ f"{dur} | {r['status']} |"
335
+ )
336
+
337
+ lines += [
338
+ "",
339
+ "## Обработанные записи",
340
+ "",
341
+ "| recording_id | Название | Дата | Статус | Дата обработки | "
342
+ "Путь транскрипта | Путь протокола |",
343
+ "|---|---|---|---|---|---|---|",
344
+ ]
345
+ for r in processed:
346
+ lines.append(
347
+ f"| {r['recording_id']} | {r['name']} | {r['date']} | {r['status']} | "
348
+ f"{r['processed_at'] or '—'} | {r['transcript_path'] or '—'} | "
349
+ f"{r['protocol_path'] or '—'} |"
350
+ )
351
+
352
+ return "\n".join(lines) + "\n"
353
+
354
+
355
+ class Registry:
356
+ """SQLite-backed registry. One connection per instance; commit per write."""
357
+
358
+ def __init__(self, db_path: str | Path) -> None:
359
+ self._conn = sqlite3.connect(str(db_path))
360
+ self._conn.row_factory = sqlite3.Row
361
+ self._conn.execute("PRAGMA journal_mode=WAL")
362
+ self._conn.execute("PRAGMA busy_timeout=5000")
363
+ self._conn.execute("PRAGMA foreign_keys=ON")
364
+ self._conn.executescript(_SCHEMA)
365
+ self._conn.commit()
366
+ if self.get_meta("schema_version") is None:
367
+ self.set_meta("schema_version", SCHEMA_VERSION)
368
+
369
+ def __enter__(self) -> Registry:
370
+ return self
371
+
372
+ def __exit__(self, *args: object) -> None:
373
+ self.close()
374
+
375
+ def close(self) -> None:
376
+ self._conn.close()
377
+
378
+ def get_meta(self, key: str) -> str | None:
379
+ row = self._conn.execute("SELECT value FROM meta WHERE key=?", (key,)).fetchone()
380
+ return row[0] if row else None
381
+
382
+ def set_meta(self, key: str, value: str) -> None:
383
+ self._conn.execute(
384
+ "INSERT INTO meta(key, value) VALUES(?, ?) "
385
+ "ON CONFLICT(key) DO UPDATE SET value=excluded.value",
386
+ (key, value),
387
+ )
388
+ self._conn.commit()
389
+
390
+ def _row_to_dict(self, row: sqlite3.Row | None) -> dict | None:
391
+ return dict(row) if row is not None else None
392
+
393
+ def get_recording(self, recording_id: str) -> dict | None:
394
+ row = self._conn.execute(
395
+ "SELECT * FROM recordings WHERE recording_id=?", (recording_id,)
396
+ ).fetchone()
397
+ return self._row_to_dict(row)
398
+
399
+ def get_participants(self, recording_id: str) -> list[dict]:
400
+ rows = self._conn.execute(
401
+ "SELECT * FROM participants WHERE recording_id=? ORDER BY ktalk_id",
402
+ (recording_id,),
403
+ ).fetchall()
404
+ return [dict(r) for r in rows]
405
+
406
+ def upsert_recording(
407
+ self,
408
+ fields: dict,
409
+ participants: list[dict] | None = None,
410
+ *,
411
+ now: str | None = None,
412
+ ) -> str:
413
+ now = now or today_str()
414
+ rid = fields["recording_id"]
415
+ existing = self.get_recording(rid)
416
+ if existing is None:
417
+ self._conn.execute(
418
+ "INSERT INTO recordings("
419
+ "recording_id, name, date, duration_min, status, raw_json, "
420
+ "created_at, updated_at) VALUES(?,?,?,?,?,?,?,?)",
421
+ (
422
+ rid,
423
+ fields.get("name", ""),
424
+ fields.get("date", ""),
425
+ fields.get("duration_min"),
426
+ fields.get("status", "new"),
427
+ fields.get("raw_json"),
428
+ now,
429
+ now,
430
+ ),
431
+ )
432
+ result = "inserted"
433
+ else:
434
+ self._conn.execute(
435
+ "UPDATE recordings SET name=?, date=?, duration_min=?, raw_json=?, "
436
+ "updated_at=? WHERE recording_id=?",
437
+ (
438
+ fields.get("name", existing["name"]),
439
+ fields.get("date", existing["date"]),
440
+ fields.get("duration_min", existing["duration_min"]),
441
+ fields.get("raw_json", existing["raw_json"]),
442
+ now,
443
+ rid,
444
+ ),
445
+ )
446
+ result = "updated"
447
+ if participants is not None:
448
+ self._conn.execute(
449
+ "DELETE FROM participants WHERE recording_id=?", (rid,)
450
+ )
451
+ self._conn.executemany(
452
+ "INSERT INTO participants(recording_id, ktalk_id, name, vault_id) "
453
+ "VALUES(?,?,?,?)",
454
+ [
455
+ (rid, p["ktalk_id"], p["name"], p.get("vault_id"))
456
+ for p in participants
457
+ ],
458
+ )
459
+ self._conn.commit()
460
+ return result
461
+
462
+ def set_status(
463
+ self, recording_id: str, status: str, *, now: str | None = None, **fields
464
+ ) -> None:
465
+ if status not in _VALID_STATUSES:
466
+ raise ValueError(f"Недопустимый статус: {status}")
467
+ if self.get_recording(recording_id) is None:
468
+ raise KeyError(recording_id)
469
+ now = now or today_str()
470
+ cols = ["status=?", "updated_at=?"]
471
+ vals: list = [status, now]
472
+ for key in ("transcript_path", "protocol_path", "meeting_type", "processed_at"):
473
+ if key in fields:
474
+ cols.append(f"{key}=?")
475
+ vals.append(fields[key])
476
+ vals.append(recording_id)
477
+ self._conn.execute(
478
+ f"UPDATE recordings SET {', '.join(cols)} WHERE recording_id=?", vals
479
+ )
480
+ self._conn.commit()
481
+
482
+ def mark_processing(self, recording_id: str, *, now: str | None = None) -> None:
483
+ self.set_status(recording_id, "processing", now=now)
484
+
485
+ def mark_done(
486
+ self,
487
+ recording_id: str,
488
+ *,
489
+ transcript_path: str | None = None,
490
+ protocol_path: str | None = None,
491
+ meeting_type: str | None = None,
492
+ now: str | None = None,
493
+ ) -> None:
494
+ now = now or today_str()
495
+ self.set_status(
496
+ recording_id,
497
+ "done",
498
+ now=now,
499
+ transcript_path=transcript_path,
500
+ protocol_path=protocol_path,
501
+ meeting_type=meeting_type,
502
+ processed_at=now,
503
+ )
504
+
505
+ def mark_partial(
506
+ self,
507
+ recording_id: str,
508
+ *,
509
+ transcript_path: str | None = None,
510
+ protocol_path: str | None = None,
511
+ now: str | None = None,
512
+ ) -> None:
513
+ self.set_status(
514
+ recording_id,
515
+ "partial",
516
+ now=now,
517
+ transcript_path=transcript_path,
518
+ protocol_path=protocol_path,
519
+ )
520
+
521
+ def mark_skipped(self, recording_id: str, *, now: str | None = None) -> None:
522
+ now = now or today_str()
523
+ self.set_status(recording_id, "skipped", now=now, processed_at=now)
524
+
525
+ def list_recordings(self, status: str | None = None) -> list[dict]:
526
+ if status is None:
527
+ rows = self._conn.execute(
528
+ "SELECT * FROM recordings ORDER BY date DESC, recording_id"
529
+ ).fetchall()
530
+ else:
531
+ rows = self._conn.execute(
532
+ "SELECT * FROM recordings WHERE status=? ORDER BY date DESC, recording_id",
533
+ (status,),
534
+ ).fetchall()
535
+ return [dict(r) for r in rows]
536
+
537
+ def set_vault_id(self, recording_id: str, ktalk_id: str, vault_id: str) -> None:
538
+ cur = self._conn.execute(
539
+ "UPDATE participants SET vault_id=? WHERE recording_id=? AND ktalk_id=?",
540
+ (vault_id, recording_id, ktalk_id),
541
+ )
542
+ if cur.rowcount == 0:
543
+ raise KeyError((recording_id, ktalk_id))
544
+ self._conn.commit()
545
+
546
+ def expire_new(self, *, now: str | None = None, days: int = 7) -> list[str]:
547
+ now = now or today_str()
548
+ cutoff = (date.fromisoformat(now) - timedelta(days=days)).isoformat()
549
+ rows = self._conn.execute(
550
+ "SELECT recording_id FROM recordings WHERE status='new' AND date < ? "
551
+ "ORDER BY recording_id",
552
+ (cutoff,),
553
+ ).fetchall()
554
+ expired = [r[0] for r in rows]
555
+ if expired:
556
+ self._conn.execute(
557
+ "UPDATE recordings SET status='skipped', processed_at=?, updated_at=? "
558
+ "WHERE status='new' AND date < ?",
559
+ (now, now, cutoff),
560
+ )
561
+ self._conn.commit()
562
+ return expired
ktalk_cli/rooms.py ADDED
@@ -0,0 +1,62 @@
1
+ """FR-17: чтение комнаты — маппер 18 полей + сетевой вызов вне `client.py` (гейт C13).
2
+
3
+ Вынесено свободной функцией по тому же приёму, что `auth.py::full_participants_apikey`
4
+ применяет к клиенту сегодня — не новый метод `KTalkClient`.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from ktalk_cli.auth import quote_path_param
10
+ from ktalk_cli.client import KTalkClient
11
+ from ktalk_cli.contour_diagnostics import (
12
+ TRANSIENT_ERRORS,
13
+ diagnose_undocumented_failure,
14
+ require_contract_field,
15
+ )
16
+
17
+ ROOM_FIELDS = (
18
+ "roomName",
19
+ "sessionHalls",
20
+ "stageConferenceId",
21
+ "moderators",
22
+ "anonymousModerators",
23
+ "allowAnonymous",
24
+ "anonymousAccessExpirationDate",
25
+ "anonymousAccessModifiedDate",
26
+ "audioPolicy",
27
+ "videoPolicy",
28
+ "screenSharePolicy",
29
+ "isModerator",
30
+ "conferenceId",
31
+ "sipSettings",
32
+ "onlineUsers",
33
+ "simultaneousTranslation",
34
+ "chatChannelSettings",
35
+ "maskingSettings",
36
+ )
37
+
38
+
39
+ def map_room(raw: dict) -> dict:
40
+ require_contract_field(raw, "roomName", "get_room") # якорь контракта
41
+ return {field: raw.get(field) for field in ROOM_FIELDS}
42
+
43
+
44
+ async def get_room(client: KTalkClient, room_name: str) -> dict:
45
+ """Читает конфигурацию комнаты по имени.
46
+
47
+ ПОБОЧНЫЙ ЭФФЕКТ (ADR-006): сервер не различает существующее и ранее не
48
+ встречавшееся имя — оба дают 200 с идентичной по форме заготовкой (Ф-45).
49
+ Вызов с именем, которое в этом контуре ещё не читалось, создаёт объект
50
+ комнаты (Ф-49); отменить средствами проекта нельзя (инструмента удаления
51
+ нет). НЕ использовать эту операцию для проверки занятости/свободности
52
+ имени — сам факт проверки создаёт занятость.
53
+ """
54
+ profile = client._profile_for("get_room") # noqa: SLF001 - fail-closed до сети (api-key)
55
+ path = profile.path_template.format(room_name=quote_path_param(room_name))
56
+ try:
57
+ response = await client._client.get(path) # noqa: SLF001
58
+ client._classify(response, profile.required_scope) # noqa: SLF001
59
+ except TRANSIENT_ERRORS as exc:
60
+ await diagnose_undocumented_failure(client, "get_room", exc)
61
+ raise # недостижимо: diagnose_undocumented_failure всегда поднимает исключение
62
+ return map_room(response.json())