msteams-transcripts 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,228 @@
1
+ """Command line: browser, list, get, batch."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import asyncio
6
+ import json
7
+ import re
8
+ import sys
9
+ from datetime import datetime, timezone
10
+ from pathlib import Path
11
+
12
+ from . import __version__
13
+ from .browser import ensure_browser, log
14
+ from .config import PROG, SETTINGS, configure
15
+ from .download import download_ref
16
+ from .errors import TranscriptError
17
+ from .model import parse_date, refs_from_recap_url, thread_id_from, transcript_refs_from_contents
18
+ from .session import TeamsSession
19
+
20
+ DESCRIPTION = """\
21
+ Download Microsoft Teams meeting transcripts from the command line.
22
+
23
+ Start the dedicated browser once with the 'browser' command and sign in to Teams
24
+ there. Everything else replays the requests the Teams web client makes, so it
25
+ reaches exactly the meetings you can already open.
26
+ """
27
+
28
+ EPILOG = f"""\
29
+ examples:
30
+ {PROG} browser
31
+ {PROG} list --from 2026-09-01 --to today
32
+ {PROG} get 3 --details
33
+ {PROG} get "https://teams.cloud.microsoft/l/meetingrecap?driveId=...&driveItemId=..." -f all
34
+ {PROG} batch --from 2026-09-01 --to 2026-09-30 --details -o ./transcripts
35
+ """
36
+
37
+
38
+ def default_out() -> str:
39
+ return str(Path.cwd() / "transcripts")
40
+
41
+
42
+ async def cmd_list(args) -> list[dict]:
43
+ """Print one row per transcript and cache the rows for 'get N'."""
44
+ start, end = parse_date(args.frm), parse_date(args.to, end=True)
45
+ async with TeamsSession() as s:
46
+ events = await s.calendar(start, end)
47
+ now = datetime.now(timezone.utc)
48
+ meetings = []
49
+ for e in events:
50
+ if not e.get("isOnlineMeeting") or e.get("isCancelled"):
51
+ continue
52
+ tid = thread_id_from(e.get("skypeTeamsMeetingUrl", "")) or thread_id_from(
53
+ json.dumps(e.get("skypeTeamsDataObject") or e.get("skypeTeamsData") or "")
54
+ )
55
+ if not tid:
56
+ continue
57
+ try:
58
+ ends = datetime.fromisoformat(e.get("endTime", "").replace("Z", "+00:00"))
59
+ except Exception:
60
+ ends = now
61
+ if ends > now:
62
+ continue
63
+ meetings.append(
64
+ {
65
+ "subject": e.get("subject", ""),
66
+ "start": e.get("startTime", ""),
67
+ "end": e.get("endTime", ""),
68
+ "threadId": tid,
69
+ "iCalUID": e.get("iCalUID", ""),
70
+ "organizer": e.get("organizerName", ""),
71
+ "objectId": e.get("objectId", ""),
72
+ }
73
+ )
74
+ log(f"{len(events)} calendar events, {len(meetings)} past online meetings. Checking for transcripts ...")
75
+
76
+ rows: list[dict] = []
77
+ seen_threads: set[str] = set()
78
+ for m in meetings:
79
+ if m["threadId"] in seen_threads:
80
+ continue
81
+ seen_threads.add(m["threadId"])
82
+ contents = await s.meeting_contents(m["threadId"], m["iCalUID"])
83
+ same_thread = [x for x in meetings if x["threadId"] == m["threadId"]]
84
+ for ref in transcript_refs_from_contents(contents, m["subject"]):
85
+ # Match a transcript to the occurrence whose window contains its start.
86
+ occ = next((x for x in same_thread if ref["start"] and x["start"] <= ref["start"] <= x["end"]), None)
87
+ ref["subject"] = m["subject"]
88
+ ref["meetingStart"] = (occ or m)["start"]
89
+ ref["threadId"] = m["threadId"]
90
+ ref["organizer"] = m["organizer"]
91
+ ref["objectId"] = (occ or m)["objectId"]
92
+ ref["iCalUID"] = (occ or m)["iCalUID"]
93
+ if start.isoformat() <= (ref["start"] or ref["meetingStart"]) <= end.isoformat():
94
+ rows.append(ref)
95
+ rows.sort(key=lambda r: r["start"] or r["meetingStart"])
96
+ for i, r in enumerate(rows, 1):
97
+ r["n"] = i
98
+
99
+ SETTINGS.state_dir.mkdir(parents=True, exist_ok=True)
100
+ SETTINGS.last_list.write_text(json.dumps(rows, indent=1), encoding="utf-8")
101
+ if args.json:
102
+ print(json.dumps(rows, indent=1))
103
+ else:
104
+ print(f"{'#':>3} {'Start (UTC)':<17} {'Subject':<50} Organizer")
105
+ for r in rows:
106
+ st = (r["start"] or r["meetingStart"])[:16].replace("T", " ")
107
+ print(f"{r['n']:>3} {st:<17} {r['subject'][:50]:<50} {r['organizer']}")
108
+ log(f"{len(rows)} transcript(s). Use: {PROG} get <#> or {PROG} batch --from ... --to ...")
109
+ return rows
110
+
111
+
112
+ async def _download_by_thread(s: TeamsSession, thread_id: str, args) -> None:
113
+ refs = transcript_refs_from_contents(await s.meeting_contents(thread_id))
114
+ if not refs:
115
+ raise TranscriptError("No transcript was found for that meeting.")
116
+ for r in refs:
117
+ await download_ref(s, r, Path(args.out), args.format, args.details)
118
+
119
+
120
+ async def cmd_get(args) -> None:
121
+ """Download one transcript, named by row number, recap link or thread id."""
122
+ target: str = args.target
123
+ async with TeamsSession() as s:
124
+ if target.isdigit():
125
+ if not SETTINGS.last_list.exists():
126
+ raise TranscriptError(f"No previous list. Run: {PROG} list --from ... --to ...")
127
+ rows = json.loads(SETTINGS.last_list.read_text(encoding="utf-8"))
128
+ ref = next((r for r in rows if r["n"] == int(target)), None)
129
+ if not ref:
130
+ raise TranscriptError(f"There is no row {target} in the last list.")
131
+ elif target.startswith("http"):
132
+ ref = refs_from_recap_url(target)
133
+ if ref is None:
134
+ tid = thread_id_from(target)
135
+ if not tid:
136
+ raise TranscriptError("That link has neither drive identifiers nor a meeting thread id.")
137
+ await _download_by_thread(s, tid, args)
138
+ return
139
+ else:
140
+ await _download_by_thread(s, thread_id_from(target) or target, args)
141
+ return
142
+ if args.sharepoint_host:
143
+ ref["host"] = args.sharepoint_host
144
+ await download_ref(s, ref, Path(args.out), args.format, args.details)
145
+
146
+
147
+ async def cmd_batch(args) -> None:
148
+ """Download every transcript in a date range."""
149
+ rows = await cmd_list(argparse.Namespace(frm=args.frm, to=args.to, json=False))
150
+ if args.only:
151
+ want = {int(x) for x in re.split(r"[,\s]+", args.only.strip()) if x}
152
+ rows = [r for r in rows if r["n"] in want]
153
+ if not rows:
154
+ return
155
+ async with TeamsSession() as s:
156
+ ok = fail = 0
157
+ for r in rows:
158
+ try:
159
+ await download_ref(s, r, Path(args.out), args.format, args.details)
160
+ ok += 1
161
+ except Exception as ex:
162
+ fail += 1
163
+ log(f" FAILED {r.get('subject')}: {ex}")
164
+ log(f"Done: {ok} downloaded, {fail} failed. Output: {Path(args.out).resolve()}")
165
+
166
+
167
+ def build_parser() -> argparse.ArgumentParser:
168
+ ap = argparse.ArgumentParser(
169
+ prog=PROG,
170
+ description=DESCRIPTION,
171
+ epilog=EPILOG,
172
+ formatter_class=argparse.RawDescriptionHelpFormatter,
173
+ )
174
+ ap.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
175
+ ap.add_argument("--port", type=int, default=0, help="browser debugging port (default 9222, or TT_CDP_PORT)")
176
+ ap.add_argument("--profile", default="", help="browser profile directory (or TT_PROFILE_DIR)")
177
+ ap.add_argument("--browser", default="", help="path to the Edge or Chrome executable (or TT_BROWSER_EXE)")
178
+ sub = ap.add_subparsers(dest="cmd", required=True)
179
+
180
+ sub.add_parser("browser", help="start or check the dedicated browser instance")
181
+
182
+ p = sub.add_parser("list", help="list past online meetings that have transcripts")
183
+ p.add_argument("--from", dest="frm", required=True, metavar="YYYY-MM-DD")
184
+ p.add_argument("--to", required=True, metavar="YYYY-MM-DD", help="a date, or 'today'")
185
+ p.add_argument("--json", action="store_true", help="print JSON instead of a table")
186
+
187
+ p = sub.add_parser("get", help="download one transcript")
188
+ p.add_argument("target", help="row number from 'list', a recap URL, or 19:meeting_...@thread.v2")
189
+ p.add_argument("-o", "--out", default=None, help="output directory (default ./transcripts)")
190
+ p.add_argument("-f", "--format", choices=["txt", "json", "vtt", "all"], default="txt")
191
+ p.add_argument("--sharepoint-host", default="", help="override the SharePoint host, e.g. contoso-my.sharepoint.com")
192
+ p.add_argument(
193
+ "-d",
194
+ "--details",
195
+ action="store_true",
196
+ help="add organizer, location, invited attendees, speakers, shared files and invitation text; also writes .meta.json",
197
+ )
198
+
199
+ p = sub.add_parser("batch", help="download every transcript in a date range")
200
+ p.add_argument("--from", dest="frm", required=True, metavar="YYYY-MM-DD")
201
+ p.add_argument("--to", required=True, metavar="YYYY-MM-DD")
202
+ p.add_argument("-o", "--out", default=None, help="output directory (default ./transcripts)")
203
+ p.add_argument("-f", "--format", choices=["txt", "json", "vtt", "all"], default="txt")
204
+ p.add_argument("-d", "--details", action="store_true", help="same as for 'get'")
205
+ p.add_argument("--only", default="", help="comma-separated row numbers instead of all, e.g. 3,7,12")
206
+ return ap
207
+
208
+
209
+ def main(argv: list[str] | None = None) -> None:
210
+ args = build_parser().parse_args(argv)
211
+ configure(port=args.port, profile=args.profile, browser=args.browser)
212
+ if getattr(args, "out", None) is None:
213
+ args.out = default_out()
214
+ try:
215
+ if args.cmd == "browser":
216
+ ensure_browser()
217
+ log(f"Debugging endpoint: {SETTINGS.cdp_url} profile: {SETTINGS.profile_dir}")
218
+ return
219
+ runner = {"list": cmd_list, "get": cmd_get, "batch": cmd_batch}[args.cmd]
220
+ asyncio.run(runner(args))
221
+ except TranscriptError as ex:
222
+ sys.exit(str(ex))
223
+ except KeyboardInterrupt:
224
+ sys.exit(130)
225
+
226
+
227
+ if __name__ == "__main__":
228
+ main()
@@ -0,0 +1,88 @@
1
+ """Runtime settings: where the browser is, which profile it uses, where state lives.
2
+
3
+ Every value can be set by an environment variable, and the command line can
4
+ override them again before any command runs.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ import shutil
10
+ import sys
11
+ from dataclasses import dataclass, field
12
+ from pathlib import Path
13
+
14
+ PROG = "msteams-transcripts"
15
+ TEAMS_URL = "https://teams.cloud.microsoft/"
16
+
17
+ WINDOWS_BROWSERS = [
18
+ r"C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe",
19
+ r"C:\Program Files\Microsoft\Edge\Application\msedge.exe",
20
+ r"C:\Program Files\Google\Chrome\Application\chrome.exe",
21
+ r"C:\Program Files (x86)\Google\Chrome\Application\chrome.exe",
22
+ ]
23
+ MACOS_BROWSERS = [
24
+ "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge",
25
+ "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome",
26
+ ]
27
+ LINUX_BROWSERS = ["microsoft-edge", "microsoft-edge-stable", "google-chrome", "chromium", "chromium-browser"]
28
+
29
+
30
+ def _default_profile_dir() -> Path:
31
+ if sys.platform == "win32":
32
+ base = Path(os.environ.get("LOCALAPPDATA") or Path.home() / "AppData" / "Local")
33
+ elif sys.platform == "darwin":
34
+ base = Path.home() / "Library" / "Application Support"
35
+ else:
36
+ base = Path(os.environ.get("XDG_DATA_HOME") or Path.home() / ".local" / "share")
37
+ return base / "TeamsTranscriptProfile"
38
+
39
+
40
+ def browser_candidates(explicit: str = "") -> list[str]:
41
+ """Browser executables to try, most preferred first."""
42
+ found = [explicit] if explicit else []
43
+ if sys.platform == "win32":
44
+ found += WINDOWS_BROWSERS
45
+ elif sys.platform == "darwin":
46
+ found += MACOS_BROWSERS
47
+ else:
48
+ found += [p for name in LINUX_BROWSERS if (p := shutil.which(name))]
49
+ return [p for p in found if p]
50
+
51
+
52
+ @dataclass
53
+ class Settings:
54
+ cdp_port: int = field(default_factory=lambda: int(os.environ.get("TT_CDP_PORT", "9222")))
55
+ profile_dir: Path = field(
56
+ default_factory=lambda: Path(os.environ["TT_PROFILE_DIR"]).expanduser()
57
+ if os.environ.get("TT_PROFILE_DIR")
58
+ else _default_profile_dir()
59
+ )
60
+ browser_exe: str = field(default_factory=lambda: os.environ.get("TT_BROWSER_EXE", ""))
61
+ state_dir: Path = field(
62
+ default_factory=lambda: Path(os.environ["TT_STATE_DIR"]).expanduser()
63
+ if os.environ.get("TT_STATE_DIR")
64
+ else Path.home() / ".teams_transcripts"
65
+ )
66
+
67
+ @property
68
+ def cdp_url(self) -> str:
69
+ return f"http://127.0.0.1:{self.cdp_port}"
70
+
71
+ @property
72
+ def last_list(self) -> Path:
73
+ """Where the rows printed by the last `list` run are cached."""
74
+ return self.state_dir / "last_list.json"
75
+
76
+
77
+ SETTINGS = Settings()
78
+
79
+
80
+ def configure(port: int | None = None, profile: str = "", browser: str = "") -> Settings:
81
+ """Apply command-line overrides to the process-wide settings."""
82
+ if port:
83
+ SETTINGS.cdp_port = port
84
+ if profile:
85
+ SETTINGS.profile_dir = Path(profile).expanduser()
86
+ if browser:
87
+ SETTINGS.browser_exe = browser
88
+ return SETTINGS
@@ -0,0 +1,108 @@
1
+ """Writing a transcript to disk, optionally with the meeting details."""
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ import re
6
+ from pathlib import Path
7
+
8
+ from .browser import log
9
+ from .errors import TranscriptError
10
+ from .formats import invite_description, speakers_of, transcript_to_text
11
+ from .model import safe_name, shared_files_from_contents
12
+ from .session import TeamsSession
13
+
14
+ # Recording file names look like "<subject>-20260903_141403-Transcription....mp4"
15
+ RECORDING_NAME_RE = re.compile(r"^(.*?)-(\d{8})_(\d{6})(?:-.*)?$")
16
+
17
+
18
+ async def collect_details(s: TeamsSession, ref: dict) -> dict:
19
+ """Calendar event and shared files for a transcript reference."""
20
+ details = {"event": await s.event_details(ref.get("objectId", ""), ref.get("iCalUID", ""))}
21
+ if ref.get("threadId"):
22
+ try:
23
+ details["files"] = shared_files_from_contents(await s.meeting_contents(ref["threadId"], ref.get("iCalUID", "")))
24
+ except TranscriptError:
25
+ details["files"] = []
26
+ return details
27
+
28
+
29
+ def _meta_document(ref: dict, det: dict, doc: dict, subject: str, when: str) -> dict:
30
+ ev = det.get("event") or {}
31
+ return {
32
+ "subject": subject,
33
+ "start": when,
34
+ "organizer": {"name": ev.get("organizerName", ""), "email": ev.get("organizerAddress", "")},
35
+ "location": ev.get("location", ""),
36
+ "attendees": ev.get("attendees", []),
37
+ "speakers": [{"name": n, "segments": c} for n, c in speakers_of(doc)],
38
+ "files": det.get("files", []),
39
+ "description": invite_description(ev.get("bodyContent", "")),
40
+ "threadId": ref.get("threadId", ""),
41
+ "iCalUID": ref.get("iCalUID", ""),
42
+ "source": ref.get("webUrl") or ref.get("location") or "",
43
+ }
44
+
45
+
46
+ async def download_ref(
47
+ s: TeamsSession,
48
+ ref: dict,
49
+ out_dir: Path,
50
+ fmt: str = "txt",
51
+ details: bool = False,
52
+ ) -> list[Path]:
53
+ """Download one transcript reference. Returns the files written.
54
+
55
+ fmt is 'txt', 'json', 'vtt' or 'all'. With details, a header is added to the
56
+ text file and a '.meta.json' sidecar is written alongside it.
57
+ """
58
+ host, drive, item = ref["host"], ref["driveId"], ref["driveItemId"]
59
+ if not host:
60
+ raise TranscriptError("No SharePoint host is known for this transcript; pass --sharepoint-host.")
61
+ tids = [ref["transcriptId"]] if ref.get("transcriptId") else [t["id"] for t in await s.list_transcripts(host, drive, item)]
62
+ if not tids:
63
+ log(f" no transcript found for {ref.get('subject')}")
64
+ return []
65
+
66
+ subject = ref.get("subject") or ref.get("fileTitle") or ""
67
+ name_match = RECORDING_NAME_RE.match(subject)
68
+ if name_match:
69
+ subject = name_match.group(1)
70
+ when = (ref.get("start") or ref.get("meetingStart") or "")[:16].replace("T", " ")
71
+ if not when:
72
+ # The recording's creation time is UTC, matching the times shown by 'list'.
73
+ when = (await s.item_created(host, drive, item) or "")[:16].replace("T", " ")
74
+ if not when and name_match:
75
+ d, t = name_match.group(2), name_match.group(3)
76
+ when = f"{d[:4]}-{d[4:6]}-{d[6:]} {t[:2]}:{t[2:4]}"
77
+
78
+ det = await collect_details(s, ref) if details else None
79
+ if det and det.get("event", {}).get("subject"):
80
+ subject = det["event"]["subject"]
81
+
82
+ stem = safe_name(f"{when.replace(':', '')} - {subject or 'meeting'}".strip(" -"))
83
+ out_dir.mkdir(parents=True, exist_ok=True)
84
+ written: list[Path] = []
85
+ for i, tid in enumerate(tids):
86
+ suffix = f" ({i + 1})" if len(tids) > 1 else ""
87
+ base = out_dir / (stem + suffix)
88
+ doc_json = await s.fetch_transcript(host, drive, item, tid, "json")
89
+ doc = json.loads(doc_json)
90
+ source = ref.get("webUrl") or ref.get("location") or ""
91
+ if fmt in ("txt", "all"):
92
+ p = base.with_suffix(".txt")
93
+ p.write_text(transcript_to_text(doc, subject or stem, when or "unknown", source, det), encoding="utf-8")
94
+ written.append(p)
95
+ if det:
96
+ p = base.with_suffix(".meta.json")
97
+ p.write_text(json.dumps(_meta_document(ref, det, doc, subject, when), indent=1, ensure_ascii=False), encoding="utf-8")
98
+ written.append(p)
99
+ if fmt in ("json", "all"):
100
+ p = base.with_suffix(".json")
101
+ p.write_text(doc_json, encoding="utf-8")
102
+ written.append(p)
103
+ if fmt in ("vtt", "all"):
104
+ p = base.with_suffix(".vtt")
105
+ p.write_text(await s.fetch_transcript(host, drive, item, tid, "vtt"), encoding="utf-8")
106
+ written.append(p)
107
+ log(f" saved {written[-1].name} ({len(doc.get('entries', []))} entries)")
108
+ return written
@@ -0,0 +1,10 @@
1
+ """Exception types raised by the package."""
2
+ from __future__ import annotations
3
+
4
+
5
+ class TranscriptError(Exception):
6
+ """Something went wrong that the user can act on.
7
+
8
+ The command line catches this and prints the message without a traceback,
9
+ so messages should read as advice, not as internal state.
10
+ """
@@ -0,0 +1,120 @@
1
+ """Turning the raw transcript document into readable text."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ from html.parser import HTMLParser
6
+
7
+
8
+ def fmt_offset(off: str | None) -> str:
9
+ """'00:01:02.3450000' to '00:01:02'."""
10
+ if not off:
11
+ return "--:--:--"
12
+ return off.split(".")[0]
13
+
14
+
15
+ class _Text(HTMLParser):
16
+ """Minimal HTML to text, keeping link targets and dropping style blocks."""
17
+
18
+ SKIP = ("style", "script", "head", "title")
19
+
20
+ def __init__(self):
21
+ super().__init__()
22
+ self.parts: list[str] = []
23
+ self.skip = 0
24
+ self.href = ""
25
+
26
+ def handle_data(self, d):
27
+ if not self.skip:
28
+ self.parts.append(d)
29
+
30
+ def handle_starttag(self, tag, attrs):
31
+ if tag in self.SKIP:
32
+ self.skip += 1
33
+ elif tag in ("br", "p", "div", "li", "tr"):
34
+ self.parts.append("\n")
35
+ elif tag == "a":
36
+ href = dict(attrs).get("href", "") or ""
37
+ keep = href.startswith("http") and "/meetup-join/" not in href and "/meet/" not in href
38
+ self.href = href if keep else ""
39
+
40
+ def handle_endtag(self, tag):
41
+ if tag in self.SKIP and self.skip:
42
+ self.skip -= 1
43
+ elif tag == "a" and self.href:
44
+ self.parts.append(f" <{self.href}>")
45
+ self.href = ""
46
+
47
+
48
+ def html_to_text(html: str) -> str:
49
+ t = _Text()
50
+ t.feed(html or "")
51
+ lines = [re.sub(r"\s+", " ", line).strip() for line in "".join(t.parts).splitlines()]
52
+ return "\n".join(line for line in lines if re.search(r"[A-Za-z0-9]", line))
53
+
54
+
55
+ def invite_description(body_html: str) -> str:
56
+ """The invitation text, with the Teams join boilerplate cut off."""
57
+ txt = html_to_text(body_html)
58
+ marks = [i for i in (txt.find("Microsoft Teams meeting"), txt.find("____"), txt.find("Join the meeting now")) if i >= 0]
59
+ return txt[: min(marks)].strip() if marks else txt.strip()
60
+
61
+
62
+ def speakers_of(doc: dict) -> list[tuple[str, int]]:
63
+ """Who spoke, and how many segments each, most talkative first."""
64
+ counts: dict[str, int] = {}
65
+ for e in doc.get("entries", []):
66
+ n = (e.get("speakerDisplayName") or e.get("speakerId") or "Unknown").strip()
67
+ counts[n] = counts.get(n, 0) + 1
68
+ return sorted(counts.items(), key=lambda x: -x[1])
69
+
70
+
71
+ def details_header(details: dict, doc: dict) -> list[str]:
72
+ """Comment lines describing the meeting, placed above the transcript body."""
73
+ ev = details.get("event") or {}
74
+ out: list[str] = []
75
+ if ev.get("organizerName") or ev.get("organizerAddress"):
76
+ out.append(f"# Organizer: {ev.get('organizerName', '')} <{ev.get('organizerAddress', '')}>".replace(" <>", ""))
77
+ if ev.get("location"):
78
+ out.append(f"# Location: {ev['location']}")
79
+ att = ev.get("attendees") or []
80
+ if att:
81
+ out.append(f"# Invited ({len(att)}):")
82
+ for a in att:
83
+ kind = a.get("type") or "Required"
84
+ resp = (a.get("status") or {}).get("response") or ""
85
+ resp = "" if resp in ("None", "NotResponded", "") else f", {resp}"
86
+ out.append(f"# {a.get('name', '')} <{a.get('address', '')}> ({kind}{resp})")
87
+ sp = speakers_of(doc)
88
+ if sp:
89
+ out.append(f"# Speakers in transcript ({len(sp)}):")
90
+ out.extend(f"# {n} ({c} segments)" for n, c in sp)
91
+ files = details.get("files") or []
92
+ if files:
93
+ out.append(f"# Files shared in meeting chat ({len(files)}):")
94
+ out.extend(f"# {f['title']} {f['url']}" for f in files)
95
+ is_html = (ev.get("bodyContentType") or "html").lower() == "html"
96
+ desc = invite_description(ev.get("bodyContent", "")) if is_html else (ev.get("bodyContent") or "")
97
+ if desc:
98
+ out.append("# Invitation text:")
99
+ out.extend(f"# {line}" for line in desc.splitlines()[:40])
100
+ return out
101
+
102
+
103
+ def transcript_to_text(doc: dict, title: str, when: str, source: str, details: dict | None = None) -> str:
104
+ """Plain text: a comment header, then '[hh:mm:ss] Speaker:' with their lines."""
105
+ lines = [f"# {title}", f"# Date: {when}", f"# Source: {source}"]
106
+ if details:
107
+ lines += details_header(details, doc)
108
+ lines.append("")
109
+ prev_speaker = None
110
+ for e in doc.get("entries", []):
111
+ speaker = (e.get("speakerDisplayName") or e.get("speakerId") or "Unknown").strip()
112
+ text = (e.get("text") or "").strip()
113
+ if not text:
114
+ continue
115
+ if speaker != prev_speaker:
116
+ lines.append("")
117
+ lines.append(f"[{fmt_offset(e.get('startOffset'))}] {speaker}:")
118
+ prev_speaker = speaker
119
+ lines.append(f" {text}")
120
+ return "\n".join(lines).strip() + "\n"