teems 0.3.1 → 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,534 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'English'
4
+ require 'date'
5
+ require 'open3'
6
+ require 'tmpdir'
7
+
8
+ module Teems
9
+ module Commands
10
+ # Fetches saved Teams transcripts for calendar meetings. Transcript retrieval still
11
+ # goes through `teems meeting`, keeping its SharePoint/auth implementation in one place.
12
+ class Transcripts < Base
13
+ HELP = <<~HELP
14
+ teems transcripts - Sync saved meeting transcripts onto this machine
15
+
16
+ USAGE:
17
+ teems transcripts sync [--since DAYS | --date YYYY-MM-DD] [--dry-run] [--no-post-sync]
18
+
19
+ OPTIONS:
20
+ --since DAYS Calendar lookback (default 7; first run 30)
21
+ --date YYYY-MM-DD Only scan one calendar date
22
+ --dry-run List meetings and recording counts without downloading
23
+ --no-post-sync Skip the configured post-sync command for this run
24
+ -q, --quiet Suppress progress (not errors)
25
+
26
+ WebVTTs: ~/.local/share/teems/transcripts/
27
+ Markdown (for local search, e.g. qmd): ~/.local/share/teems/transcripts-md/
28
+ Sync state: ~/.local/state/teems/transcript-sync.json
29
+ Only meetings on your Teams calendar with a saved, accessible recording
30
+ transcript can be retrieved; each recording's transcript is kept (a meeting
31
+ restarted mid-session has several). Does not alter meeting-capture files.
32
+
33
+ POST-SYNC COMMAND:
34
+ Set "transcripts": {"post_sync_command": "..."} in ~/.config/teems/config.json to
35
+ run a shell command (e.g. a qmd index refresh) after a sync that changed Markdown.
36
+ It receives TEEMS_TRANSCRIPTS_CHANGED (newline-separated Markdown paths),
37
+ TEEMS_TRANSCRIPTS_CHANGED_COUNT, TEEMS_TRANSCRIPTS_MARKDOWN_DIR, and
38
+ TEEMS_TRANSCRIPTS_DIR. "post_sync_timeout" (seconds, default 300) bounds it.
39
+ A failing or timed-out command is reported but does not fail the sync.
40
+ HELP
41
+
42
+ OPTION_HANDLERS = {
43
+ '--since' => ->(opts, args) { opts[:since] = Integer(args.shift, exception: false) },
44
+ '--date' => ->(opts, args) { opts.merge!(Transcripts.date_option(args.shift)) },
45
+ '--dry-run' => ->(opts, _args) { opts[:dry_run] = true },
46
+ '--no-post-sync' => ->(opts, _args) { opts[:no_post_sync] = true }
47
+ }.freeze
48
+
49
+ def self.date_option(value)
50
+ { date: Date.iso8601(value) }
51
+ rescue ArgumentError, TypeError
52
+ { invalid_date: true }
53
+ end
54
+
55
+ def initialize(args, runner:)
56
+ @options = {}
57
+ super
58
+ end
59
+
60
+ def execute
61
+ validation = validate_options
62
+ return validation if validation
63
+ unless positional_args == ['sync']
64
+ return error('Usage: teems transcripts sync [--since DAYS | --date YYYY-MM-DD]')
65
+ end
66
+ return error('--since must be between 1 and 366') unless (1..366).cover?(@options.fetch(:since, 7))
67
+ return error('Invalid --date (expected YYYY-MM-DD)') if @options[:invalid_date]
68
+
69
+ TranscriptSyncEngine.new(@options, output, hook: transcript_settings).run
70
+ end
71
+
72
+ protected
73
+
74
+ def handle_option(arg, pending)
75
+ handler = OPTION_HANDLERS[arg]
76
+ return super unless handler
77
+
78
+ handler.call(@options, pending)
79
+ end
80
+
81
+ def transcript_settings
82
+ settings = config['transcripts']
83
+ settings.is_a?(Hash) ? settings : {}
84
+ end
85
+
86
+ def help_text = HELP
87
+ end
88
+
89
+ # Where transcript sync keeps its files: XDG data for transcripts, XDG state for the manifest
90
+ TranscriptPaths = Data.define(:output_dir, :markdown_dir, :state_dir) do
91
+ def self.from_env
92
+ data_dir = Support::XdgPaths.new.data_dir
93
+ state_home = ENV.fetch('XDG_STATE_HOME') { File.join(Dir.home, '.local', 'state') }
94
+ new(output_dir: File.join(data_dir, 'transcripts'), markdown_dir: File.join(data_dir, 'transcripts-md'),
95
+ state_dir: File.join(state_home, 'teems'))
96
+ end
97
+
98
+ def manifest = File.join(state_dir, 'transcript-sync.json')
99
+
100
+ def lock = File.join(state_dir, 'transcript-sync.lock')
101
+ end
102
+
103
+ # A scheduled Teams meeting on one calendar date (one occurrence of a recurring series)
104
+ TranscriptMeeting = Data.define(:date, :event) do
105
+ def id = event.fetch('id')
106
+
107
+ def name = event['subject'].to_s.gsub(/[\r\n]/, ' ')[0, 100]
108
+
109
+ def status_line(status) = "#{date}: #{status}: #{name}"
110
+
111
+ def command_args = ['meeting', id, '--date', date.iso8601]
112
+
113
+ # Manifest key for one recording's transcript
114
+ def recording_key(url) = Digest::SHA256.hexdigest("#{date.iso8601}:#{id}:#{url}")
115
+
116
+ # Manifest key used before transcripts were tracked per recording
117
+ def legacy_key = Digest::SHA256.hexdigest("#{date.iso8601}:#{id}")
118
+ end
119
+
120
+ # One recording of a meeting; each has its own transcript
121
+ TranscriptRecording = Data.define(:meeting, :url) do
122
+ def key = meeting.recording_key(url)
123
+
124
+ def legacy_key = meeting.legacy_key
125
+
126
+ def date = meeting.date
127
+
128
+ def transcript_args(dir) = [*meeting.command_args, '--transcript', '--recording-url', url, '-o', dir]
129
+
130
+ def manifest_entry(file_name)
131
+ { 'file' => file_name, 'downloaded_at' => Time.now.iso8601, 'date' => date.iso8601, 'subject' => meeting.name }
132
+ end
133
+ end
134
+
135
+ # Private on-disk storage, separate from meeting-capture.
136
+ module TranscriptSyncFiles
137
+ private
138
+
139
+ def prepare_private_directories
140
+ [@paths.output_dir, @paths.state_dir].each { |dir| private_directory(dir) }
141
+ end
142
+
143
+ def private_directory(dir)
144
+ FileUtils.mkdir_p(dir, mode: 0o700)
145
+ File.chmod(0o700, dir)
146
+ end
147
+
148
+ # Transcripts live flat in the output dir; manifest entries are trusted only for their basename
149
+ def transcript_path(file) = File.join(@paths.output_dir, File.basename(file))
150
+
151
+ def log_locked
152
+ log 'Another transcript sync is running; skipping.'
153
+ 0
154
+ end
155
+
156
+ def load_manifest
157
+ path = @paths.manifest
158
+ File.file?(path) ? JSON.parse(File.read(path)) : { 'events' => {} }
159
+ end
160
+
161
+ def downloaded?(key, manifest)
162
+ prior = manifest.fetch('events', {})[key]
163
+ prior && valid_vtt?(transcript_path(prior.fetch('file')))
164
+ end
165
+
166
+ def persist_download(recording, file, manifest)
167
+ key = recording.key
168
+ adopted = adopt_legacy_download(recording, file, manifest)
169
+ target_name = adopted || store_download(file, key)
170
+ manifest['events'][key] = recording.manifest_entry(target_name)
171
+ save_manifest(manifest)
172
+ log "#{recording.date}: #{adopted ? 'already had' : 'saved'} #{target_name}"
173
+ end
174
+
175
+ def store_download(file, key)
176
+ target_name = "#{File.basename(file, '.vtt')}--#{key[0, 10]}.vtt"
177
+ target = transcript_path(target_name)
178
+ File.rename(file, target) unless valid_vtt?(target)
179
+ File.chmod(0o600, target)
180
+ target_name
181
+ end
182
+
183
+ # Earlier syncs kept one transcript per event. Reuse that file when it is
184
+ # this recording's transcript instead of saving a duplicate.
185
+ def adopt_legacy_download(recording, file, manifest)
186
+ events = manifest['events']
187
+ key = recording.legacy_key
188
+ legacy = events[key]
189
+ return unless legacy
190
+
191
+ existing = transcript_path(legacy.fetch('file'))
192
+ return unless valid_vtt?(existing) && FileUtils.identical?(existing, file)
193
+
194
+ events.delete(key)
195
+ File.basename(existing)
196
+ end
197
+
198
+ def valid_vtt?(file)
199
+ File.file?(file) && File.size(file) > 10 && File.open(file, 'rb') { |io| io.read(6) == 'WEBVTT' }
200
+ end
201
+
202
+ # The single valid WebVTT `teems meeting --transcript` wrote into dir, if any
203
+ def downloaded_vtt(dir)
204
+ vtt, *extra = Dir.glob(File.join(dir, '*.vtt'))
205
+ vtt if extra.empty? && valid_vtt?(vtt.to_s)
206
+ end
207
+
208
+ def save_manifest(manifest)
209
+ write_private(@paths.manifest, "#{JSON.pretty_generate(manifest)}\n")
210
+ end
211
+
212
+ def write_private(path, content)
213
+ temp = "#{path}.tmp.#{$PROCESS_ID}"
214
+ File.open(temp, File::WRONLY | File::CREAT | File::TRUNC, 0o600) { |file| file.write(content) }
215
+ File.rename(temp, path)
216
+ File.chmod(0o600, path)
217
+ ensure
218
+ File.delete(temp) if temp && File.exist?(temp)
219
+ end
220
+ end
221
+
222
+ # Speaker-turn Markdown copies of each WebVTT, suitable for a local search index.
223
+ module TranscriptSyncMarkdown
224
+ private
225
+
226
+ # Returns [Markdown paths written this run, whether every export succeeded]
227
+ def sync_markdown(manifest)
228
+ private_directory(@paths.markdown_dir)
229
+ results = manifest.fetch('events', {}).values.map { |entry| export_markdown(entry) }
230
+ written = results.grep(String)
231
+ log "Markdown: updated #{written.length} transcript(s)" unless written.empty?
232
+ [written, !results.include?(:failed)]
233
+ end
234
+
235
+ # Returns the Markdown path it wrote, or :missing, :current, or :failed
236
+ def export_markdown(entry)
237
+ vtt = transcript_path(entry.fetch('file'))
238
+ return :missing unless valid_vtt?(vtt)
239
+
240
+ target = File.join(@paths.markdown_dir, "#{File.basename(vtt, '.vtt')}.md")
241
+ return :current if markdown_current?(target, vtt)
242
+
243
+ write_private(target, markdown_for(vtt, entry))
244
+ target
245
+ rescue StandardError => e
246
+ markdown_failure(entry, "#{e.class}: #{e.message}")
247
+ end
248
+
249
+ def markdown_current?(target, vtt) = File.file?(target) && File.mtime(target) >= File.mtime(vtt)
250
+
251
+ def markdown_failure(entry, reason)
252
+ failure "Markdown export failed for #{File.basename(entry['file'].to_s)} (#{reason})"
253
+ :failed
254
+ end
255
+
256
+ def markdown_for(vtt, entry)
257
+ subject, date = entry.values_at('subject', 'date')
258
+ stem = File.basename(vtt, '.vtt').sub(/--\h{10}\z/, '')
259
+ Formatters::TranscriptMarkdown.new(
260
+ File.read(vtt, encoding: 'bom|utf-8'),
261
+ title: subject || stem.sub(/\A\d{4}-\d{2}-\d{2} - /, '').sub(/-\d{8}_\d{6}UTC\z/, ''),
262
+ date: date || transcript_date(stem),
263
+ source: File.basename(vtt)
264
+ ).render
265
+ end
266
+
267
+ def transcript_date(stem)
268
+ return stem[0, 10] if stem.match?(/\A\d{4}-\d{2}-\d{2} - /)
269
+
270
+ stem.match(/-(\d{4})(\d{2})(\d{2})_\d{6}UTC\z/)&.captures&.join('-')
271
+ end
272
+ end
273
+
274
+ # Calendar discovery is intentionally limited to scheduled Teams meetings.
275
+ module TranscriptSyncCalendar
276
+ private
277
+
278
+ def calendar_events(date)
279
+ stdout, stderr, status = Open3.capture3(teems_executable, 'cal', '--date', date.iso8601, '--json')
280
+ raise "teems cal failed (exit #{status.exitstatus}): #{redacted_error(stderr)}" unless status.success?
281
+
282
+ return [] if stdout.strip == 'No events found'
283
+
284
+ events = JSON.parse(stdout)
285
+ raise 'teems cal returned a non-array response' unless events.is_a?(Array)
286
+
287
+ events.select { |event| teams_event?(event) }
288
+ end
289
+
290
+ def teams_event?(event)
291
+ return false unless event.is_a?(Hash)
292
+ return false if event['is_cancelled'] || event['is_all_day'] || event['response_status'] == 'declined'
293
+
294
+ event['id'].to_s.start_with?('AAMk') &&
295
+ event['online_meeting_url'].to_s.start_with?('https://teams.microsoft.com/')
296
+ end
297
+
298
+ def teems_executable = ENV.fetch('TEEMS_EXECUTABLE', 'teems')
299
+
300
+ def redacted_error(text)
301
+ lines = text.to_s.scrub.lines.map(&:strip).reject { |line| line.empty? || line.include?('warning:') }
302
+ message = lines.find { |line| line.start_with?('Error:') } || lines.first || 'unknown error'
303
+ message.gsub(%r{https?://\S+}, '[URL]').gsub(/Bearer\s+\S+/i, 'Bearer [REDACTED]')[0, 220]
304
+ end
305
+ end
306
+
307
+ # One transcript per recording: a meeting restarted mid-session has several.
308
+ module TranscriptSyncRecordings
309
+ NO_TRANSCRIPT = Regexp.union(/No recording sharing link found/i,
310
+ /No transcripts found for this recording/i,
311
+ /No meeting activity found for/i)
312
+
313
+ private
314
+
315
+ def sync_event(meeting, manifest)
316
+ urls = recording_urls(meeting)
317
+ return unavailable?(meeting) if urls.empty?
318
+ return preview_event?(meeting, urls, manifest) if @options[:dry_run]
319
+
320
+ urls.map { |url| sync_recording?(TranscriptRecording.new(meeting: meeting, url: url), manifest) }.all?
321
+ rescue StandardError => e
322
+ failure "#{meeting.status_line('failed')} (#{e.class}: #{e.message})"
323
+ false
324
+ end
325
+
326
+ def recording_urls(meeting)
327
+ stdout, stderr, status = Open3.capture3(teems_executable, *meeting.command_args, '--json')
328
+ return recordings_from(stdout) if status.success?
329
+
330
+ message = "#{stderr} #{stdout}"
331
+ return [] if NO_TRANSCRIPT.match?(message)
332
+
333
+ raise "teems meeting failed (exit #{status.exitstatus}): #{redacted_error(message)}"
334
+ end
335
+
336
+ # Recording URLs in start-time order, skipping recordings without a sharing link
337
+ def recordings_from(json)
338
+ pairs = JSON.parse(json).fetch('recordings', []).map { |rec| rec.values_at('time', 'url') }
339
+ pairs.select(&:last).sort_by { |time, _url| time.to_s }.map(&:last).uniq
340
+ end
341
+
342
+ def sync_recording?(recording, manifest)
343
+ downloaded?(recording.key, manifest) || download_recording?(recording, manifest)
344
+ end
345
+
346
+ def preview_event?(meeting, urls, manifest)
347
+ pending = urls.count { |url| !downloaded?(meeting.recording_key(url), manifest) }
348
+ pending -= 1 if pending.positive? && downloaded?(meeting.legacy_key, manifest)
349
+ log "#{meeting.status_line('candidate')} (#{urls.length} recording(s), #{pending} not yet saved)"
350
+ true
351
+ end
352
+
353
+ def download_recording?(recording, manifest)
354
+ Dir.mktmpdir('teems-transcript-', @paths.output_dir) do |temp_dir|
355
+ stdout, stderr, status = Open3.capture3(teems_executable, *recording.transcript_args(temp_dir))
356
+ vtt = downloaded_vtt(temp_dir)
357
+ return download_failure?(recording.meeting, status, "#{stderr} #{stdout}") unless status.success? && vtt
358
+
359
+ persist_download(recording, vtt, manifest)
360
+ end
361
+ true
362
+ end
363
+
364
+ def download_failure?(meeting, status, message)
365
+ return unavailable?(meeting) if NO_TRANSCRIPT.match?(message)
366
+
367
+ failure "#{meeting.status_line('failed')} (exit #{status.exitstatus}; #{redacted_error(message)})"
368
+ false
369
+ end
370
+
371
+ def unavailable?(meeting)
372
+ log meeting.status_line('unavailable via teems')
373
+ true
374
+ end
375
+ end
376
+
377
+ # Optional user command run after a sync changes Markdown, e.g. to refresh a qmd index.
378
+ # A failing or slow command is reported but never fails the sync itself.
379
+ class TranscriptPostSyncHook
380
+ DEFAULT_TIMEOUT = 300
381
+
382
+ def initialize(settings, paths:, output:, quiet:)
383
+ @settings = settings
384
+ @paths = paths
385
+ @output = output
386
+ @quiet = quiet
387
+ end
388
+
389
+ def run(changed)
390
+ command = @settings['post_sync_command'].to_s.strip
391
+ return if command.empty? || changed.empty?
392
+
393
+ log "Post-sync: running command for #{changed.length} changed transcript(s)"
394
+ report(*execute(command, changed))
395
+ rescue StandardError => e
396
+ @output.warn("Post-sync command could not run (#{e.class}: #{e.message})")
397
+ end
398
+
399
+ private
400
+
401
+ def log(message)
402
+ @output.puts(message) unless @quiet
403
+ end
404
+
405
+ # Runs in its own process group so a timeout can stop the whole pipeline.
406
+ def execute(command, changed)
407
+ Open3.popen2e(env(changed), 'sh', '-c', command, pgroup: true) do |stdin, stdout, wait|
408
+ stdin.close
409
+ reader = Thread.new { stdout.read }
410
+ finished = wait.join(timeout)
411
+ stop(wait.pid) unless finished
412
+ [finished ? wait.value : nil, reader.value]
413
+ end
414
+ end
415
+
416
+ def stop(pid)
417
+ Process.kill('KILL', -pid)
418
+ rescue Errno::ESRCH
419
+ nil
420
+ end
421
+
422
+ def env(changed)
423
+ { 'TEEMS_TRANSCRIPTS_CHANGED' => changed.join("\n"),
424
+ 'TEEMS_TRANSCRIPTS_CHANGED_COUNT' => changed.length.to_s,
425
+ 'TEEMS_TRANSCRIPTS_MARKDOWN_DIR' => @paths.markdown_dir,
426
+ 'TEEMS_TRANSCRIPTS_DIR' => @paths.output_dir }
427
+ end
428
+
429
+ def timeout = [@settings['post_sync_timeout']].grep(Numeric).find(&:positive?) || DEFAULT_TIMEOUT
430
+
431
+ def report(status, hook_output)
432
+ return log('Post-sync: done') if status&.success?
433
+
434
+ reason = status ? "exit #{status.exitstatus}" : "timed out after #{timeout}s"
435
+ @output.warn("Post-sync command failed (#{reason}): #{tail(hook_output)}")
436
+ end
437
+
438
+ def tail(text)
439
+ lines = text.to_s.scrub.lines.map(&:strip).reject(&:empty?).last(3)
440
+ lines.empty? ? 'no output' : lines.join(' | ')[0, 300]
441
+ end
442
+ end
443
+
444
+ # Local manifest and replay engine. Never transfers transcripts to another machine.
445
+ class TranscriptSyncEngine
446
+ include TranscriptSyncFiles
447
+ include TranscriptSyncCalendar
448
+ include TranscriptSyncMarkdown
449
+ include TranscriptSyncRecordings
450
+
451
+ DEFAULT_LOOKBACK = 7
452
+ INITIAL_LOOKBACK = 30
453
+
454
+ def initialize(options, output, hook: {})
455
+ @options = options
456
+ @output = output
457
+ @paths = TranscriptPaths.from_env
458
+ @hook = TranscriptPostSyncHook.new(hook, paths: @paths, output: output, quiet: options[:quiet])
459
+ end
460
+
461
+ def run
462
+ File.umask(0o077)
463
+ @options[:dry_run] ? scan(load_manifest) : locked_scan
464
+ end
465
+
466
+ private
467
+
468
+ def locked_scan
469
+ prepare_private_directories
470
+ File.open(@paths.lock, File::RDWR | File::CREAT, 0o600) do |lock|
471
+ return log_locked unless lock.flock(File::LOCK_EX | File::LOCK_NB)
472
+
473
+ scan(load_manifest)
474
+ end
475
+ end
476
+
477
+ def log(message)
478
+ @output.puts(message) unless @options[:quiet]
479
+ end
480
+
481
+ def failure(message)
482
+ @output.error(message)
483
+ end
484
+
485
+ def dates_to_scan(manifest)
486
+ date = @options[:date]
487
+ return [date] if date
488
+
489
+ today = Date.today
490
+ days = lookback_days(manifest, today)
491
+ log "Scanning #{days} day(s) through #{today} on this machine"
492
+ ((today - days + 1)..today).to_a
493
+ end
494
+
495
+ def lookback_days(manifest, today)
496
+ days = @options.fetch(:since, DEFAULT_LOOKBACK)
497
+ return days unless days == DEFAULT_LOOKBACK
498
+
499
+ previous = manifest['last_scan_date']
500
+ return INITIAL_LOOKBACK unless previous
501
+
502
+ # Catch up after time away, capped at the requested initial window.
503
+ (today - Date.iso8601(previous) + 1).to_i.clamp(days, INITIAL_LOOKBACK)
504
+ end
505
+
506
+ def scan(manifest)
507
+ results = dates_to_scan(manifest).map { |date| scan_date(date, manifest) }
508
+ results << finish_scan(manifest, results) unless @options[:dry_run]
509
+ results.all? ? 0 : 1
510
+ end
511
+
512
+ def scan_date(date, manifest)
513
+ results = calendar_events(date).map do |event|
514
+ sync_event(TranscriptMeeting.new(date: date, event: event), manifest)
515
+ end
516
+ results.all?
517
+ rescue StandardError => e
518
+ failure "#{date}: calendar error (#{e.class}: #{e.message})"
519
+ nil
520
+ end
521
+
522
+ # Records the scan, refreshes Markdown, and runs the post-sync hook; returns whether Markdown export succeeded.
523
+ # Failed artifacts stay in the log and get seven-day retries. A calendar failure (nil result)
524
+ # leaves the wider scan pending so historical days aren't lost.
525
+ def finish_scan(manifest, results)
526
+ manifest['last_scan_date'] = Date.today.iso8601 unless @options[:date] || results.include?(nil)
527
+ save_manifest(manifest)
528
+ changed, markdown_ok = sync_markdown(manifest)
529
+ @hook.run(changed) unless @options[:no_post_sync]
530
+ markdown_ok
531
+ end
532
+ end
533
+ end
534
+ end
@@ -26,6 +26,9 @@ module Teems
26
26
  att.is_a?(Hash) ? (att['fileName'] || att['name'] || 'file') : att.to_s
27
27
  end
28
28
 
29
+ # One-line description of a message's inline images, e.g. "image (881x179), chart"
30
+ def image_summary(images) = images.map(&:label).join(', ')
31
+
29
32
  def format_single_reaction(reaction, emoji_map)
30
33
  type = reaction[:type]
31
34
  emoji = emoji_map[type] || type
@@ -15,10 +15,12 @@ module Teems
15
15
  '1f440_eyes' => "\u{1F440}", 'thumbsdown' => "\u{1F44E}"
16
16
  }.freeze
17
17
 
18
- def initialize(chat_name:, chat_type: nil, synced_at: nil)
18
+ # image_link: optional callable returning a relative path for a locally saved InlineImage
19
+ def initialize(chat_name:, chat_type: nil, synced_at: nil, image_link: nil)
19
20
  @chat_name = chat_name
20
21
  @chat_type = chat_type
21
22
  @synced_at = synced_at
23
+ @image_link = image_link
22
24
  end
23
25
 
24
26
  # Format an array of Message objects into a Markdown string.
@@ -68,6 +70,7 @@ module Teems
68
70
  content = msg.content
69
71
  result = content.to_s.empty? ? [] : [content]
70
72
  result.concat(format_message_attachments(msg))
73
+ result.concat(msg.images.map { |image| format_image(image) })
71
74
  result.concat(format_message_reactions(msg))
72
75
  end
73
76
 
@@ -93,6 +96,12 @@ module Teems
93
96
  url&.start_with?('https://') ? "\u{1F4CE} [#{name}](#{url})" : "\u{1F4CE} #{name}"
94
97
  end
95
98
 
99
+ def format_image(image)
100
+ label = "image: #{image.label}"
101
+ path = @image_link&.call(image)
102
+ path ? "![#{label}](#{path})" : "[#{label}]"
103
+ end
104
+
96
105
  def format_message_reactions(msg)
97
106
  reactions = msg.reactions
98
107
  return [] unless displayable_reactions?(reactions)
@@ -21,8 +21,8 @@ module Teems
21
21
 
22
22
  def format(message)
23
23
  content = highlight_mentions(message.content, message.mentions)
24
- [format_header(message), " #{content}", format_attachments(message), format_reactions(message)]
25
- .compact.join("\n")
24
+ [format_header(message), " #{content}", format_attachments(message), format_images(message),
25
+ format_reactions(message)].compact.join("\n")
26
26
  end
27
27
 
28
28
  private
@@ -44,6 +44,11 @@ module Teems
44
44
  " #{@output.gray("\u{1F4CE} #{attachment_names(attachments)}")}"
45
45
  end
46
46
 
47
+ def format_images(message)
48
+ summary = FormatUtils.image_summary(message.images)
49
+ " #{@output.gray("\u{1F5BC}\u{FE0F} #{summary}")}" unless summary.empty?
50
+ end
51
+
47
52
  def format_reactions(message)
48
53
  reactions = message.reactions
49
54
  return unless reactions.any?
@@ -0,0 +1,97 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+
5
+ module Teems
6
+ module Formatters
7
+ # Converts Teams WebVTT transcripts into speaker-turn Markdown for local search indexes.
8
+ # Consecutive cues from the same speaker are merged so each paragraph carries context.
9
+ class TranscriptMarkdown
10
+ # One timed WebVTT cue: its start timestamp, speaker (nil when unattributed), and plain text
11
+ Cue = Data.define(:start, :speaker, :text)
12
+
13
+ # Consecutive cues from one speaker, merged into a paragraph until it is too long to index well
14
+ Turn = Struct.new(:start, :speaker, :text) do
15
+ def self.from(cue) = new(cue.start, cue.speaker, +cue.text)
16
+
17
+ def accepts?(cue) = cue.speaker == speaker && text.length < MAX_TURN_CHARS
18
+
19
+ def add(cue) = text << ' ' << cue.text
20
+ end
21
+
22
+ TIMING = /\A(?<start>(?:\d+:)?\d{2}:\d{2}\.\d{3})\s+-->/
23
+ VOICE = /<v(?:\.[^\s>]+)?\s+([^>]+)>/
24
+ ENTITIES = {
25
+ '&amp;' => '&', '&lt;' => '<', '&gt;' => '>', '&quot;' => '"', '&#39;' => "'",
26
+ '&apos;' => "'", '&nbsp;' => ' ', '&lrm;' => '', '&rlm;' => ''
27
+ }.freeze
28
+ ENTITY_PATTERN = Regexp.union(ENTITIES.keys)
29
+ MAX_TURN_CHARS = 1500
30
+
31
+ def initialize(vtt, title:, date: nil, source: nil)
32
+ @vtt = vtt.to_s
33
+ @title = single_line(title).then { |value| value.empty? ? 'Teams transcript' : value }
34
+ @date = date
35
+ @source = source
36
+ end
37
+
38
+ def render
39
+ lines = front_matter + ["# #{@title}", '', ['Teams transcript', @date].compact.join(' · '), '']
40
+ turns.each { |turn| lines.push(format_turn(turn), '') }
41
+ "#{lines.join("\n").rstrip}\n"
42
+ end
43
+
44
+ def turns
45
+ cues.each_with_object([]) { |cue, turns| append_cue(turns, cue) }
46
+ end
47
+
48
+ private
49
+
50
+ def append_cue(turns, cue)
51
+ last = turns.last
52
+ last&.accepts?(cue) ? last.add(cue) : turns << Turn.from(cue)
53
+ end
54
+
55
+ def cues
56
+ normalized = @vtt.scrub.delete_prefix("\uFEFF").gsub(/\r\n?/, "\n")
57
+ normalized.split(/\n{2,}/).filter_map { |block| parse_cue(block) }
58
+ end
59
+
60
+ # Cue text is every line after the timing line; identifier lines before it are skipped
61
+ def parse_cue(block)
62
+ timing, *text = block.lines(chomp: true).drop_while { |line| !TIMING.match?(line) }
63
+ build_cue(timing[TIMING, :start], text.join(' ')) if timing
64
+ end
65
+
66
+ def build_cue(start, raw)
67
+ text = clean(raw)
68
+ return if text.empty?
69
+
70
+ Cue.new(start: start, speaker: raw[VOICE, 1]&.then { |name| clean(name) }, text: text)
71
+ end
72
+
73
+ def clean(text) = single_line(decode(text.gsub(/<[^>]*>/, '')))
74
+
75
+ def decode(text) = text.gsub(ENTITY_PATTERN, ENTITIES)
76
+
77
+ def single_line(text) = text.to_s.gsub(/\s+/, ' ').strip
78
+
79
+ def front_matter
80
+ [
81
+ '---',
82
+ "title: #{@title.to_json}",
83
+ ("date: #{@date}" if @date),
84
+ 'source: teams-transcript',
85
+ ("vtt: #{@source.to_json}" if @source),
86
+ '---',
87
+ ''
88
+ ].compact
89
+ end
90
+
91
+ def format_turn(turn)
92
+ speaker = turn.speaker || 'Unknown speaker'
93
+ "**#{speaker}** (#{turn.start.sub(/\.\d+\z/, '')}): #{turn.text}"
94
+ end
95
+ end
96
+ end
97
+ end