teems 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,331 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Teems
4
+ module Commands
5
+ # Parses DASH MPD manifest XML to extract video/audio segment info
6
+ class DashManifestParser
7
+ # Describes a media track (video or audio) with its segment template info
8
+ Track = Data.define(:type, :init_url, :media_template, :segment_count, :timescale, :segments)
9
+
10
+ # A single media segment with start time and duration in timescale units
11
+ Segment = Data.define(:start, :duration)
12
+
13
+ ADAPTATION_RE = %r{<AdaptationSet[^>]*contentType="(video|audio)"(.*?)</AdaptationSet>}m
14
+ TEMPLATE_RE = /<SegmentTemplate[^>]*/
15
+ TIMELINE_RE = %r{<SegmentTimeline>(.*?)</SegmentTimeline>}m
16
+ SEGMENT_RE = /<S\s[^>]*>/
17
+ REP_RE = /<Representation[^>]*id="([^"]+)"/
18
+
19
+ def initialize(mpd_xml)
20
+ @xml = mpd_xml
21
+ @attrs = nil
22
+ @rep_id = nil
23
+ end
24
+
25
+ def parse
26
+ tracks = []
27
+ @xml.scan(ADAPTATION_RE) do |type, body|
28
+ track = parse_adaptation(type, body)
29
+ tracks << track if track
30
+ end
31
+ tracks
32
+ end
33
+
34
+ private
35
+
36
+ def parse_adaptation(type, body)
37
+ template_match = body.match(TEMPLATE_RE)
38
+ return nil unless template_match
39
+
40
+ @attrs = template_match[0]
41
+ @rep_id = body.match(REP_RE)&.then { _1[1] } || ''
42
+ segments = parse_timeline(body)
43
+ return nil if segments.empty?
44
+
45
+ build_track(type, segments)
46
+ end
47
+
48
+ def build_track(type, segments)
49
+ timescale_val = extract_attr(@attrs, 'timescale').to_i
50
+ Track.new(type: type,
51
+ init_url: decode_template(extract_attr(@attrs, 'initialization')),
52
+ media_template: decode_template(extract_attr(@attrs, 'media')),
53
+ segment_count: segments.length,
54
+ timescale: timescale_val.positive? ? timescale_val : 1,
55
+ segments: segments)
56
+ end
57
+
58
+ def decode_template(url)
59
+ return '' unless url
60
+
61
+ url.gsub('&amp;', '&').gsub('$RepresentationID$', @rep_id.to_s)
62
+ end
63
+
64
+ def parse_timeline(body)
65
+ timeline_match = body.match(TIMELINE_RE)
66
+ return [] unless timeline_match
67
+
68
+ segments = []
69
+ time = 0
70
+ timeline_match[1].scan(SEGMENT_RE) do |seg_tag|
71
+ time = parse_segment_tag(seg_tag, time, segments)
72
+ end
73
+ segments
74
+ end
75
+
76
+ def parse_segment_tag(seg_tag, time, segments)
77
+ start_time = extract_attr(seg_tag, 't').to_i
78
+ duration = extract_attr(seg_tag, 'd').to_i
79
+ repeat_count = extract_attr(seg_tag, 'r').to_i
80
+
81
+ time = start_time if start_time.positive?
82
+ (repeat_count + 1).times do
83
+ segments << Segment.new(start: time, duration: duration)
84
+ time += duration
85
+ end
86
+ time
87
+ end
88
+
89
+ def extract_attr(tag, name)
90
+ match = tag.match(/#{name}="([^"]*)"/)
91
+ match ? match[1] : nil
92
+ end
93
+ end
94
+
95
+ # Downloads DASH segments in parallel and assembles track data
96
+ module SegmentDownloader
97
+ PARALLEL_DOWNLOADS = 5
98
+
99
+ private
100
+
101
+ def download_track_segments(track, base_url)
102
+ urls = track.segments.map { |seg| resolve_url(base_url, segment_path(track, seg)) }
103
+ init = fetch_segment(resolve_url(base_url, track.init_url))
104
+ results = parallel_fetch(urls, track.segment_count)
105
+ String.new(init).tap { |data| results.each { |seg| data << seg } }
106
+ end
107
+
108
+ def segment_path(track, seg)
109
+ track.media_template.gsub('$Time$', seg.start.to_s)
110
+ end
111
+
112
+ def parallel_fetch(urls, total)
113
+ init_seg_state(urls.length, total)
114
+ urls.each_with_index { |url, idx| @seg_state[:queue] << [url, idx] }
115
+ PARALLEL_DOWNLOADS.times.map { Thread.new { fetch_from_queue } }.each(&:join)
116
+ check_seg_errors
117
+ @seg_state[:results]
118
+ end
119
+
120
+ def init_seg_state(count, total)
121
+ @seg_state = { results: Array.new(count), errors: [],
122
+ queue: Queue.new, total: total, mutex: Mutex.new }
123
+ end
124
+
125
+ def check_seg_errors
126
+ first_error = @seg_state[:errors].first
127
+ raise first_error if first_error
128
+ end
129
+
130
+ def fetch_from_queue
131
+ loop do
132
+ item = @seg_state[:queue].pop(true)
133
+ @seg_state[:results][item[1]] = fetch_segment(item[0])
134
+ record_progress
135
+ rescue ThreadError
136
+ break
137
+ rescue StandardError => e
138
+ @seg_state[:mutex].synchronize { @seg_state[:errors] << e }
139
+ break
140
+ end
141
+ end
142
+
143
+ def record_progress
144
+ @seg_state[:mutex].synchronize do
145
+ print "\r Downloading: #{@seg_state[:results].count { _1 }}/#{@seg_state[:total]} segments"
146
+ end
147
+ end
148
+
149
+ def resolve_url(base_url, path)
150
+ return path if path.start_with?('http')
151
+
152
+ "#{base_url.sub(%r{/[^/]*$}, '/')}#{path}"
153
+ end
154
+
155
+ def fetch_segment(url)
156
+ Net::HTTP.get_response(URI(url)).tap do |resp|
157
+ raise Teems::Error, "Segment download failed (#{resp.code})" unless resp.is_a?(Net::HTTPSuccess)
158
+ end.body
159
+ end
160
+ end
161
+
162
+ # Resolves recording file info by fetching embed page HTML directly
163
+ module RecordingResolver
164
+ include EmbedPageParser
165
+
166
+ private
167
+
168
+ def resolve_recording_file_info(sharing_url)
169
+ embed_url = fetch_embed_url(sharing_url)
170
+ return error('Could not get embed URL for recording') && nil unless embed_url
171
+
172
+ file_info = fetch_and_parse_embed(embed_url)
173
+ return error('Could not extract file info from embed page') && nil unless file_info&.dig(:transform_url)
174
+
175
+ { transform_url: file_info[:transform_url], name: file_info[:name] || 'recording.mp4' }
176
+ end
177
+ end
178
+
179
+ # Writes video/audio output files from downloaded DASH tracks via ffmpeg
180
+ module RecordingOutputWriter
181
+ private
182
+
183
+ def produce_outputs(video_path, audio_path, media)
184
+ video_result = media[:video] ? write_video_output(video_path, audio_path) : true
185
+ audio_result = media[:audio] ? write_audio_output(audio_path) : true
186
+ FileUtils.rm_f([video_path, audio_path].compact)
187
+ video_result && audio_result ? 0 : 1
188
+ end
189
+
190
+ def write_video_output(video_path, audio_path)
191
+ output_path = File.join(@rec_dir, "#{@recording_stem}.mp4")
192
+ info('Merging video and audio tracks...')
193
+ merged = run_ffmpeg('-i', video_path, '-i', audio_path, '-c', 'copy', output_path)
194
+ return error('ffmpeg merge failed') && false unless merged
195
+
196
+ embed_subtitle_if_available(output_path)
197
+ success("Recording saved to #{output_path}")
198
+ true
199
+ end
200
+
201
+ def write_audio_output(audio_path)
202
+ output_path = File.join(@rec_dir, "#{@recording_stem}.m4a")
203
+ info('Remuxing audio track...')
204
+ remuxed = run_ffmpeg('-i', audio_path, '-c', 'copy', output_path)
205
+ return error('ffmpeg audio remux failed') && false unless remuxed
206
+
207
+ success("Audio saved to #{output_path}")
208
+ true
209
+ end
210
+
211
+ def embed_subtitle_if_available(video_path)
212
+ vtt_path = paired_vtt_path(video_path)
213
+ return unless vtt_path
214
+
215
+ info('Embedding transcript as subtitle track...')
216
+ final_path = video_path.sub('.mp4', '_subs.mp4')
217
+ return unless run_ffmpeg('-i', video_path, '-i', vtt_path, '-c', 'copy', '-c:s', 'mov_text', final_path)
218
+
219
+ FileUtils.mv(final_path, video_path)
220
+ end
221
+
222
+ def paired_vtt_path(video_path)
223
+ paired = video_path.sub(/\.mp4\z/, '.vtt')
224
+ return paired if File.exist?(paired)
225
+
226
+ Dir.glob(File.join(File.dirname(video_path), '*.vtt')).first
227
+ end
228
+ end
229
+
230
+ # Downloads meeting recordings via DASH streaming from SharePoint
231
+ module MeetingRecording
232
+ include MeetingFilename
233
+ include RecordingResolver
234
+ include SegmentDownloader
235
+ include RecordingOutputWriter
236
+
237
+ MANIFEST_PARAMS = 'format=dash&part=index'
238
+
239
+ private
240
+
241
+ def download_recording(target, classified, media: { video: true, audio: false })
242
+ sharing_url = target[:fileUrl] || classified[:recordings].filter_map { |rec| rec[:url] }.first
243
+ return error('No recordings found for this meeting') unless sharing_url
244
+ return error('ffmpeg is required. Install with: brew install ffmpeg') unless ffmpeg?
245
+
246
+ execute_recording_pipeline(sharing_url, media)
247
+ end
248
+
249
+ def execute_recording_pipeline(sharing_url, media)
250
+ info("Fetching #{media[:video] ? 'recording' : 'audio'} via SharePoint...")
251
+ file_info = resolve_recording_file_info(sharing_url)
252
+ return 1 unless file_info
253
+
254
+ @recording_stem = derive_output_stem(file_info[:name])
255
+ manifest = fetch_dash_manifest(file_info[:transform_url])
256
+ return error('Could not fetch DASH manifest') unless manifest
257
+
258
+ download_and_assemble(manifest, media)
259
+ end
260
+
261
+ def fetch_dash_manifest(transform_url)
262
+ url = build_manifest_url(transform_url)
263
+ debug("Fetching DASH manifest: #{url}")
264
+ fetch_manifest_content(url)
265
+ end
266
+
267
+ def build_manifest_url(transform_url)
268
+ url = transform_url.sub('/thumbnail', '/videomanifest')
269
+ separator = url.include?('?') ? '&' : '?'
270
+ "#{url}#{separator}#{MANIFEST_PARAMS}"
271
+ end
272
+
273
+ def fetch_manifest_content(url)
274
+ result = Net::HTTP.get_response(URI(url))
275
+ return result.body if result.is_a?(Net::HTTPSuccess)
276
+
277
+ debug("Manifest download failed: HTTP #{result.code}")
278
+ nil
279
+ rescue IOError, SystemCallError, SocketError, Timeout::Error, OpenSSL::SSL::SSLError => e
280
+ debug("Manifest download error: #{e.message}")
281
+ nil
282
+ end
283
+
284
+ def download_and_assemble(manifest_xml, media)
285
+ tracks = DashManifestParser.new(manifest_xml).parse
286
+ selected = select_tracks(tracks, media)
287
+ return error('No video/audio tracks found in manifest') unless selected
288
+
289
+ assemble_recording(selected, manifest_xml, media)
290
+ end
291
+
292
+ def select_tracks(tracks, media)
293
+ grouped = tracks.group_by(&:type)
294
+ audio = grouped['audio']&.first
295
+ return nil unless audio
296
+
297
+ video = grouped['video']&.first
298
+ return nil if media[:video] && !video
299
+
300
+ { video: video, audio: audio }
301
+ end
302
+
303
+ def assemble_recording(selected, manifest_xml, media)
304
+ @rec_dir = @options[:output_dir] || '.'
305
+ FileUtils.mkdir_p(@rec_dir)
306
+ @rec_base_url = manifest_xml.match(%r{<BaseURL>([^<]+)</BaseURL>})&.then { _1[1] } || ''
307
+
308
+ audio_path = write_track(selected[:audio], 'audio')
309
+ video_path = media[:video] ? write_track(selected[:video], 'video') : nil
310
+ produce_outputs(video_path, audio_path, media)
311
+ end
312
+
313
+ def write_track(track, label)
314
+ info("Downloading #{label} track (#{track.segment_count} segments)...")
315
+ data = download_track_segments(track, @rec_base_url)
316
+ path = File.join(@rec_dir, ".teems_#{label}.mp4")
317
+ File.binwrite(path, data)
318
+ puts
319
+ path
320
+ end
321
+
322
+ def run_ffmpeg(*)
323
+ system('ffmpeg', '-y', *, out: File::NULL, err: File::NULL)
324
+ end
325
+
326
+ def ffmpeg?
327
+ system('which', 'ffmpeg', out: File::NULL, err: File::NULL)
328
+ end
329
+ end
330
+ end
331
+ end
@@ -0,0 +1,270 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Teems
4
+ module Commands
5
+ # Derives a human-friendly output filename stem from a SharePoint recording/transcript name.
6
+ # SharePoint format is typically "<subject>-YYYYMMDD_HHMMSS[-Meeting Recording].<ext>"
7
+ # which we reshape into "YYYY-MM-DD - <subject>".
8
+ module MeetingFilename
9
+ SHAREPOINT_NAME_RE = /\A(.+)-(\d{4})(\d{2})(\d{2})_\d{6}\z/
10
+ UNSAFE_FILENAME_CHARS = %r{[/\\:*?"<>|\x00]}
11
+
12
+ private
13
+
14
+ def derive_output_stem(sharepoint_name)
15
+ base = File.basename(sharepoint_name.to_s, '.*').sub(/-Meeting Recording\z/, '')
16
+ sanitize_filename(reshape_sharepoint_name(base))
17
+ end
18
+
19
+ def reshape_sharepoint_name(base)
20
+ captures = base.match(SHAREPOINT_NAME_RE)&.captures
21
+ return base unless captures
22
+
23
+ subject, year, month, day = captures
24
+ "#{year}-#{month}-#{day} - #{subject}"
25
+ end
26
+
27
+ def sanitize_filename(name)
28
+ cleaned = name.gsub(UNSAFE_FILENAME_CHARS, '_').strip
29
+ cleaned.empty? ? 'recording' : cleaned
30
+ end
31
+ end
32
+
33
+ # Parses SharePoint embed page HTML to extract file metadata
34
+ module EmbedPageParser
35
+ SP_ITEM_RE = %r{(https://[^/]+(?::\d+)?/.+?)/_api/v2\.\d+/drives/([^/]+)/items/([^/?]+)}
36
+ FILE_INFO_RE = /g_fileInfo\s*=\s*(\{.*?\});/m
37
+
38
+ private
39
+
40
+ def fetch_embed_url(sharing_url)
41
+ debug("Resolving sharing link: #{sharing_url}")
42
+ result = with_token_refresh { runner.meetings_api.share_preview(sharing_url) }
43
+ result['getUrl']
44
+ rescue ApiError => e
45
+ debug("Share preview failed: #{e.message}")
46
+ nil
47
+ end
48
+
49
+ def fetch_and_parse_embed(embed_url)
50
+ debug("Fetching embed page: #{embed_url}")
51
+ html = fetch_embed_page(embed_url)
52
+ return nil unless html
53
+
54
+ parse_embed_file_info(html)
55
+ end
56
+
57
+ def fetch_embed_page(url, limit: 3)
58
+ response = Net::HTTP.get_response(URI(url))
59
+ return response.body if response.is_a?(Net::HTTPSuccess)
60
+ return debug("Embed page failed: HTTP #{response.code}") unless response.is_a?(Net::HTTPRedirection)
61
+ return debug('Too many redirects fetching embed page') unless limit.positive?
62
+
63
+ fetch_embed_page(URI.join(url, response['location']).to_s, limit: limit - 1)
64
+ rescue IOError, SystemCallError, SocketError, Timeout::Error, URI::InvalidURIError, OpenSSL::SSL::SSLError => e
65
+ debug("Embed page fetch error: #{e.message}")
66
+ nil
67
+ end
68
+
69
+ def parse_embed_file_info(html)
70
+ match = html.match(FILE_INFO_RE)
71
+ unless match
72
+ debug('g_fileInfo not found in embed page HTML')
73
+ return nil
74
+ end
75
+
76
+ data = JSON.parse(match[1])
77
+ build_file_info(data)
78
+ rescue JSON::ParserError => e
79
+ debug("Embed page JSON parse error: #{e.message}")
80
+ nil
81
+ end
82
+
83
+ def build_file_info(data)
84
+ extras = extract_extras(data)
85
+ extract_ids(data)&.merge(extras)
86
+ end
87
+
88
+ def extract_extras(data)
89
+ { drive_token: data['.driveAccessTokenV21'],
90
+ transform_url: data['.transformUrl'] || data['transformUrl'],
91
+ name: data['name'] }
92
+ end
93
+
94
+ def extract_ids(data)
95
+ build_from_sp_item_url(data['.spItemUrl']) || build_from_fields(data)
96
+ end
97
+
98
+ def build_from_sp_item_url(sp_url)
99
+ return nil unless sp_url
100
+
101
+ match = sp_url.match(SP_ITEM_RE)
102
+ match ? { site_url: match[1], drive_id: match[2], item_id: match[3] } : nil
103
+ end
104
+
105
+ def build_from_fields(data)
106
+ drive_id = data.dig('libraryId', 'siteId') || data['driveId']
107
+ item_id = data['itemId'] || data['id']
108
+ site_url = data['siteUrl'] || data.dig('libraryId', 'siteUrl')
109
+ [drive_id, item_id, site_url].all? ? { drive_id: drive_id, item_id: item_id, site_url: site_url } : nil
110
+ end
111
+ end
112
+
113
+ # Converts JSON transcript entries to WebVTT with speaker names
114
+ class TranscriptFormatter
115
+ def initialize(entries)
116
+ @entries = entries
117
+ @cue = nil
118
+ end
119
+
120
+ def to_vtt
121
+ cues = @entries.each_with_index.map { |entry, idx| format_cue(entry, idx) }
122
+ "WEBVTT\n\n#{cues.join}"
123
+ end
124
+
125
+ private
126
+
127
+ def format_cue(entry, idx)
128
+ @cue = entry
129
+ "#{idx + 1}\n#{cue_timestamps}\n<v #{@cue['speakerDisplayName']}>#{@cue['text']}</v>\n\n"
130
+ end
131
+
132
+ def cue_timestamps
133
+ "#{truncate_ts(@cue['startOffset'])} --> #{truncate_ts(@cue['endOffset'])}"
134
+ end
135
+
136
+ def truncate_ts(offset)
137
+ offset ? offset.to_s[0, 12] : '00:00:00.000'
138
+ end
139
+ end
140
+
141
+ # Downloads meeting transcripts via SharePoint API (no Safari required)
142
+ module MeetingTranscript
143
+ include EmbedPageParser
144
+ include MeetingFilename
145
+
146
+ private
147
+
148
+ def download_transcript(target, classified)
149
+ sharing_url = target[:fileUrl] || first_recording_url(classified)
150
+ return error('No recording sharing link found for transcript download') unless sharing_url
151
+
152
+ execute_transcript_pipeline(sharing_url)
153
+ end
154
+
155
+ def first_recording_url(classified)
156
+ classified[:recordings].filter_map { |rec| rec[:url] }.first
157
+ end
158
+
159
+ def execute_transcript_pipeline(sharing_url)
160
+ info('Fetching transcript via SharePoint...')
161
+ embed_url = fetch_embed_url(sharing_url)
162
+ return error('Could not get embed URL from sharing link') unless embed_url
163
+
164
+ @transcript_info = fetch_and_parse_embed(embed_url)
165
+ return error('Could not extract file info from embed page') unless @transcript_info
166
+
167
+ transcript = resolve_transcript
168
+ return error('No transcripts found for this recording') unless transcript
169
+
170
+ save_transcript(transcript)
171
+ end
172
+
173
+ def resolve_transcript
174
+ url = build_transcripts_url
175
+ debug("Fetching transcripts from: #{url}")
176
+ fetch_transcript_list(url)
177
+ end
178
+
179
+ def build_transcripts_url
180
+ "#{@transcript_info[:site_url]}/_api/v2.1" \
181
+ "/drives/#{@transcript_info[:drive_id]}" \
182
+ "/items/#{@transcript_info[:item_id]}/media/transcripts"
183
+ end
184
+
185
+ def fetch_transcript_list(url)
186
+ response = fetch_with_drive_token(url)
187
+ return nil unless response
188
+
189
+ parse_transcript_response(response)
190
+ rescue JSON::ParserError => e
191
+ debug("Transcript list JSON parse error: #{e.message}")
192
+ nil
193
+ end
194
+
195
+ def fetch_with_drive_token(url)
196
+ result = Net::HTTP.get_response(URI(url), drive_token_headers)
197
+ return result.body if result.is_a?(Net::HTTPSuccess)
198
+
199
+ debug("Transcript list request failed: HTTP #{result.code}")
200
+ nil
201
+ rescue IOError, SystemCallError, SocketError, Timeout::Error, OpenSSL::SSL::SSLError => e
202
+ debug("Transcript list fetch error: #{e.message}")
203
+ nil
204
+ end
205
+
206
+ def drive_token_headers
207
+ token = @transcript_info[:drive_token]
208
+ debug('No drive token; request will be unauthenticated') unless token
209
+ hdrs = { 'Accept' => 'application/json' }
210
+ hdrs['Authorization'] = "Bearer #{token}" if token
211
+ hdrs
212
+ end
213
+
214
+ def parse_transcript_response(body)
215
+ parsed = JSON.parse(body)
216
+ api_error = parsed['error']
217
+ return debug("SharePoint API error: #{api_error}") if api_error
218
+
219
+ extract_transcript_entry(parsed)
220
+ end
221
+
222
+ def extract_transcript_entry(parsed)
223
+ entry = (parsed['value'] || [parsed]).first
224
+ url = entry&.dig('temporaryDownloadUrl') || entry&.dig('downloadUrl')
225
+ url ? { url: url, name: File.basename(entry['name'] || 'transcript.vtt') } : nil
226
+ end
227
+
228
+ # Prefers the recording's name (via @transcript_info) so transcript and recording share a stem.
229
+ def save_transcript(transcript)
230
+ dir = @options[:output_dir] || '.'
231
+ FileUtils.mkdir_p(dir)
232
+ path = File.join(dir, "#{derive_output_stem(@transcript_info[:name] || transcript[:name])}.vtt")
233
+ info("Downloading transcript to #{path}...")
234
+ vtt = fetch_and_convert_transcript(transcript[:url])
235
+ return error('Failed to download transcript content') unless vtt
236
+
237
+ File.write(path, vtt)
238
+ success("Transcript saved to #{path}")
239
+ rescue SystemCallError => e
240
+ error("Could not save transcript: #{e.message}")
241
+ end
242
+
243
+ def fetch_and_convert_transcript(url)
244
+ json_url = url.include?('?') ? "#{url}&format=json" : "#{url}?format=json"
245
+ json_content = parse_transcript_json(fetch_transcript_content(json_url))
246
+ json_content ? TranscriptFormatter.new(json_content['entries']).to_vtt : fetch_transcript_content(url)
247
+ end
248
+
249
+ def parse_transcript_json(raw)
250
+ return nil unless raw
251
+
252
+ data = JSON.parse(raw)
253
+ data['entries'] ? data : nil
254
+ rescue JSON::ParserError
255
+ nil
256
+ end
257
+
258
+ def fetch_transcript_content(url)
259
+ response = Net::HTTP.get_response(URI(url))
260
+ return response.body if response.is_a?(Net::HTTPSuccess)
261
+
262
+ debug("Transcript download failed: HTTP #{response.code}")
263
+ nil
264
+ rescue IOError, SystemCallError, SocketError, Timeout::Error, OpenSSL::SSL::SSLError => e
265
+ debug("Transcript download error: #{e.message}")
266
+ nil
267
+ end
268
+ end
269
+ end
270
+ end