sys2txt 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sys2txt/__init__.py +1 -0
- sys2txt/__main__.py +172 -0
- sys2txt/audio.py +169 -0
- sys2txt/constants.py +4 -0
- sys2txt/pulse.py +48 -0
- sys2txt/transcribe.py +82 -0
- sys2txt/utils.py +11 -0
- sys2txt-0.1.1.dist-info/METADATA +167 -0
- sys2txt-0.1.1.dist-info/RECORD +13 -0
- sys2txt-0.1.1.dist-info/WHEEL +5 -0
- sys2txt-0.1.1.dist-info/entry_points.txt +2 -0
- sys2txt-0.1.1.dist-info/licenses/LICENSE +21 -0
- sys2txt-0.1.1.dist-info/top_level.txt +1 -0
sys2txt/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
sys2txt/__main__.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Main entry point for sys2txt CLI."""
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import os
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
|
|
10
|
+
from .audio import record_once, segment_and_transcribe_live
|
|
11
|
+
from .constants import WHISPER_MODEL
|
|
12
|
+
from .pulse import get_default_monitor_source, list_pulse_sources
|
|
13
|
+
from .transcribe import transcribe_file
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def get_timestamp_filename() -> str:
|
|
17
|
+
"""Generate a timestamp-based filename for output files.
|
|
18
|
+
|
|
19
|
+
Returns:
|
|
20
|
+
A filename string in the format: YYYY-MM-DD_HH-MM-SS.txt
|
|
21
|
+
"""
|
|
22
|
+
return datetime.now().strftime("%Y-%m-%d_%H-%M-%S.txt")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def ensure_output_dir() -> str:
|
|
26
|
+
"""Ensure the output directory exists and return its path.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
Absolute path to the output directory
|
|
30
|
+
"""
|
|
31
|
+
output_dir = os.path.join(os.getcwd(), "output")
|
|
32
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
33
|
+
return output_dir
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def main():
|
|
37
|
+
"""Main CLI entry point."""
|
|
38
|
+
parser = argparse.ArgumentParser(description="Record Ubuntu system audio and transcribe with Whisper.")
|
|
39
|
+
sub = parser.add_subparsers(dest="mode", required=True)
|
|
40
|
+
|
|
41
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
42
|
+
common.add_argument(
|
|
43
|
+
"--source", help="PulseAudio source name (e.g., <sink>.monitor). Defaults to auto.", default=None
|
|
44
|
+
)
|
|
45
|
+
common.add_argument(
|
|
46
|
+
"--model",
|
|
47
|
+
dest="model_size",
|
|
48
|
+
default=WHISPER_MODEL,
|
|
49
|
+
help=f"Whisper model size (default: {WHISPER_MODEL})",
|
|
50
|
+
)
|
|
51
|
+
common.add_argument(
|
|
52
|
+
"--engine", choices=["auto", "faster", "whisper"], default="auto", help="Transcription engine (default: auto)"
|
|
53
|
+
)
|
|
54
|
+
common.add_argument("--language", default=None, help="Force language code (e.g., en). Defaults to auto-detect")
|
|
55
|
+
common.add_argument("--timestamps", action="store_true", help="Print timestamps with transcript")
|
|
56
|
+
common.add_argument("--list-sources", action="store_true", help="List PulseAudio sources and exit")
|
|
57
|
+
|
|
58
|
+
once = sub.add_parser("once", parents=[common], help="Record once and transcribe after")
|
|
59
|
+
once.add_argument("--duration", type=int, default=None, help="Record for N seconds instead of Ctrl-C")
|
|
60
|
+
once.add_argument("--output", default=None, help="Write transcript to file")
|
|
61
|
+
once.add_argument("--input", default=None, help="Skip recording and transcribe this existing audio file")
|
|
62
|
+
|
|
63
|
+
live = sub.add_parser("live", parents=[common], help="Segmented live transcription")
|
|
64
|
+
live.add_argument("--segment-seconds", type=int, default=8, help="Segment length in seconds (default: 8)")
|
|
65
|
+
live.add_argument("--output", default=None, help="Append live transcript to this file as it's produced")
|
|
66
|
+
|
|
67
|
+
args = parser.parse_args()
|
|
68
|
+
|
|
69
|
+
if args.list_sources:
|
|
70
|
+
sources = list_pulse_sources()
|
|
71
|
+
if not sources:
|
|
72
|
+
print("No PulseAudio sources found. Is PulseAudio/PipeWire running?", file=sys.stderr)
|
|
73
|
+
sys.exit(1)
|
|
74
|
+
print("Available PulseAudio sources:")
|
|
75
|
+
for name, _ in sources:
|
|
76
|
+
print(" ", name)
|
|
77
|
+
return
|
|
78
|
+
|
|
79
|
+
# Determine source
|
|
80
|
+
source = args.source or get_default_monitor_source()
|
|
81
|
+
|
|
82
|
+
if args.mode == "once":
|
|
83
|
+
# Determine output file path
|
|
84
|
+
output_dir = ensure_output_dir()
|
|
85
|
+
if args.output:
|
|
86
|
+
# If user specified a path, use it as-is (could be relative or absolute)
|
|
87
|
+
output_file = args.output
|
|
88
|
+
else:
|
|
89
|
+
# Generate timestamp-based filename in output/ directory
|
|
90
|
+
output_file = os.path.join(output_dir, get_timestamp_filename())
|
|
91
|
+
|
|
92
|
+
if args.input:
|
|
93
|
+
audio_path = args.input
|
|
94
|
+
else:
|
|
95
|
+
# Make a temp WAV, record until duration/ctrl-c, then transcribe
|
|
96
|
+
with tempfile.TemporaryDirectory(prefix="sys2txt_") as tmp:
|
|
97
|
+
wav = os.path.join(tmp, "capture.wav")
|
|
98
|
+
record_once(source=source, out_wav=wav, sample_rate=16000, channels=1, duration=args.duration)
|
|
99
|
+
audio_path = wav
|
|
100
|
+
text = transcribe_file(
|
|
101
|
+
audio_path,
|
|
102
|
+
engine=args.engine,
|
|
103
|
+
model_size=args.model_size,
|
|
104
|
+
language=args.language,
|
|
105
|
+
timestamps=args.timestamps,
|
|
106
|
+
)
|
|
107
|
+
print(text)
|
|
108
|
+
with open(output_file, "w", encoding="utf-8") as w:
|
|
109
|
+
w.write(text + "\n")
|
|
110
|
+
print(f"Transcript saved to: {output_file}")
|
|
111
|
+
return
|
|
112
|
+
# If input provided, just transcribe it
|
|
113
|
+
text = transcribe_file(
|
|
114
|
+
audio_path,
|
|
115
|
+
engine=args.engine,
|
|
116
|
+
model_size=args.model_size,
|
|
117
|
+
language=args.language,
|
|
118
|
+
timestamps=args.timestamps,
|
|
119
|
+
)
|
|
120
|
+
print(text)
|
|
121
|
+
with open(output_file, "w", encoding="utf-8") as w:
|
|
122
|
+
w.write(text + "\n")
|
|
123
|
+
print(f"Transcript saved to: {output_file}")
|
|
124
|
+
|
|
125
|
+
elif args.mode == "live":
|
|
126
|
+
# Determine output file path
|
|
127
|
+
output_dir = ensure_output_dir()
|
|
128
|
+
if args.output:
|
|
129
|
+
# If user specified a path, use it as-is (could be relative or absolute)
|
|
130
|
+
output_file = args.output
|
|
131
|
+
else:
|
|
132
|
+
# Generate timestamp-based filename in output/ directory
|
|
133
|
+
output_file = os.path.join(output_dir, get_timestamp_filename())
|
|
134
|
+
|
|
135
|
+
print(f"Live transcript will be saved to: {output_file}")
|
|
136
|
+
|
|
137
|
+
def transcribe_segment(file_path: str, segment_index: int) -> str:
|
|
138
|
+
"""Transcribe a segment and format with optional timestamp prefix."""
|
|
139
|
+
text = transcribe_file(
|
|
140
|
+
file_path,
|
|
141
|
+
engine=args.engine,
|
|
142
|
+
model_size=args.model_size,
|
|
143
|
+
language=args.language,
|
|
144
|
+
timestamps=args.timestamps,
|
|
145
|
+
)
|
|
146
|
+
if args.timestamps:
|
|
147
|
+
# Add segment time window prefix
|
|
148
|
+
start = segment_index * args.segment_seconds
|
|
149
|
+
end = start + args.segment_seconds
|
|
150
|
+
prefix = f"[{start:>5d}-{end:>5d}s] "
|
|
151
|
+
return prefix + text.strip()
|
|
152
|
+
else:
|
|
153
|
+
return text.strip()
|
|
154
|
+
|
|
155
|
+
segment_and_transcribe_live(
|
|
156
|
+
source=source,
|
|
157
|
+
sample_rate=16000,
|
|
158
|
+
channels=1,
|
|
159
|
+
segment_seconds=args.segment_seconds,
|
|
160
|
+
transcribe_callback=transcribe_segment,
|
|
161
|
+
output_path=output_file,
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
if __name__ == "__main__":
|
|
166
|
+
try:
|
|
167
|
+
main()
|
|
168
|
+
except RuntimeError as e:
|
|
169
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
170
|
+
sys.exit(1)
|
|
171
|
+
except KeyboardInterrupt:
|
|
172
|
+
pass
|
sys2txt/audio.py
ADDED
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Audio recording functionality using ffmpeg and PulseAudio/PipeWire."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import signal
|
|
5
|
+
import subprocess
|
|
6
|
+
import tempfile
|
|
7
|
+
import time
|
|
8
|
+
from typing import Optional
|
|
9
|
+
|
|
10
|
+
from .utils import which
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def record_once(source: str, out_wav: str, sample_rate: int, channels: int, duration: Optional[int]) -> None:
|
|
14
|
+
"""Record audio once from a PulseAudio source to a WAV file.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
source: PulseAudio source name (e.g., "sink.monitor")
|
|
18
|
+
out_wav: Output WAV file path
|
|
19
|
+
sample_rate: Sample rate in Hz (e.g., 16000)
|
|
20
|
+
channels: Number of audio channels (1 for mono, 2 for stereo)
|
|
21
|
+
duration: Optional recording duration in seconds. If None, records until interrupted.
|
|
22
|
+
"""
|
|
23
|
+
ffmpeg = which("ffmpeg")
|
|
24
|
+
args = [
|
|
25
|
+
ffmpeg,
|
|
26
|
+
"-nostdin",
|
|
27
|
+
"-hide_banner",
|
|
28
|
+
"-loglevel",
|
|
29
|
+
"error",
|
|
30
|
+
"-f",
|
|
31
|
+
"pulse",
|
|
32
|
+
"-i",
|
|
33
|
+
source,
|
|
34
|
+
"-ac",
|
|
35
|
+
str(channels),
|
|
36
|
+
"-ar",
|
|
37
|
+
str(sample_rate),
|
|
38
|
+
"-f",
|
|
39
|
+
"wav",
|
|
40
|
+
]
|
|
41
|
+
if duration is not None and duration > 0:
|
|
42
|
+
args.extend(["-t", str(duration)])
|
|
43
|
+
args.append(out_wav)
|
|
44
|
+
|
|
45
|
+
print(f"Recording system audio from source '{source}' at {sample_rate} Hz, mono -> {out_wav}")
|
|
46
|
+
print("Press Ctrl-C to stop early..." if duration is None else f"Recording for {duration} seconds...")
|
|
47
|
+
|
|
48
|
+
proc = subprocess.Popen(args)
|
|
49
|
+
try:
|
|
50
|
+
proc.wait()
|
|
51
|
+
except KeyboardInterrupt:
|
|
52
|
+
try:
|
|
53
|
+
proc.send_signal(signal.SIGINT)
|
|
54
|
+
except Exception:
|
|
55
|
+
pass
|
|
56
|
+
proc.wait()
|
|
57
|
+
print("Recording finished.")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def segment_and_transcribe_live(
|
|
61
|
+
source: str,
|
|
62
|
+
sample_rate: int,
|
|
63
|
+
channels: int,
|
|
64
|
+
segment_seconds: int,
|
|
65
|
+
transcribe_callback,
|
|
66
|
+
output_path: Optional[str],
|
|
67
|
+
) -> None:
|
|
68
|
+
"""Record audio in segments and transcribe each segment as it's created.
|
|
69
|
+
|
|
70
|
+
Args:
|
|
71
|
+
source: PulseAudio source name
|
|
72
|
+
sample_rate: Sample rate in Hz
|
|
73
|
+
channels: Number of audio channels
|
|
74
|
+
segment_seconds: Length of each segment in seconds
|
|
75
|
+
transcribe_callback: Function to call for each segment. Should accept (file_path, segment_index) and return text
|
|
76
|
+
output_path: Optional file path to append transcripts to
|
|
77
|
+
"""
|
|
78
|
+
ffmpeg = which("ffmpeg")
|
|
79
|
+
with tempfile.TemporaryDirectory(prefix="sys2txt_") as tmp:
|
|
80
|
+
pattern = os.path.join(tmp, "seg_%05d.wav")
|
|
81
|
+
args = [
|
|
82
|
+
ffmpeg,
|
|
83
|
+
"-hide_banner",
|
|
84
|
+
"-loglevel",
|
|
85
|
+
"error",
|
|
86
|
+
"-f",
|
|
87
|
+
"pulse",
|
|
88
|
+
"-i",
|
|
89
|
+
source,
|
|
90
|
+
"-ac",
|
|
91
|
+
str(channels),
|
|
92
|
+
"-ar",
|
|
93
|
+
str(sample_rate),
|
|
94
|
+
"-f",
|
|
95
|
+
"segment",
|
|
96
|
+
"-segment_time",
|
|
97
|
+
str(segment_seconds),
|
|
98
|
+
"-reset_timestamps",
|
|
99
|
+
"1",
|
|
100
|
+
pattern,
|
|
101
|
+
]
|
|
102
|
+
|
|
103
|
+
print(f"Live mode: segmenting every {segment_seconds}s from '{source}'. Press Ctrl-C to stop.")
|
|
104
|
+
proc = subprocess.Popen(args, stdin=subprocess.PIPE, stderr=subprocess.PIPE)
|
|
105
|
+
processed: set[str] = set()
|
|
106
|
+
try:
|
|
107
|
+
while True:
|
|
108
|
+
# sorted ensures we process in chronological order
|
|
109
|
+
files = sorted(f for f in os.listdir(tmp) if f.startswith("seg_") and f.endswith(".wav"))
|
|
110
|
+
new_files = [f for f in files if f not in processed]
|
|
111
|
+
for f in new_files:
|
|
112
|
+
full = os.path.join(tmp, f)
|
|
113
|
+
# Ensure the segment has been finalized and has content
|
|
114
|
+
if os.path.getsize(full) < 64:
|
|
115
|
+
continue
|
|
116
|
+
processed.add(f)
|
|
117
|
+
|
|
118
|
+
# Extract segment index from filename
|
|
119
|
+
try:
|
|
120
|
+
idx = int(os.path.splitext(f)[0].split("_")[-1])
|
|
121
|
+
except Exception:
|
|
122
|
+
idx = 0
|
|
123
|
+
|
|
124
|
+
text = transcribe_callback(full, idx)
|
|
125
|
+
print(text, flush=True)
|
|
126
|
+
if output_path:
|
|
127
|
+
with open(output_path, "a", encoding="utf-8") as w:
|
|
128
|
+
w.write(text + "\n")
|
|
129
|
+
|
|
130
|
+
# If ffmpeg has exited and no new files pending, break
|
|
131
|
+
ret = proc.poll()
|
|
132
|
+
if ret is not None:
|
|
133
|
+
# flush remaining unprocessed files
|
|
134
|
+
files = sorted(f for f in os.listdir(tmp) if f.startswith("seg_") and f.endswith(".wav"))
|
|
135
|
+
new_files = [f for f in files if f not in processed]
|
|
136
|
+
for f in new_files:
|
|
137
|
+
full = os.path.join(tmp, f)
|
|
138
|
+
if os.path.getsize(full) < 64:
|
|
139
|
+
continue
|
|
140
|
+
processed.add(f)
|
|
141
|
+
try:
|
|
142
|
+
idx = int(os.path.splitext(f)[0].split("_")[-1])
|
|
143
|
+
except Exception:
|
|
144
|
+
idx = 0
|
|
145
|
+
text = transcribe_callback(full, idx)
|
|
146
|
+
print(text, flush=True)
|
|
147
|
+
if output_path:
|
|
148
|
+
with open(output_path, "a", encoding="utf-8") as w:
|
|
149
|
+
w.write(text + "\n")
|
|
150
|
+
break
|
|
151
|
+
time.sleep(0.3)
|
|
152
|
+
except KeyboardInterrupt:
|
|
153
|
+
print("\nStopping live capture...")
|
|
154
|
+
try:
|
|
155
|
+
# Send 'q' command to ffmpeg to quit gracefully
|
|
156
|
+
if proc.stdin:
|
|
157
|
+
proc.stdin.write(b"q")
|
|
158
|
+
proc.stdin.flush()
|
|
159
|
+
proc.stdin.close()
|
|
160
|
+
# Give ffmpeg time to finish the current segment
|
|
161
|
+
try:
|
|
162
|
+
proc.wait(timeout=3.0)
|
|
163
|
+
except subprocess.TimeoutExpired:
|
|
164
|
+
# If it doesn't finish in time, terminate it
|
|
165
|
+
proc.terminate()
|
|
166
|
+
proc.wait()
|
|
167
|
+
except Exception:
|
|
168
|
+
pass
|
|
169
|
+
print("Stopped live capture.")
|
sys2txt/constants.py
ADDED
sys2txt/pulse.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""PulseAudio/PipeWire integration for source enumeration and selection."""
|
|
2
|
+
|
|
3
|
+
import subprocess
|
|
4
|
+
from typing import List, Tuple
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def run_command(cmd: List[str]) -> Tuple[int, str, str]:
|
|
8
|
+
"""Run a command and return exit code, stdout, stderr."""
|
|
9
|
+
p = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)
|
|
10
|
+
out, err = p.communicate()
|
|
11
|
+
return p.returncode, out, err
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def list_pulse_sources() -> List[Tuple[str, str]]:
|
|
15
|
+
"""Return list of (name, description) for PulseAudio sources."""
|
|
16
|
+
try:
|
|
17
|
+
code, out, _ = run_command(["pactl", "list", "short", "sources"])
|
|
18
|
+
if code != 0:
|
|
19
|
+
return []
|
|
20
|
+
items: List[Tuple[str, str]] = []
|
|
21
|
+
for line in out.splitlines():
|
|
22
|
+
# Format: index\tname\tmodule\tsampleSpec\tstate
|
|
23
|
+
parts = line.split("\t")
|
|
24
|
+
if len(parts) >= 2:
|
|
25
|
+
name = parts[1]
|
|
26
|
+
items.append((name, name))
|
|
27
|
+
return items
|
|
28
|
+
except FileNotFoundError:
|
|
29
|
+
return []
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def get_default_monitor_source() -> str:
|
|
33
|
+
"""Pick the default sink's .monitor if available; otherwise the first *.monitor source; else 'default'."""
|
|
34
|
+
try:
|
|
35
|
+
code, sink_name, _ = run_command(["pactl", "get-default-sink"])
|
|
36
|
+
sink_name = sink_name.strip()
|
|
37
|
+
if code == 0 and sink_name:
|
|
38
|
+
candidate = f"{sink_name}.monitor"
|
|
39
|
+
sources = [s for s, _ in list_pulse_sources()]
|
|
40
|
+
if candidate in sources:
|
|
41
|
+
return candidate
|
|
42
|
+
# fallback: first *.monitor source
|
|
43
|
+
for s, _ in list_pulse_sources():
|
|
44
|
+
if s.endswith(".monitor"):
|
|
45
|
+
return s
|
|
46
|
+
except Exception:
|
|
47
|
+
pass
|
|
48
|
+
return "default"
|
sys2txt/transcribe.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Transcription functionality using Whisper models."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def transcribe_file(path: str, engine: str, model_size: str, language: Optional[str], timestamps: bool) -> str:
|
|
8
|
+
"""Transcribe an audio file using the specified Whisper engine.
|
|
9
|
+
|
|
10
|
+
Args:
|
|
11
|
+
path: Path to audio file
|
|
12
|
+
engine: Engine to use ("auto", "faster", or "whisper")
|
|
13
|
+
model_size: Whisper model size (tiny, base, small, medium, large-v2)
|
|
14
|
+
language: Optional language code (e.g., "en"). If None, auto-detect.
|
|
15
|
+
timestamps: Whether to include timestamps in output
|
|
16
|
+
|
|
17
|
+
Returns:
|
|
18
|
+
Transcribed text
|
|
19
|
+
"""
|
|
20
|
+
engine = engine.lower()
|
|
21
|
+
if engine == "auto":
|
|
22
|
+
try:
|
|
23
|
+
engine = "faster"
|
|
24
|
+
except Exception:
|
|
25
|
+
engine = "whisper"
|
|
26
|
+
|
|
27
|
+
if engine == "faster":
|
|
28
|
+
return _transcribe_faster_whisper(path, model_size, language, timestamps)
|
|
29
|
+
elif engine == "whisper":
|
|
30
|
+
return _transcribe_openai_whisper(path, model_size, language, timestamps)
|
|
31
|
+
else:
|
|
32
|
+
raise ValueError(f"Unknown engine: {engine}")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _transcribe_faster_whisper(path: str, model_size: str, language: Optional[str], timestamps: bool) -> str:
|
|
36
|
+
"""Transcribe using faster-whisper (ctranslate2 backend)."""
|
|
37
|
+
try:
|
|
38
|
+
from faster_whisper import WhisperModel # type: ignore
|
|
39
|
+
except Exception as e:
|
|
40
|
+
raise RuntimeError("faster-whisper is not installed. pip install faster-whisper") from e
|
|
41
|
+
|
|
42
|
+
# Auto device selection
|
|
43
|
+
device = "cpu"
|
|
44
|
+
compute_type = "int8"
|
|
45
|
+
# If user has a CUDA-enabled ctranslate2 build installed, they can switch manually by editing below
|
|
46
|
+
# or by setting environment variable SYS2TXT_DEVICE=cuda
|
|
47
|
+
if os.environ.get("SYS2TXT_DEVICE") == "cuda":
|
|
48
|
+
device = "cuda"
|
|
49
|
+
compute_type = "float16"
|
|
50
|
+
|
|
51
|
+
model = WhisperModel(model_size, device=device, compute_type=compute_type)
|
|
52
|
+
segments, info = model.transcribe(path, vad_filter=True, language=language)
|
|
53
|
+
if timestamps:
|
|
54
|
+
lines = []
|
|
55
|
+
for seg in segments:
|
|
56
|
+
s = f"[{seg.start:6.2f}-{seg.end:6.2f}] {seg.text.strip()}"
|
|
57
|
+
lines.append(s)
|
|
58
|
+
return "\n".join(lines)
|
|
59
|
+
else:
|
|
60
|
+
text_parts = [seg.text for seg in segments]
|
|
61
|
+
return " ".join(t.strip() for t in text_parts).strip()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _transcribe_openai_whisper(path: str, model_size: str, language: Optional[str], timestamps: bool) -> str:
|
|
65
|
+
"""Transcribe using openai-whisper (reference implementation)."""
|
|
66
|
+
try:
|
|
67
|
+
import whisper # type: ignore
|
|
68
|
+
except Exception as e:
|
|
69
|
+
raise RuntimeError("openai-whisper is not installed. pip install openai-whisper") from e
|
|
70
|
+
|
|
71
|
+
model = whisper.load_model(model_size)
|
|
72
|
+
result = model.transcribe(path, language=language)
|
|
73
|
+
if timestamps and "segments" in result:
|
|
74
|
+
lines = []
|
|
75
|
+
for seg in result.get("segments", []):
|
|
76
|
+
start = seg.get("start", 0.0)
|
|
77
|
+
end = seg.get("end", 0.0)
|
|
78
|
+
text = seg.get("text", "").strip()
|
|
79
|
+
lines.append(f"[{start:6.2f}-{end:6.2f}] {text}")
|
|
80
|
+
return "\n".join(lines)
|
|
81
|
+
else:
|
|
82
|
+
return result.get("text", "").strip()
|
sys2txt/utils.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Utility functions."""
|
|
2
|
+
|
|
3
|
+
import shutil
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def which(cmd: str) -> str:
|
|
7
|
+
"""Find command in PATH or raise RuntimeError if not found."""
|
|
8
|
+
path = shutil.which(cmd)
|
|
9
|
+
if not path:
|
|
10
|
+
raise RuntimeError(f"Required command not found: {cmd}. Please install it and try again.")
|
|
11
|
+
return path
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sys2txt
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Record system audio and transcribe to text using AI
|
|
5
|
+
Author-email: Joe Heffer <jheffer@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: faster-whisper>=1.0.0
|
|
11
|
+
Requires-Dist: openai-whisper>=20231117
|
|
12
|
+
Provides-Extra: dev
|
|
13
|
+
Requires-Dist: ruff>=0.8.0; extra == "dev"
|
|
14
|
+
Dynamic: license-file
|
|
15
|
+
|
|
16
|
+
[](https://github.com/Joe-Heffer/sys2txt/actions/workflows/ci.yml)
|
|
17
|
+
[](https://badge.fury.io/py/sys2txt)
|
|
18
|
+
[](https://pypi.org/project/sys2txt/)
|
|
19
|
+
|
|
20
|
+
# System audio to text
|
|
21
|
+
|
|
22
|
+
Record system audio and automatically transcribe to text using ✨AI✨.
|
|
23
|
+
|
|
24
|
+
## Overview
|
|
25
|
+
|
|
26
|
+
`sys2txt` is a command-line tool that records your system audio (via PulseAudio/PipeWire monitor sources) with `ffmpeg` and transcribes it locally using [Whisper](https://github.com/openai/whisper). It supports both:
|
|
27
|
+
|
|
28
|
+
- On-demand: Record until you stop, then transcribe once
|
|
29
|
+
- Live-ish: Segment the recording every *N* seconds and transcribe each segment as it’s created (prints continuously)
|
|
30
|
+
|
|
31
|
+
You can use either the `openai-whisper` (Python) reference implementation or the [`faster-whisper`](https://github.com/SYSTRAN/faster-whisper) engine if installed. The tool auto-selects `faster-whisper` when available for better speed on CPU and especially GPU.
|
|
32
|
+
|
|
33
|
+
## Installation
|
|
34
|
+
|
|
35
|
+
### Prerequisites
|
|
36
|
+
|
|
37
|
+
- Ubuntu with PulseAudio or PipeWire (default on modern Ubuntu)
|
|
38
|
+
- ffmpeg
|
|
39
|
+
- Python 3.9+ (recommended)
|
|
40
|
+
|
|
41
|
+
### Install
|
|
42
|
+
|
|
43
|
+
1) System packages
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
sudo apt update
|
|
47
|
+
sudo apt install -y ffmpeg python3-venv python3-pip
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
2) Create a virtual environment and install sys2txt
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
cd sys2txt
|
|
54
|
+
python3 -m venv .venv
|
|
55
|
+
source .venv/bin/activate
|
|
56
|
+
pip install -e .
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
This installs both faster-whisper (for speed) and openai-whisper (reference implementation). The tool auto-selects faster-whisper when available or falls back to openai-whisper.
|
|
60
|
+
|
|
61
|
+
## Quick start
|
|
62
|
+
|
|
63
|
+
Record and transcribe once (press Ctrl-C to stop recording):
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
sys2txt once --model small.en
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Live segmented transcription (prints ongoing transcript every 8s by default; Ctrl-C to stop):
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
sys2txt live --model small.en --segment-seconds 8
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
### Useful flags
|
|
76
|
+
|
|
77
|
+
- `--source <pulse_source_name>` - Explicit PulseAudio/PipeWire source (e.g., alsa_output.pci-0000_00_1f.3.analog-stereo.monitor)
|
|
78
|
+
- `--list-sources` - List available Pulse sources and exit
|
|
79
|
+
- `--model <size>` - tiny|base|small|medium|large-v2 (default: small)
|
|
80
|
+
- `--engine <auto|faster|whisper>` - Force a specific engine (default: auto)
|
|
81
|
+
- `--language <code>` - Force language code (e.g., en). Omit to auto-detect
|
|
82
|
+
- `--output <path>` - Write final transcript to a file (in live mode, appends)
|
|
83
|
+
- `--duration <seconds>` - (once mode) Record fixed duration instead of waiting for Ctrl-C
|
|
84
|
+
- `--segment-seconds <n>` - (live mode) Segment length in seconds (default: 8)
|
|
85
|
+
- `--timestamps` - Print timestamps alongside text
|
|
86
|
+
|
|
87
|
+
## Examples
|
|
88
|
+
|
|
89
|
+
Record 30s of system audio from the default monitor and transcribe:
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
sys2txt once --duration 30 --model small --output transcript.txt
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Use a specific PulseAudio source:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
sys2txt once --source alsa_output.usb-Focusrite_Scarlett.monitor --model base
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Live mode with shorter latency and timestamps:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
sys2txt live --segment-seconds 5 --timestamps
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Force the reference openai-whisper engine:
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
sys2txt once --engine whisper --model base
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Transcribe an existing audio file:
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
sys2txt once --input recording.wav --model small
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
### Just want one-liners (no sys2txt)?
|
|
120
|
+
|
|
121
|
+
Find the default sink and its monitor source:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
pactl get-default-sink
|
|
125
|
+
pactl list short sources | grep monitor
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Record 30s of system audio from the default monitor to a WAV at 16 kHz mono (good for Whisper):
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
ffmpeg -hide_banner -loglevel error -f pulse -i "$(pactl get-default-sink).monitor" -ac 1 -ar 16000 -t 30 out.wav
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Transcribe with openai-whisper CLI:
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
whisper out.wav --model small --task transcribe --language en
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Tips and troubleshooting
|
|
141
|
+
|
|
142
|
+
- If you get silence, ensure you are using the monitor source for your output device (the name ends with `.monitor`). Use `--list-sources` to view options.
|
|
143
|
+
- Make sure the application you want to capture is playing through the same output sink as your default sink. You can manage routes with `pavucontrol`.
|
|
144
|
+
- PipeWire systems expose PulseAudio-compatible sources, so `-f pulse` in ffmpeg still works.
|
|
145
|
+
- For better performance on CPU, use faster-whisper with model `base` or `small`. For the best accuracy, use `medium` or `large-v2` (these are heavier).
|
|
146
|
+
- GPU acceleration for faster-whisper requires a compatible ctranslate2 CUDA wheel. Set `SYS2TXT_DEVICE=cuda` to enable it. If not available, it will run on CPU.
|
|
147
|
+
|
|
148
|
+
## Development
|
|
149
|
+
|
|
150
|
+
Install with development dependencies:
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
pip install -e ".[dev]"
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Run unit tests:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
python -m unittest discover -s tests -p "test_*.py"
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
Format and lint code:
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
ruff format src/
|
|
166
|
+
ruff check src/
|
|
167
|
+
```
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
sys2txt/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
2
|
+
sys2txt/__main__.py,sha256=aTFF9J885f-548RmiurZCCJrm0bar09gM0XuARRxQ6E,6525
|
|
3
|
+
sys2txt/audio.py,sha256=cTcYAAgiRipmLkaEsl5VIUCEliGW3fbjZ9wWm6rq2bw,5987
|
|
4
|
+
sys2txt/constants.py,sha256=iqZGZIHROECa4iQTcVNQxFSmyEbZ5Eufd7O1Ikp7zjg,90
|
|
5
|
+
sys2txt/pulse.py,sha256=d1vcrS-7opktxRaes8Yguo_0WVTSbfkm9mes7wyUGS8,1694
|
|
6
|
+
sys2txt/transcribe.py,sha256=-6o9nSWzW2U8oIRMzA58nXxumYESv0s_9xIEYgBZSEA,3129
|
|
7
|
+
sys2txt/utils.py,sha256=h94hGzrIcp9aTSZqrWCeU9oUvxxHENQ1sq2Co6i47qM,298
|
|
8
|
+
sys2txt-0.1.1.dist-info/licenses/LICENSE,sha256=IuEA70YNQHOd_2f9RbqyzAvpGOz3qbA__ms65xlQn6Y,1067
|
|
9
|
+
sys2txt-0.1.1.dist-info/METADATA,sha256=0iMaalfrlq_X-A9y6Gb8w8-2McMU8HqsLJF7OBDVZOo,5196
|
|
10
|
+
sys2txt-0.1.1.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
11
|
+
sys2txt-0.1.1.dist-info/entry_points.txt,sha256=-g_M1WBoKXNaKfByGAPFjz_3fLCA2SE4BwSr4PlDcgc,50
|
|
12
|
+
sys2txt-0.1.1.dist-info/top_level.txt,sha256=V_PwMcyUAVJZdER8Jr616x_wWC7wJ7OdQu6nD9Hbv2E,8
|
|
13
|
+
sys2txt-0.1.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Joe Heffer
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
sys2txt
|