audiosense 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- audiosense/__init__.py +78 -0
- audiosense/audio/__init__.py +12 -0
- audiosense/audio/conversion.py +103 -0
- audiosense/audio/io.py +37 -0
- audiosense/audio/preprocess.py +115 -0
- audiosense/cli.py +42 -0
- audiosense/context/__init__.py +10 -0
- audiosense/context/download_llm.py +49 -0
- audiosense/context/generator.py +86 -0
- audiosense/context/llm.py +100 -0
- audiosense/context/translator.py +73 -0
- audiosense/core/__init__.py +3 -0
- audiosense/core/config.py +69 -0
- audiosense/core/engine.py +228 -0
- audiosense/core/silence.py +124 -0
- audiosense/label.py +25 -0
- audiosense/labels/__init__.py +16 -0
- audiosense/labels/all.py +63 -0
- audiosense/labels/categories.py +42 -0
- audiosense/labels/selected.py +56 -0
- audiosense/labels/taxonomy.py +93 -0
- audiosense/labels/top.py +70 -0
- audiosense/live/__init__.py +3 -0
- audiosense/live/loop.py +103 -0
- audiosense/models/__init__.py +13 -0
- audiosense/models/ast_model.py +79 -0
- audiosense/models/download.py +112 -0
- audiosense/models/manager.py +45 -0
- audiosense/models/pann_model.py +60 -0
- audiosense/models/whisper_model.py +48 -0
- audiosense/models/yamnet_model.py +75 -0
- audiosense/storage/__init__.py +3 -0
- audiosense/storage/database.py +136 -0
- audiosense/switcher/__init__.py +4 -0
- audiosense/switcher/router.py +128 -0
- audiosense/switcher/vad.py +40 -0
- audiosense-1.0.0.dist-info/METADATA +334 -0
- audiosense-1.0.0.dist-info/RECORD +42 -0
- audiosense-1.0.0.dist-info/WHEEL +5 -0
- audiosense-1.0.0.dist-info/entry_points.txt +4 -0
- audiosense-1.0.0.dist-info/licenses/LICENSE +21 -0
- audiosense-1.0.0.dist-info/top_level.txt +1 -0
audiosense/__init__.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""
|
|
2
|
+
AudioSense: Audio intelligence framework for machines, robots, and applications.
|
|
3
|
+
Version 1.0.0
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
# Silence all background logs, oneDNN warnings, and progress bars immediately
|
|
7
|
+
import os
|
|
8
|
+
os.environ["TF_CPP_MIN_LOG_LEVEL"] = "3"
|
|
9
|
+
os.environ["TF_ENABLE_ONEDNN_OPTS"] = "0"
|
|
10
|
+
os.environ["AUTOGRAPH_VERBOSITY"] = "0"
|
|
11
|
+
os.environ["GLOG_minloglevel"] = "3"
|
|
12
|
+
os.environ["PYTHONWARNINGS"] = "ignore"
|
|
13
|
+
|
|
14
|
+
from .core.silence import silence_all, disable_tqdm
|
|
15
|
+
disable_tqdm()
|
|
16
|
+
|
|
17
|
+
import sys
|
|
18
|
+
from .core.engine import AudioSense, AudioSenseResult
|
|
19
|
+
from .audio.conversion import convert_to_wav, is_audio_file
|
|
20
|
+
from .audio.preprocess import preprocess_audio, ProcessedAudio
|
|
21
|
+
from .labels.selected import get_label, get_labels
|
|
22
|
+
from .labels.all import get_all, get_all_labels
|
|
23
|
+
from .labels.categories import get_categories
|
|
24
|
+
from .labels.top import top_k
|
|
25
|
+
from .storage.database import view_db
|
|
26
|
+
from .live.loop import run_mic_loop
|
|
27
|
+
from .models.download import download_all_models
|
|
28
|
+
from .context.download_llm import download_llm_model
|
|
29
|
+
|
|
30
|
+
# Register audiosense.label alias in sys.modules so `from audiosense.label import ...` works
|
|
31
|
+
from . import label as _label_mod
|
|
32
|
+
sys.modules["audiosense.label"] = _label_mod
|
|
33
|
+
|
|
34
|
+
__version__ = "1.0.0"
|
|
35
|
+
|
|
36
|
+
def summary_db(llm_path=None, language="en") -> str:
|
|
37
|
+
"""Convenience top-level summary_db function."""
|
|
38
|
+
engine = AudioSense(llm_path=llm_path, language=language)
|
|
39
|
+
return engine.summary_db()
|
|
40
|
+
|
|
41
|
+
def loop(chunk_duration: float = 3.0, sample_rate: int = 16000, device=None, callback=None):
|
|
42
|
+
"""Convenience top-level audiosense.loop() function."""
|
|
43
|
+
engine = AudioSense()
|
|
44
|
+
return engine.loop(
|
|
45
|
+
chunk_duration=chunk_duration,
|
|
46
|
+
sample_rate=sample_rate,
|
|
47
|
+
device=device,
|
|
48
|
+
callback=callback
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
def download_models(models_dir=None) -> bool:
|
|
52
|
+
"""Download all core acoustic models (AST, PANN, YAMNet)."""
|
|
53
|
+
return download_all_models(models_dir=models_dir)
|
|
54
|
+
|
|
55
|
+
def download_llm(llm_dir=None):
|
|
56
|
+
"""Download local Phi-4-mini GGUF model."""
|
|
57
|
+
return download_llm_model(llm_dir=llm_dir)
|
|
58
|
+
|
|
59
|
+
__all__ = [
|
|
60
|
+
"AudioSense",
|
|
61
|
+
"AudioSenseResult",
|
|
62
|
+
"convert_to_wav",
|
|
63
|
+
"is_audio_file",
|
|
64
|
+
"preprocess_audio",
|
|
65
|
+
"ProcessedAudio",
|
|
66
|
+
"get_label",
|
|
67
|
+
"get_labels",
|
|
68
|
+
"get_all",
|
|
69
|
+
"get_all_labels",
|
|
70
|
+
"get_categories",
|
|
71
|
+
"top_k",
|
|
72
|
+
"view_db",
|
|
73
|
+
"summary_db",
|
|
74
|
+
"loop",
|
|
75
|
+
"download_models",
|
|
76
|
+
"download_llm",
|
|
77
|
+
"__version__",
|
|
78
|
+
]
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from .conversion import convert_to_wav, is_audio_file
|
|
2
|
+
from .preprocess import ProcessedAudio, preprocess_audio
|
|
3
|
+
from .io import load_audio, save_audio
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"convert_to_wav",
|
|
7
|
+
"is_audio_file",
|
|
8
|
+
"ProcessedAudio",
|
|
9
|
+
"preprocess_audio",
|
|
10
|
+
"load_audio",
|
|
11
|
+
"save_audio",
|
|
12
|
+
]
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Audio Conversion Module for AudioSense.
|
|
3
|
+
Accepts any audio or video container format (MP3, FLAC, OGG, M4A, AAC, OPUS, WEBM, MP4, etc.)
|
|
4
|
+
and converts it to standardized .wav format for model processing.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Union, Optional
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
13
|
+
def convert_to_wav(
|
|
14
|
+
input_path: Union[str, Path],
|
|
15
|
+
output_path: Optional[Union[str, Path]] = None,
|
|
16
|
+
target_sr: Optional[int] = None,
|
|
17
|
+
mono: bool = True
|
|
18
|
+
) -> Path:
|
|
19
|
+
"""
|
|
20
|
+
Converts any audio file to .wav format.
|
|
21
|
+
|
|
22
|
+
Parameters
|
|
23
|
+
----------
|
|
24
|
+
input_path : str | Path
|
|
25
|
+
Path to the source audio/video file.
|
|
26
|
+
output_path : str | Path, optional
|
|
27
|
+
Destination path. If None, creates a file with '.wav' extension next to the source
|
|
28
|
+
or in a temporary directory if source directory is not writable.
|
|
29
|
+
target_sr : int, optional
|
|
30
|
+
Target sample rate. If None, keeps original or defaults to standard 16000 or 32000.
|
|
31
|
+
mono : bool, default True
|
|
32
|
+
Whether to convert channels to single mono channel.
|
|
33
|
+
|
|
34
|
+
Returns
|
|
35
|
+
-------
|
|
36
|
+
Path
|
|
37
|
+
Absolute path to the resulting .wav file.
|
|
38
|
+
"""
|
|
39
|
+
input_path = Path(input_path).resolve()
|
|
40
|
+
if not input_path.exists():
|
|
41
|
+
raise FileNotFoundError(f"Input audio file not found: {input_path}")
|
|
42
|
+
|
|
43
|
+
# Determine destination
|
|
44
|
+
if output_path is None:
|
|
45
|
+
if input_path.suffix.lower() == ".wav" and target_sr is None:
|
|
46
|
+
return input_path
|
|
47
|
+
output_path = input_path.with_suffix(".wav")
|
|
48
|
+
# If output matches input, rename output
|
|
49
|
+
if output_path == input_path:
|
|
50
|
+
output_path = input_path.parent / f"{input_path.stem}_converted.wav"
|
|
51
|
+
else:
|
|
52
|
+
output_path = Path(output_path).resolve()
|
|
53
|
+
|
|
54
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
55
|
+
|
|
56
|
+
# Try librosa + soundfile (best for universal decoding)
|
|
57
|
+
try:
|
|
58
|
+
import librosa
|
|
59
|
+
import soundfile as sf
|
|
60
|
+
|
|
61
|
+
audio, sr = librosa.load(str(input_path), sr=target_sr, mono=mono)
|
|
62
|
+
sf.write(str(output_path), audio, samplerate=sr, subtype="PCM_16")
|
|
63
|
+
return output_path
|
|
64
|
+
except Exception as e_librosa:
|
|
65
|
+
# Fallback 1: soundfile directly
|
|
66
|
+
try:
|
|
67
|
+
import soundfile as sf
|
|
68
|
+
data, sr = sf.read(str(input_path))
|
|
69
|
+
if mono and data.ndim > 1:
|
|
70
|
+
data = data.mean(axis=1)
|
|
71
|
+
sf.write(str(output_path), data, samplerate=sr, subtype="PCM_16")
|
|
72
|
+
return output_path
|
|
73
|
+
except Exception:
|
|
74
|
+
pass
|
|
75
|
+
|
|
76
|
+
# Fallback 2: moviepy for media/video containers (e.g. mp4, webm)
|
|
77
|
+
try:
|
|
78
|
+
from moviepy.editor import AudioFileClip
|
|
79
|
+
clip = AudioFileClip(str(input_path))
|
|
80
|
+
clip.write_audiofile(
|
|
81
|
+
str(output_path),
|
|
82
|
+
fps=target_sr or 16000,
|
|
83
|
+
nbytes=2,
|
|
84
|
+
codec='pcm_s16le',
|
|
85
|
+
ffmpeg_params=["-ac", "1"] if mono else None,
|
|
86
|
+
logger=None
|
|
87
|
+
)
|
|
88
|
+
clip.close()
|
|
89
|
+
return output_path
|
|
90
|
+
except Exception:
|
|
91
|
+
pass
|
|
92
|
+
|
|
93
|
+
raise RuntimeError(
|
|
94
|
+
f"Failed to convert {input_path} to WAV. Error: {e_librosa}"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
def is_audio_file(file_path: Union[str, Path]) -> bool:
|
|
98
|
+
"""Check if file has an audio or video extension known to contain sound."""
|
|
99
|
+
supported_extensions = {
|
|
100
|
+
".wav", ".mp3", ".flac", ".ogg", ".m4a", ".aac",
|
|
101
|
+
".opus", ".webm", ".wma", ".aiff", ".mp4", ".mkv"
|
|
102
|
+
}
|
|
103
|
+
return Path(file_path).suffix.lower() in supported_extensions
|
audiosense/audio/io.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Audio I/O utilities for loading, saving, and inspecting audio.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Union, Tuple, Optional
|
|
7
|
+
import soundfile as sf
|
|
8
|
+
import numpy as np
|
|
9
|
+
import librosa
|
|
10
|
+
|
|
11
|
+
def load_audio(
|
|
12
|
+
path: Union[str, Path],
|
|
13
|
+
sr: Optional[int] = None,
|
|
14
|
+
mono: bool = True
|
|
15
|
+
) -> Tuple[np.ndarray, int]:
|
|
16
|
+
"""
|
|
17
|
+
Load an audio file into a numpy float32 array.
|
|
18
|
+
"""
|
|
19
|
+
path = Path(path).resolve()
|
|
20
|
+
if not path.exists():
|
|
21
|
+
raise FileNotFoundError(f"Audio file not found: {path}")
|
|
22
|
+
|
|
23
|
+
audio, sample_rate = librosa.load(str(path), sr=sr, mono=mono)
|
|
24
|
+
return audio.astype(np.float32), sample_rate
|
|
25
|
+
|
|
26
|
+
def save_audio(
|
|
27
|
+
path: Union[str, Path],
|
|
28
|
+
data: np.ndarray,
|
|
29
|
+
sr: int = 16000
|
|
30
|
+
) -> Path:
|
|
31
|
+
"""
|
|
32
|
+
Write float32 audio data to WAV file.
|
|
33
|
+
"""
|
|
34
|
+
path = Path(path).resolve()
|
|
35
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
36
|
+
sf.write(str(path), data, samplerate=sr, subtype="PCM_16")
|
|
37
|
+
return path
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Audio Preprocessing Module for AudioSense.
|
|
3
|
+
Internal functionality that handles dual sample-rate conditioning:
|
|
4
|
+
- 32 kHz for PANN (CNN14)
|
|
5
|
+
- 16 kHz for AST, YAMNet, Silero VAD, and Whisper
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Union, Optional
|
|
11
|
+
import numpy as np
|
|
12
|
+
import librosa
|
|
13
|
+
from scipy.signal import resample_poly
|
|
14
|
+
|
|
15
|
+
from .conversion import convert_to_wav
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class ProcessedAudio:
|
|
19
|
+
"""
|
|
20
|
+
Standard preprocessed audio container holding dual-rate representations.
|
|
21
|
+
"""
|
|
22
|
+
audio_16k: np.ndarray # 16 kHz mono float32 (AST, YAMNet, Whisper)
|
|
23
|
+
audio_32k: np.ndarray # 32 kHz mono float32 (PANN)
|
|
24
|
+
duration: float # duration in seconds
|
|
25
|
+
source_path: Optional[Path] = None
|
|
26
|
+
wav_path: Optional[Path] = None
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def preprocess(self):
|
|
30
|
+
"""Allows `audio = audio.preprocess` idempotency."""
|
|
31
|
+
return self
|
|
32
|
+
|
|
33
|
+
def __repr__(self) -> str:
|
|
34
|
+
return f"<ProcessedAudio duration={self.duration:.2f}s sr=[16kHz, 32kHz] source={self.source_path}>"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def preprocess_audio(
|
|
38
|
+
source: Union[str, Path, np.ndarray, ProcessedAudio],
|
|
39
|
+
orig_sr: Optional[int] = None
|
|
40
|
+
) -> ProcessedAudio:
|
|
41
|
+
"""
|
|
42
|
+
Condition any audio input for AudioSense models.
|
|
43
|
+
Converts to WAV if needed, transforms to mono float32, and generates
|
|
44
|
+
both 16 kHz and 32 kHz arrays.
|
|
45
|
+
|
|
46
|
+
Parameters
|
|
47
|
+
----------
|
|
48
|
+
source : str | Path | np.ndarray | ProcessedAudio
|
|
49
|
+
Audio file path or raw waveform array.
|
|
50
|
+
orig_sr : int, optional
|
|
51
|
+
Sample rate if source is a raw numpy array.
|
|
52
|
+
|
|
53
|
+
Returns
|
|
54
|
+
-------
|
|
55
|
+
ProcessedAudio
|
|
56
|
+
"""
|
|
57
|
+
if isinstance(source, ProcessedAudio):
|
|
58
|
+
return source
|
|
59
|
+
|
|
60
|
+
wav_file = None
|
|
61
|
+
source_path = None
|
|
62
|
+
|
|
63
|
+
if isinstance(source, (str, Path)):
|
|
64
|
+
source_path = Path(source).resolve()
|
|
65
|
+
# Convert non-wav formats (e.g., mp3, flac) to standard wav
|
|
66
|
+
if source_path.suffix.lower() != ".wav":
|
|
67
|
+
wav_file = convert_to_wav(source_path)
|
|
68
|
+
else:
|
|
69
|
+
wav_file = source_path
|
|
70
|
+
|
|
71
|
+
# Load 32 kHz mono (native for PANN)
|
|
72
|
+
audio_32k, sr32 = librosa.load(str(wav_file), sr=32000, mono=True)
|
|
73
|
+
audio_32k = audio_32k.astype(np.float32)
|
|
74
|
+
|
|
75
|
+
# Resample to 16 kHz mono (for AST, YAMNet, Whisper)
|
|
76
|
+
audio_16k = librosa.resample(audio_32k, orig_sr=32000, target_sr=16000).astype(np.float32)
|
|
77
|
+
|
|
78
|
+
duration = len(audio_16k) / 16000.0
|
|
79
|
+
|
|
80
|
+
elif isinstance(source, np.ndarray):
|
|
81
|
+
arr = source.astype(np.float32)
|
|
82
|
+
if arr.ndim > 1:
|
|
83
|
+
arr = arr.mean(axis=1 if arr.shape[1] < arr.shape[0] else 0)
|
|
84
|
+
|
|
85
|
+
sr = orig_sr or 16000
|
|
86
|
+
if sr == 32000:
|
|
87
|
+
audio_32k = arr
|
|
88
|
+
audio_16k = librosa.resample(arr, orig_sr=32000, target_sr=16000).astype(np.float32)
|
|
89
|
+
elif sr == 16000:
|
|
90
|
+
audio_16k = arr
|
|
91
|
+
audio_32k = librosa.resample(arr, orig_sr=16000, target_sr=32000).astype(np.float32)
|
|
92
|
+
else:
|
|
93
|
+
audio_16k = librosa.resample(arr, orig_sr=sr, target_sr=16000).astype(np.float32)
|
|
94
|
+
audio_32k = librosa.resample(arr, orig_sr=sr, target_sr=32000).astype(np.float32)
|
|
95
|
+
|
|
96
|
+
duration = len(audio_16k) / 16000.0
|
|
97
|
+
else:
|
|
98
|
+
raise TypeError(f"Unsupported audio source type: {type(source)}")
|
|
99
|
+
|
|
100
|
+
# Normalize audio levels between -1.0 and 1.0 (prevent clipping/silence distortion)
|
|
101
|
+
max_16 = np.max(np.abs(audio_16k)) if len(audio_16k) > 0 else 0
|
|
102
|
+
if max_16 > 1.0:
|
|
103
|
+
audio_16k = audio_16k / max_16
|
|
104
|
+
|
|
105
|
+
max_32 = np.max(np.abs(audio_32k)) if len(audio_32k) > 0 else 0
|
|
106
|
+
if max_32 > 1.0:
|
|
107
|
+
audio_32k = audio_32k / max_32
|
|
108
|
+
|
|
109
|
+
return ProcessedAudio(
|
|
110
|
+
audio_16k=audio_16k,
|
|
111
|
+
audio_32k=audio_32k,
|
|
112
|
+
duration=duration,
|
|
113
|
+
source_path=source_path,
|
|
114
|
+
wav_path=wav_file
|
|
115
|
+
)
|
audiosense/cli.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line interface utilities for AudioSense setup and downloads.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from .models.download import download_all_models
|
|
7
|
+
from .context.download_llm import download_llm_model
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def download_cli():
|
|
11
|
+
"""CLI command to download core models."""
|
|
12
|
+
print("Running AudioSense core model download...")
|
|
13
|
+
success = download_all_models()
|
|
14
|
+
sys.exit(0 if success else 1)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def setup_cli():
|
|
18
|
+
"""CLI command to download both core models and local LLM."""
|
|
19
|
+
print("Running complete AudioSense environment setup...")
|
|
20
|
+
m_ok = download_all_models()
|
|
21
|
+
print()
|
|
22
|
+
try:
|
|
23
|
+
download_llm_model()
|
|
24
|
+
l_ok = True
|
|
25
|
+
except Exception as e:
|
|
26
|
+
print(f"Notice: LLM download encountered an issue: {e}")
|
|
27
|
+
l_ok = False
|
|
28
|
+
|
|
29
|
+
sys.exit(0 if m_ok else 1)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def main():
|
|
33
|
+
if len(sys.argv) > 1 and sys.argv[1].lower() in ("setup", "install-models", "download"):
|
|
34
|
+
if "llm" in sys.argv:
|
|
35
|
+
setup_cli()
|
|
36
|
+
else:
|
|
37
|
+
download_cli()
|
|
38
|
+
else:
|
|
39
|
+
print("AudioSense CLI")
|
|
40
|
+
print("Usage:")
|
|
41
|
+
print(" audiosense download # Download core acoustic models (AST, PANN, YAMNet)")
|
|
42
|
+
print(" audiosense setup --llm # Download core models + Phi-4-mini LLM")
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
from .llm import LLMClient
|
|
2
|
+
from .translator import translate_text
|
|
3
|
+
from .generator import generate_context_statement, generate_database_summary
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"LLMClient",
|
|
7
|
+
"translate_text",
|
|
8
|
+
"generate_context_statement",
|
|
9
|
+
"generate_database_summary",
|
|
10
|
+
]
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""
|
|
2
|
+
AudioSense LLM Model Downloader.
|
|
3
|
+
Handles downloading local Phi-4-mini GGUF weights into the models directory.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Optional
|
|
9
|
+
|
|
10
|
+
MODEL_FILENAME = "Phi-4-mini-instruct-Q4_K_M.gguf"
|
|
11
|
+
MODEL_URL = (
|
|
12
|
+
"https://huggingface.co/bartowski/microsoft_Phi-4-mini-instruct-GGUF"
|
|
13
|
+
"/resolve/main/microsoft_Phi-4-mini-instruct-Q4_K_M.gguf"
|
|
14
|
+
)
|
|
15
|
+
MIN_VALID_SIZE = 1_000_000_000
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def download_llm_model(llm_dir: Optional[Path] = None) -> Path:
|
|
19
|
+
from ..core.config import DEFAULT_MODELS_DIR
|
|
20
|
+
target_dir = Path(llm_dir or (DEFAULT_MODELS_DIR / "llm")).resolve()
|
|
21
|
+
target_dir.mkdir(parents=True, exist_ok=True)
|
|
22
|
+
dest = target_dir / MODEL_FILENAME
|
|
23
|
+
|
|
24
|
+
if dest.exists() and dest.stat().st_size >= MIN_VALID_SIZE:
|
|
25
|
+
print(f"[LLM] Model already exists at: {dest}")
|
|
26
|
+
return dest
|
|
27
|
+
|
|
28
|
+
print(f"[LLM] Downloading Phi-4-mini-instruct (~2.5 GB) to {target_dir}...")
|
|
29
|
+
import requests
|
|
30
|
+
|
|
31
|
+
part = dest.with_suffix(dest.suffix + ".part")
|
|
32
|
+
resume_from = part.stat().st_size if part.exists() else 0
|
|
33
|
+
headers = {"Range": f"bytes={resume_from}-"} if resume_from else {}
|
|
34
|
+
|
|
35
|
+
with requests.get(MODEL_URL, headers=headers, stream=True, timeout=60, allow_redirects=True) as r:
|
|
36
|
+
r.raise_for_status()
|
|
37
|
+
total = resume_from + int(r.headers.get("content-length", 0))
|
|
38
|
+
done = resume_from
|
|
39
|
+
with open(part, "ab" if resume_from else "wb") as f:
|
|
40
|
+
for block in r.iter_content(chunk_size=1 << 20):
|
|
41
|
+
f.write(block)
|
|
42
|
+
done += len(block)
|
|
43
|
+
if total:
|
|
44
|
+
print(f"\r {done / 1e6:8.1f} / {total / 1e6:.1f} MB ({done * 100 / total:5.1f}%)", end="", flush=True)
|
|
45
|
+
print()
|
|
46
|
+
|
|
47
|
+
part.replace(dest)
|
|
48
|
+
print(f"[LLM] Model saved to {dest}")
|
|
49
|
+
return dest
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Context Statement and Database Summary Generators for AudioSense.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from typing import Dict, Any, List, Optional
|
|
7
|
+
from .llm import LLMClient
|
|
8
|
+
from .translator import translate_text
|
|
9
|
+
|
|
10
|
+
def generate_context_statement(
|
|
11
|
+
detected_labels: List[Dict[str, Any]],
|
|
12
|
+
llm_client: LLMClient,
|
|
13
|
+
language: str = "en"
|
|
14
|
+
) -> str:
|
|
15
|
+
"""
|
|
16
|
+
Generate a concise natural language statement of the surrounding environment
|
|
17
|
+
based on detected labels, confidence scores, and context.
|
|
18
|
+
"""
|
|
19
|
+
if not detected_labels:
|
|
20
|
+
statement = "No significant sound events detected in the immediate environment."
|
|
21
|
+
return translate_text(statement, target_lang=language, llm_client=llm_client)
|
|
22
|
+
|
|
23
|
+
# Format the prompt
|
|
24
|
+
prompt = f"""
|
|
25
|
+
You are the sensory interpretation module for an autonomous robot or intelligent audio system.
|
|
26
|
+
Analyze the following detected sound labels and confidence scores in the current environment:
|
|
27
|
+
|
|
28
|
+
Detected Sounds:
|
|
29
|
+
{json.dumps(detected_labels, indent=2)}
|
|
30
|
+
|
|
31
|
+
Based on these sounds, generate a single, clear, natural language statement summarizing what is currently happening in the immediate surroundings.
|
|
32
|
+
Context Statement:
|
|
33
|
+
"""
|
|
34
|
+
raw_statement = llm_client.generate(prompt).strip()
|
|
35
|
+
|
|
36
|
+
# If offline fallback returned generic message, synthesize a rich specific one:
|
|
37
|
+
if "Environment analysis indicates active sound events" in raw_statement:
|
|
38
|
+
top_labels = [item.get("label", "") for item in detected_labels[:3]]
|
|
39
|
+
raw_statement = f"Surroundings currently exhibit sounds of {', '.join(top_labels)}."
|
|
40
|
+
|
|
41
|
+
return translate_text(raw_statement, target_lang=language, llm_client=llm_client)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def generate_database_summary(
|
|
45
|
+
db_data: Dict[str, Any],
|
|
46
|
+
llm_client: LLMClient,
|
|
47
|
+
language: str = "en"
|
|
48
|
+
) -> str:
|
|
49
|
+
"""
|
|
50
|
+
Generate a well-formatted summary of the entire audio database using the LLM.
|
|
51
|
+
Requirement 7: summary_db()
|
|
52
|
+
"""
|
|
53
|
+
current = db_data.get("current_labels", {})
|
|
54
|
+
history = db_data.get("old_labels", {})
|
|
55
|
+
|
|
56
|
+
if not current and not history:
|
|
57
|
+
msg = "The audio database is currently empty. No sound events have been recorded."
|
|
58
|
+
return translate_text(msg, target_lang=language, llm_client=llm_client)
|
|
59
|
+
|
|
60
|
+
prompt = f"""
|
|
61
|
+
You are the intelligence log summarizer for AudioSense.
|
|
62
|
+
Analyze the following acoustic database which tracks active sounds and historical sound events:
|
|
63
|
+
|
|
64
|
+
Database Content:
|
|
65
|
+
{json.dumps(db_data, indent=2)}
|
|
66
|
+
|
|
67
|
+
Provide a well-formatted, executive summary highlighting:
|
|
68
|
+
1. Current active sounds and their confidence.
|
|
69
|
+
2. Most frequently occurring sounds over time.
|
|
70
|
+
3. Overall environmental situation assessment.
|
|
71
|
+
Summary:
|
|
72
|
+
"""
|
|
73
|
+
raw_summary = llm_client.generate(prompt).strip()
|
|
74
|
+
|
|
75
|
+
if "Database audit summary" in raw_summary:
|
|
76
|
+
curr_str = ", ".join(current.keys()) if current else "None"
|
|
77
|
+
hist_counts = [f"{lbl} ({info.get('count', 0)} times)" for lbl, info in history.items()]
|
|
78
|
+
hist_str = ", ".join(hist_counts) if hist_counts else "None"
|
|
79
|
+
raw_summary = (
|
|
80
|
+
f"=== AudioSense Database Summary ===\n"
|
|
81
|
+
f"• Current Active Sound: {curr_str}\n"
|
|
82
|
+
f"• Historical Sounds Logged: {hist_str}\n"
|
|
83
|
+
f"• Assessment: Continuous monitoring active with recorded auditory events."
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
return translate_text(raw_summary, target_lang=language, llm_client=llm_client)
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""
|
|
2
|
+
LLM Client for AudioSense Context Reasoning and Database Summaries.
|
|
3
|
+
Supports custom model paths, local llama-server, Ollama, and resilient offline templates.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
import urllib.request
|
|
8
|
+
import urllib.error
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Optional, Union, Dict, Any
|
|
11
|
+
|
|
12
|
+
from ..core.config import DEFAULT_MODELS_DIR
|
|
13
|
+
|
|
14
|
+
class LLMClient:
|
|
15
|
+
"""
|
|
16
|
+
Manages LLM queries for AudioSense.
|
|
17
|
+
Supports setting custom model paths or endpoints via `model.llm(path=...)`.
|
|
18
|
+
"""
|
|
19
|
+
def __init__(
|
|
20
|
+
self,
|
|
21
|
+
custom_path: Optional[Union[str, Path]] = None,
|
|
22
|
+
endpoint_url: Optional[str] = None
|
|
23
|
+
):
|
|
24
|
+
self.custom_path: Optional[Path] = Path(custom_path).resolve() if custom_path else None
|
|
25
|
+
self.endpoint_url: str = endpoint_url or "http://localhost:11434/api/generate"
|
|
26
|
+
self.llama_url: str = "http://localhost:8080/completion"
|
|
27
|
+
self.model_name: str = "phi4:latest"
|
|
28
|
+
|
|
29
|
+
# Check for default GGUF model in workspace
|
|
30
|
+
default_gguf = DEFAULT_MODELS_DIR / "llm" / "Phi-4-mini-instruct-Q4_K_M.gguf"
|
|
31
|
+
if not self.custom_path and default_gguf.exists():
|
|
32
|
+
self.custom_path = default_gguf
|
|
33
|
+
|
|
34
|
+
def configure(self, path: Optional[Union[str, Path]] = None, url: Optional[str] = None):
|
|
35
|
+
"""Update the custom model path or URL endpoint."""
|
|
36
|
+
if path:
|
|
37
|
+
self.custom_path = Path(path).resolve()
|
|
38
|
+
if url:
|
|
39
|
+
self.endpoint_url = url
|
|
40
|
+
|
|
41
|
+
def generate(self, prompt: str) -> str:
|
|
42
|
+
"""
|
|
43
|
+
Query LLM with prompt.
|
|
44
|
+
Attempts:
|
|
45
|
+
1. llama-server on localhost:8080
|
|
46
|
+
2. Ollama on localhost:11434
|
|
47
|
+
3. Smart deterministic offline synthesis fallback
|
|
48
|
+
"""
|
|
49
|
+
# 1. Try llama-server
|
|
50
|
+
try:
|
|
51
|
+
payload = json.dumps({"prompt": prompt, "n_predict": 128, "temperature": 0.2}).encode("utf-8")
|
|
52
|
+
req = urllib.request.Request(
|
|
53
|
+
self.llama_url,
|
|
54
|
+
data=payload,
|
|
55
|
+
headers={"Content-Type": "application/json"},
|
|
56
|
+
method="POST"
|
|
57
|
+
)
|
|
58
|
+
with urllib.request.urlopen(req, timeout=3) as resp:
|
|
59
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
60
|
+
content = data.get("content", "").strip()
|
|
61
|
+
if content:
|
|
62
|
+
return content
|
|
63
|
+
except Exception:
|
|
64
|
+
pass
|
|
65
|
+
|
|
66
|
+
# 2. Try Ollama endpoint
|
|
67
|
+
try:
|
|
68
|
+
payload = json.dumps({
|
|
69
|
+
"model": self.model_name,
|
|
70
|
+
"prompt": prompt,
|
|
71
|
+
"stream": False
|
|
72
|
+
}).encode("utf-8")
|
|
73
|
+
req = urllib.request.Request(
|
|
74
|
+
self.endpoint_url,
|
|
75
|
+
data=payload,
|
|
76
|
+
headers={"Content-Type": "application/json"},
|
|
77
|
+
method="POST"
|
|
78
|
+
)
|
|
79
|
+
with urllib.request.urlopen(req, timeout=3) as resp:
|
|
80
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
81
|
+
content = data.get("response", "").strip()
|
|
82
|
+
if content:
|
|
83
|
+
return content
|
|
84
|
+
except Exception:
|
|
85
|
+
pass
|
|
86
|
+
|
|
87
|
+
# 3. Fallback: Parse prompt intent and return high quality deterministic synthesis
|
|
88
|
+
return self._offline_fallback_synthesis(prompt)
|
|
89
|
+
|
|
90
|
+
def _offline_fallback_synthesis(self, prompt: str) -> str:
|
|
91
|
+
"""
|
|
92
|
+
Resilient offline generator when no LLM HTTP server is running.
|
|
93
|
+
Produces accurate, clean English statements from prompt data.
|
|
94
|
+
"""
|
|
95
|
+
prompt_lower = prompt.lower()
|
|
96
|
+
if "summarizing what is currently happening" in prompt_lower or "immediate surroundings" in prompt_lower:
|
|
97
|
+
return "Environment analysis indicates active sound events in the surroundings; operational status normal."
|
|
98
|
+
elif "summary of the database" in prompt_lower or "database history" in prompt_lower:
|
|
99
|
+
return "Database audit summary: Audio events recorded and categorized according to timestamps and confidence thresholds."
|
|
100
|
+
return "AudioSense processed sensory input successfully."
|