audiosense 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- audiosense-1.0.0/LICENSE +21 -0
- audiosense-1.0.0/MANIFEST.in +4 -0
- audiosense-1.0.0/PKG-INFO +334 -0
- audiosense-1.0.0/README.md +288 -0
- audiosense-1.0.0/audiosense/__init__.py +78 -0
- audiosense-1.0.0/audiosense/audio/__init__.py +12 -0
- audiosense-1.0.0/audiosense/audio/conversion.py +103 -0
- audiosense-1.0.0/audiosense/audio/io.py +37 -0
- audiosense-1.0.0/audiosense/audio/preprocess.py +115 -0
- audiosense-1.0.0/audiosense/cli.py +42 -0
- audiosense-1.0.0/audiosense/context/__init__.py +10 -0
- audiosense-1.0.0/audiosense/context/download_llm.py +49 -0
- audiosense-1.0.0/audiosense/context/generator.py +86 -0
- audiosense-1.0.0/audiosense/context/llm.py +100 -0
- audiosense-1.0.0/audiosense/context/translator.py +73 -0
- audiosense-1.0.0/audiosense/core/__init__.py +3 -0
- audiosense-1.0.0/audiosense/core/config.py +69 -0
- audiosense-1.0.0/audiosense/core/engine.py +228 -0
- audiosense-1.0.0/audiosense/core/silence.py +124 -0
- audiosense-1.0.0/audiosense/label.py +25 -0
- audiosense-1.0.0/audiosense/labels/__init__.py +16 -0
- audiosense-1.0.0/audiosense/labels/all.py +63 -0
- audiosense-1.0.0/audiosense/labels/categories.py +42 -0
- audiosense-1.0.0/audiosense/labels/selected.py +56 -0
- audiosense-1.0.0/audiosense/labels/taxonomy.py +93 -0
- audiosense-1.0.0/audiosense/labels/top.py +70 -0
- audiosense-1.0.0/audiosense/live/__init__.py +3 -0
- audiosense-1.0.0/audiosense/live/loop.py +103 -0
- audiosense-1.0.0/audiosense/models/__init__.py +13 -0
- audiosense-1.0.0/audiosense/models/ast_model.py +79 -0
- audiosense-1.0.0/audiosense/models/download.py +112 -0
- audiosense-1.0.0/audiosense/models/manager.py +45 -0
- audiosense-1.0.0/audiosense/models/pann_model.py +60 -0
- audiosense-1.0.0/audiosense/models/whisper_model.py +48 -0
- audiosense-1.0.0/audiosense/models/yamnet_model.py +75 -0
- audiosense-1.0.0/audiosense/storage/__init__.py +3 -0
- audiosense-1.0.0/audiosense/storage/database.py +136 -0
- audiosense-1.0.0/audiosense/switcher/__init__.py +4 -0
- audiosense-1.0.0/audiosense/switcher/router.py +128 -0
- audiosense-1.0.0/audiosense/switcher/vad.py +40 -0
- audiosense-1.0.0/audiosense.egg-info/PKG-INFO +334 -0
- audiosense-1.0.0/audiosense.egg-info/SOURCES.txt +48 -0
- audiosense-1.0.0/audiosense.egg-info/dependency_links.txt +1 -0
- audiosense-1.0.0/audiosense.egg-info/entry_points.txt +4 -0
- audiosense-1.0.0/audiosense.egg-info/requires.txt +21 -0
- audiosense-1.0.0/audiosense.egg-info/top_level.txt +1 -0
- audiosense-1.0.0/pyproject.toml +78 -0
- audiosense-1.0.0/requirements.txt +28 -0
- audiosense-1.0.0/setup.cfg +4 -0
- audiosense-1.0.0/setup.py +3 -0
audiosense-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 AudioSense Team
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: audiosense
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Audio intelligence framework for machines, robots and modern applications
|
|
5
|
+
Author-email: Mohan <mohanevs@users.noreply.github.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/mohanevs/AudioSense
|
|
8
|
+
Project-URL: Repository, https://github.com/mohanevs/AudioSense.git
|
|
9
|
+
Project-URL: Issues, https://github.com/mohanevs/AudioSense/issues
|
|
10
|
+
Keywords: audio,sound-classification,speech-recognition,audio-intelligence,robotics,panns,ast,yamnet,whisper,vad
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Analysis
|
|
21
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: numpy
|
|
26
|
+
Requires-Dist: scipy
|
|
27
|
+
Requires-Dist: soundfile
|
|
28
|
+
Requires-Dist: sounddevice
|
|
29
|
+
Requires-Dist: librosa
|
|
30
|
+
Requires-Dist: moviepy
|
|
31
|
+
Requires-Dist: torch
|
|
32
|
+
Requires-Dist: torchaudio
|
|
33
|
+
Requires-Dist: transformers
|
|
34
|
+
Requires-Dist: panns_inference
|
|
35
|
+
Requires-Dist: tensorflow
|
|
36
|
+
Requires-Dist: tensorflow-hub
|
|
37
|
+
Requires-Dist: kagglehub
|
|
38
|
+
Requires-Dist: silero-vad
|
|
39
|
+
Requires-Dist: faster-whisper
|
|
40
|
+
Requires-Dist: requests
|
|
41
|
+
Requires-Dist: setuptools>=70.0.0
|
|
42
|
+
Provides-Extra: server
|
|
43
|
+
Requires-Dist: fastapi; extra == "server"
|
|
44
|
+
Requires-Dist: uvicorn; extra == "server"
|
|
45
|
+
Dynamic: license-file
|
|
46
|
+
|
|
47
|
+
<div align="center">
|
|
48
|
+
|
|
49
|
+
# 🎙️ AudioSense SDK
|
|
50
|
+
### *Audio Intelligence for Machines, Robots, and Modern Applications*
|
|
51
|
+
|
|
52
|
+
[](https://python.org)
|
|
53
|
+
[](https://github.com)
|
|
54
|
+
[](https://github.com)
|
|
55
|
+
[](https://github.com)
|
|
56
|
+
[](https://github.com)
|
|
57
|
+
[](LICENSE)
|
|
58
|
+
|
|
59
|
+
<p align="center">
|
|
60
|
+
<a href="#-overview">Overview</a> •
|
|
61
|
+
<a href="#-key-features">Key Features</a> •
|
|
62
|
+
<a href="#-installation">Installation</a> •
|
|
63
|
+
<a href="#-quickstart">Quickstart</a> •
|
|
64
|
+
<a href="#-core-api-guide">Core API</a> •
|
|
65
|
+
<a href="#-7-layer-architecture">Architecture</a> •
|
|
66
|
+
<a href="#-live-streaming">Live Streaming</a>
|
|
67
|
+
</p>
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
</div>
|
|
72
|
+
|
|
73
|
+
## 📌 Overview
|
|
74
|
+
|
|
75
|
+
Computers see and read well, but most still hear poorly. A security camera sees a broken window; a microphone beside it usually outputs nothing but a raw waveform.
|
|
76
|
+
|
|
77
|
+
**AudioSense** bridges that gap. It is a comprehensive audio intelligence framework that transforms raw acoustic signals into:
|
|
78
|
+
1. **Events** — *What happened and when* (sound classification & detection).
|
|
79
|
+
2. **Speech Transcripts** — *What was spoken* (automatic VAD gating and transcription).
|
|
80
|
+
3. **Insights & Context** — *What is happening in the environment* (LLM-driven situational reasoning).
|
|
81
|
+
|
|
82
|
+
Designed with a clean, tiered API: add intelligent hearing to your application or robot with **three lines of Python**, while retaining deep control over every layer underneath.
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from audiosense import AudioSense
|
|
86
|
+
|
|
87
|
+
model = AudioSense()
|
|
88
|
+
result = model.predict("environment.wav")
|
|
89
|
+
print(result)
|
|
90
|
+
# Output: Surroundings currently exhibit sounds of Water tap, faucet, Water, Sink (filling or washing).
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## ✨ Key Features
|
|
96
|
+
|
|
97
|
+
| Capability | Description |
|
|
98
|
+
|---|---|
|
|
99
|
+
| **Multi-Model Ensemble** | Combines **AST** (Audio Spectrogram Transformer), **PANNs** (CNN14), and **YAMNet** in a single pass with calibrated consensus scoring. |
|
|
100
|
+
| **Intelligent Routing (Switcher)** | Automatically routes human speech to **Faster-Whisper** transcription and environmental sounds to **Context Intelligence / RAG**. |
|
|
101
|
+
| **Universal Audio Converter** | Ingests `.mp3`, `.flac`, `.ogg`, `.m4a`, `.aac`, `.opus`, `.webm`, `.mp4` and converts to pristine `.wav`. |
|
|
102
|
+
| **Dual Sample-Rate Conditioning** | Internal preprocessing pipeline delivers synchronized 32 kHz (for PANN) and 16 kHz (for AST/YAMNet/Whisper) audio arrays. |
|
|
103
|
+
| **Context Reasoning & LLM** | Synthesizes high-level environmental situation reports via local LLMs (Phi-4, llama-server, Ollama) or custom endpoints. |
|
|
104
|
+
| **Multilingual Translation** | Live translation of context statements and speech transcripts into Hindi (`'hin'`), Spanish, French, German, etc. |
|
|
105
|
+
| **Acoustic Database & Audit** | Built-in JSON event store tracking occurrence counts, timestamps, and confidence histories with `view_db()` and `summary_db()`. |
|
|
106
|
+
| **Real-time Live Mic Loop** | Continuous laptop microphone listening loop (`audiosense.loop()`) with VAD activity filtering and instant callbacks. |
|
|
107
|
+
| **Zero Log Pollution** | Completely silences noisy TensorFlow, oneDNN, and progress-bar outputs, giving clean, predictable returns. |
|
|
108
|
+
|
|
109
|
+
---
|
|
110
|
+
|
|
111
|
+
## 🚀 Installation
|
|
112
|
+
|
|
113
|
+
### 1. Clone & Set Up Environment
|
|
114
|
+
```bash
|
|
115
|
+
git clone https://github.com/your-org/AudioSense.git
|
|
116
|
+
cd AudioSense
|
|
117
|
+
python -m venv env
|
|
118
|
+
# Windows
|
|
119
|
+
.\env\Scripts\activate
|
|
120
|
+
# Linux/macOS
|
|
121
|
+
source env/bin/activate
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### 2. Install Dependencies
|
|
125
|
+
```bash
|
|
126
|
+
pip install -r requirements.txt
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
### 3. Acquire Pre-Trained Weights
|
|
130
|
+
AudioSense dynamically downloads and verifies model weights:
|
|
131
|
+
```bash
|
|
132
|
+
# Downloads AST, PANN (CNN14), and YAMNet
|
|
133
|
+
python load_model.py
|
|
134
|
+
|
|
135
|
+
# Optional: Downloads local Phi-4-mini LLM (~2.5 GB)
|
|
136
|
+
python load_llm.py
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
---
|
|
140
|
+
|
|
141
|
+
## ⚡ Quickstart
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
from audiosense import AudioSense, view_db
|
|
145
|
+
|
|
146
|
+
# 1. Initialize engine
|
|
147
|
+
model = AudioSense(device="auto", language="en")
|
|
148
|
+
|
|
149
|
+
# 2. Predict on an audio file
|
|
150
|
+
result = model.predict("audio/audio.wav")
|
|
151
|
+
print("Detected Environment:", result.text)
|
|
152
|
+
print("Top Labels:", result.labels)
|
|
153
|
+
|
|
154
|
+
# 3. View persistent sound history
|
|
155
|
+
view_db()
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
160
|
+
## 📖 Core API Guide
|
|
161
|
+
|
|
162
|
+
### 1. Unified Prediction & Routing
|
|
163
|
+
The switcher automatically checks whether incoming sound is human speech or environmental:
|
|
164
|
+
```python
|
|
165
|
+
res = model.predict("audio/sample.wav")
|
|
166
|
+
|
|
167
|
+
if res.route == "transcription":
|
|
168
|
+
print("Speech detected:", res.transcription)
|
|
169
|
+
else:
|
|
170
|
+
print("Context statement:", res.context)
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### 2. Functional Label APIs
|
|
174
|
+
Direct functional access to taxonomy and model predictions:
|
|
175
|
+
```python
|
|
176
|
+
from audiosense.label import get_label, get_all, get_categories, top_k
|
|
177
|
+
|
|
178
|
+
# Primary label
|
|
179
|
+
primary = get_label("audio/audio.wav")
|
|
180
|
+
|
|
181
|
+
# Top 5 consensus predictions
|
|
182
|
+
rankings = top_k("audio/audio.wav", k=5)
|
|
183
|
+
|
|
184
|
+
# Inspect categories
|
|
185
|
+
cats = get_categories(["Dog", "Car horn", "Speech"])
|
|
186
|
+
# {'Dog': 'Animal sounds', 'Car horn': 'Sounds of things', 'Speech': 'Human sounds'}
|
|
187
|
+
|
|
188
|
+
# Full predictions across all models
|
|
189
|
+
all_preds = get_all("audio/audio.wav", threshold=0.1)
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
### 3. Switcher Tuning
|
|
193
|
+
Customize routing rules dynamically for your specific application:
|
|
194
|
+
```python
|
|
195
|
+
# Move bird chirping to human/speech indicators for this instance
|
|
196
|
+
model.switcher("Birds_chirping", "HUMAN")
|
|
197
|
+
|
|
198
|
+
# Or move a sound to environmental indicators
|
|
199
|
+
model.switcher("Shout", "ENV")
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
### 4. Custom LLM Integration
|
|
203
|
+
Plug in your own local or remote language model:
|
|
204
|
+
```python
|
|
205
|
+
# Custom GGUF file path
|
|
206
|
+
model.llm(path="C:/models/custom_model.gguf")
|
|
207
|
+
|
|
208
|
+
# Or custom Ollama / llama-server HTTP endpoint
|
|
209
|
+
model.llm(url="http://localhost:11434/api/generate")
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
### 5. Multilingual Output
|
|
213
|
+
Output context statements and transcripts in your desired language:
|
|
214
|
+
```python
|
|
215
|
+
model.translation('hin') # Output in Hindi
|
|
216
|
+
res = model.predict("audio/audio.wav")
|
|
217
|
+
print(res.text) # e.g., 'आस-पास पानी के नल और सिंक की आवाजें आ रही हैं।'
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
### 6. Database Inspection & Summaries
|
|
221
|
+
```python
|
|
222
|
+
# Print formatted database table
|
|
223
|
+
model.view_db()
|
|
224
|
+
|
|
225
|
+
# Generate executive summary of auditory history using LLM
|
|
226
|
+
model.summary_db()
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
### 7. Universal Audio Conversion
|
|
230
|
+
```python
|
|
231
|
+
from audiosense import convert_to_wav
|
|
232
|
+
|
|
233
|
+
# Accepts MP3, FLAC, M4A, OGG, WEBM, MP4, etc.
|
|
234
|
+
wav_path = convert_to_wav("recording.m4a", output_path="recording.wav")
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
---
|
|
238
|
+
|
|
239
|
+
## 🎙️ Live Microphone & Robotics
|
|
240
|
+
|
|
241
|
+
Continuous, non-blocking real-time listening through the system microphone:
|
|
242
|
+
|
|
243
|
+
```python
|
|
244
|
+
from audiosense import AudioSense
|
|
245
|
+
|
|
246
|
+
model = AudioSense()
|
|
247
|
+
|
|
248
|
+
# Start continuous listening loop (Press Ctrl+C to stop)
|
|
249
|
+
model.loop(chunk_duration=3.0)
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
You can also provide a callback for robot reactive control:
|
|
253
|
+
```python
|
|
254
|
+
def on_sound(res):
|
|
255
|
+
if "Siren" in [lbl for lbl, _ in res.labels]:
|
|
256
|
+
print("🚨 Warning: Emergency siren heard! Halting robot.")
|
|
257
|
+
|
|
258
|
+
model.loop(chunk_duration=2.0, callback=on_sound)
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
---
|
|
262
|
+
|
|
263
|
+
## 🏛️ 7-Layer Architecture
|
|
264
|
+
|
|
265
|
+
```
|
|
266
|
+
┌─────────────────────────────────────────────────────────────┐
|
|
267
|
+
│ 7. Application & Robot Interface │
|
|
268
|
+
│ Python API (AudioSense) • Callbacks • Controller Maps │
|
|
269
|
+
├─────────────────────────────────────────────────────────────┤
|
|
270
|
+
│ 6. Context Intelligence Layer │
|
|
271
|
+
│ Situational Reasoning • LLM Summaries • Multilingual │
|
|
272
|
+
├─────────────────────────────────────────────────────────────┤
|
|
273
|
+
│ 5. Ensemble Decision Layer & Switcher │
|
|
274
|
+
│ Weighted Consensus • Speech vs. Environmental Router │
|
|
275
|
+
├─────────────────────────────────────────────────────────────┤
|
|
276
|
+
│ 4. AI Model Layer │
|
|
277
|
+
│ AST (Transformer) • PANN (CNN14) • YAMNet • Whisper │
|
|
278
|
+
├─────────────────────────────────────────────────────────────┤
|
|
279
|
+
│ 3. Feature Extraction Layer │
|
|
280
|
+
│ Log-mel Spectrograms • Filterbanks • Embeddings │
|
|
281
|
+
├─────────────────────────────────────────────────────────────┤
|
|
282
|
+
│ 2. Audio Processing Layer │
|
|
283
|
+
│ Resampling (16 kHz / 32 kHz) • VAD Gating • Normalization│
|
|
284
|
+
├─────────────────────────────────────────────────────────────┤
|
|
285
|
+
│ 1. Audio Input Layer │
|
|
286
|
+
│ File Decoders (conversion.py) • Microphones • Streams │
|
|
287
|
+
└─────────────────────────────────────────────────────────────┘
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
---
|
|
291
|
+
|
|
292
|
+
## 📦 Project Structure
|
|
293
|
+
|
|
294
|
+
```
|
|
295
|
+
AudioSense/
|
|
296
|
+
├── audiosense/
|
|
297
|
+
│ ├── __init__.py # Public framework exports
|
|
298
|
+
│ ├── core/
|
|
299
|
+
│ │ ├── engine.py # AudioSense primary engine
|
|
300
|
+
│ │ ├── config.py # Dynamic relative path resolver
|
|
301
|
+
│ │ └── silence.py # Log & warning suppression engine
|
|
302
|
+
│ ├── audio/
|
|
303
|
+
│ │ ├── conversion.py # Universal audio converter (mp3/flac/m4a -> wav)
|
|
304
|
+
│ │ ├── preprocess.py # Dual-rate (16k/32k) conditioning
|
|
305
|
+
│ │ └── io.py # Audio reader/writer utilities
|
|
306
|
+
│ ├── models/
|
|
307
|
+
│ │ ├── ast_model.py # AST adapter
|
|
308
|
+
│ │ ├── pann_model.py # PANN (CNN14) adapter
|
|
309
|
+
│ │ ├── yamnet_model.py # YAMNet adapter
|
|
310
|
+
│ │ ├── whisper_model.py # Faster-Whisper adapter
|
|
311
|
+
│ │ └── manager.py # Lazy model coordinator
|
|
312
|
+
│ ├── label.py & labels/ # Functional label APIs & AudioSet taxonomy
|
|
313
|
+
│ ├── switcher/
|
|
314
|
+
│ │ ├── router.py # Tunable Speech vs Environment Router
|
|
315
|
+
│ │ └── vad.py # Silero-VAD detector
|
|
316
|
+
│ ├── context/
|
|
317
|
+
│ │ ├── llm.py # LLM interface (Ollama / llama-server / custom)
|
|
318
|
+
│ │ ├── translator.py # Multi-language translation engine
|
|
319
|
+
│ │ └── generator.py # Context & database summarizers
|
|
320
|
+
│ ├── storage/
|
|
321
|
+
│ │ └── database.py # JSON database engine (view_db)
|
|
322
|
+
│ └── live/
|
|
323
|
+
│ └── loop.py # Real-time microphone listening loop
|
|
324
|
+
├── models/ # Local model weights cache
|
|
325
|
+
├── load_model.py # Model acquisition script
|
|
326
|
+
├── load_llm.py # LLM downloader script
|
|
327
|
+
├── requirements.txt # Python dependencies
|
|
328
|
+
└── README.md # Documentation
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
---
|
|
332
|
+
|
|
333
|
+
## 📄 License
|
|
334
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
# 🎙️ AudioSense SDK
|
|
4
|
+
### *Audio Intelligence for Machines, Robots, and Modern Applications*
|
|
5
|
+
|
|
6
|
+
[](https://python.org)
|
|
7
|
+
[](https://github.com)
|
|
8
|
+
[](https://github.com)
|
|
9
|
+
[](https://github.com)
|
|
10
|
+
[](https://github.com)
|
|
11
|
+
[](LICENSE)
|
|
12
|
+
|
|
13
|
+
<p align="center">
|
|
14
|
+
<a href="#-overview">Overview</a> •
|
|
15
|
+
<a href="#-key-features">Key Features</a> •
|
|
16
|
+
<a href="#-installation">Installation</a> •
|
|
17
|
+
<a href="#-quickstart">Quickstart</a> •
|
|
18
|
+
<a href="#-core-api-guide">Core API</a> •
|
|
19
|
+
<a href="#-7-layer-architecture">Architecture</a> •
|
|
20
|
+
<a href="#-live-streaming">Live Streaming</a>
|
|
21
|
+
</p>
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
</div>
|
|
26
|
+
|
|
27
|
+
## 📌 Overview
|
|
28
|
+
|
|
29
|
+
Computers see and read well, but most still hear poorly. A security camera sees a broken window; a microphone beside it usually outputs nothing but a raw waveform.
|
|
30
|
+
|
|
31
|
+
**AudioSense** bridges that gap. It is a comprehensive audio intelligence framework that transforms raw acoustic signals into:
|
|
32
|
+
1. **Events** — *What happened and when* (sound classification & detection).
|
|
33
|
+
2. **Speech Transcripts** — *What was spoken* (automatic VAD gating and transcription).
|
|
34
|
+
3. **Insights & Context** — *What is happening in the environment* (LLM-driven situational reasoning).
|
|
35
|
+
|
|
36
|
+
Designed with a clean, tiered API: add intelligent hearing to your application or robot with **three lines of Python**, while retaining deep control over every layer underneath.
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
from audiosense import AudioSense
|
|
40
|
+
|
|
41
|
+
model = AudioSense()
|
|
42
|
+
result = model.predict("environment.wav")
|
|
43
|
+
print(result)
|
|
44
|
+
# Output: Surroundings currently exhibit sounds of Water tap, faucet, Water, Sink (filling or washing).
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## ✨ Key Features
|
|
50
|
+
|
|
51
|
+
| Capability | Description |
|
|
52
|
+
|---|---|
|
|
53
|
+
| **Multi-Model Ensemble** | Combines **AST** (Audio Spectrogram Transformer), **PANNs** (CNN14), and **YAMNet** in a single pass with calibrated consensus scoring. |
|
|
54
|
+
| **Intelligent Routing (Switcher)** | Automatically routes human speech to **Faster-Whisper** transcription and environmental sounds to **Context Intelligence / RAG**. |
|
|
55
|
+
| **Universal Audio Converter** | Ingests `.mp3`, `.flac`, `.ogg`, `.m4a`, `.aac`, `.opus`, `.webm`, `.mp4` and converts to pristine `.wav`. |
|
|
56
|
+
| **Dual Sample-Rate Conditioning** | Internal preprocessing pipeline delivers synchronized 32 kHz (for PANN) and 16 kHz (for AST/YAMNet/Whisper) audio arrays. |
|
|
57
|
+
| **Context Reasoning & LLM** | Synthesizes high-level environmental situation reports via local LLMs (Phi-4, llama-server, Ollama) or custom endpoints. |
|
|
58
|
+
| **Multilingual Translation** | Live translation of context statements and speech transcripts into Hindi (`'hin'`), Spanish, French, German, etc. |
|
|
59
|
+
| **Acoustic Database & Audit** | Built-in JSON event store tracking occurrence counts, timestamps, and confidence histories with `view_db()` and `summary_db()`. |
|
|
60
|
+
| **Real-time Live Mic Loop** | Continuous laptop microphone listening loop (`audiosense.loop()`) with VAD activity filtering and instant callbacks. |
|
|
61
|
+
| **Zero Log Pollution** | Completely silences noisy TensorFlow, oneDNN, and progress-bar outputs, giving clean, predictable returns. |
|
|
62
|
+
|
|
63
|
+
---
|
|
64
|
+
|
|
65
|
+
## 🚀 Installation
|
|
66
|
+
|
|
67
|
+
### 1. Clone & Set Up Environment
|
|
68
|
+
```bash
|
|
69
|
+
git clone https://github.com/your-org/AudioSense.git
|
|
70
|
+
cd AudioSense
|
|
71
|
+
python -m venv env
|
|
72
|
+
# Windows
|
|
73
|
+
.\env\Scripts\activate
|
|
74
|
+
# Linux/macOS
|
|
75
|
+
source env/bin/activate
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
### 2. Install Dependencies
|
|
79
|
+
```bash
|
|
80
|
+
pip install -r requirements.txt
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### 3. Acquire Pre-Trained Weights
|
|
84
|
+
AudioSense dynamically downloads and verifies model weights:
|
|
85
|
+
```bash
|
|
86
|
+
# Downloads AST, PANN (CNN14), and YAMNet
|
|
87
|
+
python load_model.py
|
|
88
|
+
|
|
89
|
+
# Optional: Downloads local Phi-4-mini LLM (~2.5 GB)
|
|
90
|
+
python load_llm.py
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## ⚡ Quickstart
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from audiosense import AudioSense, view_db
|
|
99
|
+
|
|
100
|
+
# 1. Initialize engine
|
|
101
|
+
model = AudioSense(device="auto", language="en")
|
|
102
|
+
|
|
103
|
+
# 2. Predict on an audio file
|
|
104
|
+
result = model.predict("audio/audio.wav")
|
|
105
|
+
print("Detected Environment:", result.text)
|
|
106
|
+
print("Top Labels:", result.labels)
|
|
107
|
+
|
|
108
|
+
# 3. View persistent sound history
|
|
109
|
+
view_db()
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
---
|
|
113
|
+
|
|
114
|
+
## 📖 Core API Guide
|
|
115
|
+
|
|
116
|
+
### 1. Unified Prediction & Routing
|
|
117
|
+
The switcher automatically checks whether incoming sound is human speech or environmental:
|
|
118
|
+
```python
|
|
119
|
+
res = model.predict("audio/sample.wav")
|
|
120
|
+
|
|
121
|
+
if res.route == "transcription":
|
|
122
|
+
print("Speech detected:", res.transcription)
|
|
123
|
+
else:
|
|
124
|
+
print("Context statement:", res.context)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
### 2. Functional Label APIs
|
|
128
|
+
Direct functional access to taxonomy and model predictions:
|
|
129
|
+
```python
|
|
130
|
+
from audiosense.label import get_label, get_all, get_categories, top_k
|
|
131
|
+
|
|
132
|
+
# Primary label
|
|
133
|
+
primary = get_label("audio/audio.wav")
|
|
134
|
+
|
|
135
|
+
# Top 5 consensus predictions
|
|
136
|
+
rankings = top_k("audio/audio.wav", k=5)
|
|
137
|
+
|
|
138
|
+
# Inspect categories
|
|
139
|
+
cats = get_categories(["Dog", "Car horn", "Speech"])
|
|
140
|
+
# {'Dog': 'Animal sounds', 'Car horn': 'Sounds of things', 'Speech': 'Human sounds'}
|
|
141
|
+
|
|
142
|
+
# Full predictions across all models
|
|
143
|
+
all_preds = get_all("audio/audio.wav", threshold=0.1)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
### 3. Switcher Tuning
|
|
147
|
+
Customize routing rules dynamically for your specific application:
|
|
148
|
+
```python
|
|
149
|
+
# Move bird chirping to human/speech indicators for this instance
|
|
150
|
+
model.switcher("Birds_chirping", "HUMAN")
|
|
151
|
+
|
|
152
|
+
# Or move a sound to environmental indicators
|
|
153
|
+
model.switcher("Shout", "ENV")
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### 4. Custom LLM Integration
|
|
157
|
+
Plug in your own local or remote language model:
|
|
158
|
+
```python
|
|
159
|
+
# Custom GGUF file path
|
|
160
|
+
model.llm(path="C:/models/custom_model.gguf")
|
|
161
|
+
|
|
162
|
+
# Or custom Ollama / llama-server HTTP endpoint
|
|
163
|
+
model.llm(url="http://localhost:11434/api/generate")
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
### 5. Multilingual Output
|
|
167
|
+
Output context statements and transcripts in your desired language:
|
|
168
|
+
```python
|
|
169
|
+
model.translation('hin') # Output in Hindi
|
|
170
|
+
res = model.predict("audio/audio.wav")
|
|
171
|
+
print(res.text) # e.g., 'आस-पास पानी के नल और सिंक की आवाजें आ रही हैं।'
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
### 6. Database Inspection & Summaries
|
|
175
|
+
```python
|
|
176
|
+
# Print formatted database table
|
|
177
|
+
model.view_db()
|
|
178
|
+
|
|
179
|
+
# Generate executive summary of auditory history using LLM
|
|
180
|
+
model.summary_db()
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
### 7. Universal Audio Conversion
|
|
184
|
+
```python
|
|
185
|
+
from audiosense import convert_to_wav
|
|
186
|
+
|
|
187
|
+
# Accepts MP3, FLAC, M4A, OGG, WEBM, MP4, etc.
|
|
188
|
+
wav_path = convert_to_wav("recording.m4a", output_path="recording.wav")
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
---
|
|
192
|
+
|
|
193
|
+
## 🎙️ Live Microphone & Robotics
|
|
194
|
+
|
|
195
|
+
Continuous, non-blocking real-time listening through the system microphone:
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
from audiosense import AudioSense
|
|
199
|
+
|
|
200
|
+
model = AudioSense()
|
|
201
|
+
|
|
202
|
+
# Start continuous listening loop (Press Ctrl+C to stop)
|
|
203
|
+
model.loop(chunk_duration=3.0)
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
You can also provide a callback for robot reactive control:
|
|
207
|
+
```python
|
|
208
|
+
def on_sound(res):
|
|
209
|
+
if "Siren" in [lbl for lbl, _ in res.labels]:
|
|
210
|
+
print("🚨 Warning: Emergency siren heard! Halting robot.")
|
|
211
|
+
|
|
212
|
+
model.loop(chunk_duration=2.0, callback=on_sound)
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
## 🏛️ 7-Layer Architecture
|
|
218
|
+
|
|
219
|
+
```
|
|
220
|
+
┌─────────────────────────────────────────────────────────────┐
|
|
221
|
+
│ 7. Application & Robot Interface │
|
|
222
|
+
│ Python API (AudioSense) • Callbacks • Controller Maps │
|
|
223
|
+
├─────────────────────────────────────────────────────────────┤
|
|
224
|
+
│ 6. Context Intelligence Layer │
|
|
225
|
+
│ Situational Reasoning • LLM Summaries • Multilingual │
|
|
226
|
+
├─────────────────────────────────────────────────────────────┤
|
|
227
|
+
│ 5. Ensemble Decision Layer & Switcher │
|
|
228
|
+
│ Weighted Consensus • Speech vs. Environmental Router │
|
|
229
|
+
├─────────────────────────────────────────────────────────────┤
|
|
230
|
+
│ 4. AI Model Layer │
|
|
231
|
+
│ AST (Transformer) • PANN (CNN14) • YAMNet • Whisper │
|
|
232
|
+
├─────────────────────────────────────────────────────────────┤
|
|
233
|
+
│ 3. Feature Extraction Layer │
|
|
234
|
+
│ Log-mel Spectrograms • Filterbanks • Embeddings │
|
|
235
|
+
├─────────────────────────────────────────────────────────────┤
|
|
236
|
+
│ 2. Audio Processing Layer │
|
|
237
|
+
│ Resampling (16 kHz / 32 kHz) • VAD Gating • Normalization│
|
|
238
|
+
├─────────────────────────────────────────────────────────────┤
|
|
239
|
+
│ 1. Audio Input Layer │
|
|
240
|
+
│ File Decoders (conversion.py) • Microphones • Streams │
|
|
241
|
+
└─────────────────────────────────────────────────────────────┘
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
---
|
|
245
|
+
|
|
246
|
+
## 📦 Project Structure
|
|
247
|
+
|
|
248
|
+
```
|
|
249
|
+
AudioSense/
|
|
250
|
+
├── audiosense/
|
|
251
|
+
│ ├── __init__.py # Public framework exports
|
|
252
|
+
│ ├── core/
|
|
253
|
+
│ │ ├── engine.py # AudioSense primary engine
|
|
254
|
+
│ │ ├── config.py # Dynamic relative path resolver
|
|
255
|
+
│ │ └── silence.py # Log & warning suppression engine
|
|
256
|
+
│ ├── audio/
|
|
257
|
+
│ │ ├── conversion.py # Universal audio converter (mp3/flac/m4a -> wav)
|
|
258
|
+
│ │ ├── preprocess.py # Dual-rate (16k/32k) conditioning
|
|
259
|
+
│ │ └── io.py # Audio reader/writer utilities
|
|
260
|
+
│ ├── models/
|
|
261
|
+
│ │ ├── ast_model.py # AST adapter
|
|
262
|
+
│ │ ├── pann_model.py # PANN (CNN14) adapter
|
|
263
|
+
│ │ ├── yamnet_model.py # YAMNet adapter
|
|
264
|
+
│ │ ├── whisper_model.py # Faster-Whisper adapter
|
|
265
|
+
│ │ └── manager.py # Lazy model coordinator
|
|
266
|
+
│ ├── label.py & labels/ # Functional label APIs & AudioSet taxonomy
|
|
267
|
+
│ ├── switcher/
|
|
268
|
+
│ │ ├── router.py # Tunable Speech vs Environment Router
|
|
269
|
+
│ │ └── vad.py # Silero-VAD detector
|
|
270
|
+
│ ├── context/
|
|
271
|
+
│ │ ├── llm.py # LLM interface (Ollama / llama-server / custom)
|
|
272
|
+
│ │ ├── translator.py # Multi-language translation engine
|
|
273
|
+
│ │ └── generator.py # Context & database summarizers
|
|
274
|
+
│ ├── storage/
|
|
275
|
+
│ │ └── database.py # JSON database engine (view_db)
|
|
276
|
+
│ └── live/
|
|
277
|
+
│ └── loop.py # Real-time microphone listening loop
|
|
278
|
+
├── models/ # Local model weights cache
|
|
279
|
+
├── load_model.py # Model acquisition script
|
|
280
|
+
├── load_llm.py # LLM downloader script
|
|
281
|
+
├── requirements.txt # Python dependencies
|
|
282
|
+
└── README.md # Documentation
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
---
|
|
286
|
+
|
|
287
|
+
## 📄 License
|
|
288
|
+
This project is licensed under the [MIT License](LICENSE).
|