audiosense 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. audiosense-1.0.0/LICENSE +21 -0
  2. audiosense-1.0.0/MANIFEST.in +4 -0
  3. audiosense-1.0.0/PKG-INFO +334 -0
  4. audiosense-1.0.0/README.md +288 -0
  5. audiosense-1.0.0/audiosense/__init__.py +78 -0
  6. audiosense-1.0.0/audiosense/audio/__init__.py +12 -0
  7. audiosense-1.0.0/audiosense/audio/conversion.py +103 -0
  8. audiosense-1.0.0/audiosense/audio/io.py +37 -0
  9. audiosense-1.0.0/audiosense/audio/preprocess.py +115 -0
  10. audiosense-1.0.0/audiosense/cli.py +42 -0
  11. audiosense-1.0.0/audiosense/context/__init__.py +10 -0
  12. audiosense-1.0.0/audiosense/context/download_llm.py +49 -0
  13. audiosense-1.0.0/audiosense/context/generator.py +86 -0
  14. audiosense-1.0.0/audiosense/context/llm.py +100 -0
  15. audiosense-1.0.0/audiosense/context/translator.py +73 -0
  16. audiosense-1.0.0/audiosense/core/__init__.py +3 -0
  17. audiosense-1.0.0/audiosense/core/config.py +69 -0
  18. audiosense-1.0.0/audiosense/core/engine.py +228 -0
  19. audiosense-1.0.0/audiosense/core/silence.py +124 -0
  20. audiosense-1.0.0/audiosense/label.py +25 -0
  21. audiosense-1.0.0/audiosense/labels/__init__.py +16 -0
  22. audiosense-1.0.0/audiosense/labels/all.py +63 -0
  23. audiosense-1.0.0/audiosense/labels/categories.py +42 -0
  24. audiosense-1.0.0/audiosense/labels/selected.py +56 -0
  25. audiosense-1.0.0/audiosense/labels/taxonomy.py +93 -0
  26. audiosense-1.0.0/audiosense/labels/top.py +70 -0
  27. audiosense-1.0.0/audiosense/live/__init__.py +3 -0
  28. audiosense-1.0.0/audiosense/live/loop.py +103 -0
  29. audiosense-1.0.0/audiosense/models/__init__.py +13 -0
  30. audiosense-1.0.0/audiosense/models/ast_model.py +79 -0
  31. audiosense-1.0.0/audiosense/models/download.py +112 -0
  32. audiosense-1.0.0/audiosense/models/manager.py +45 -0
  33. audiosense-1.0.0/audiosense/models/pann_model.py +60 -0
  34. audiosense-1.0.0/audiosense/models/whisper_model.py +48 -0
  35. audiosense-1.0.0/audiosense/models/yamnet_model.py +75 -0
  36. audiosense-1.0.0/audiosense/storage/__init__.py +3 -0
  37. audiosense-1.0.0/audiosense/storage/database.py +136 -0
  38. audiosense-1.0.0/audiosense/switcher/__init__.py +4 -0
  39. audiosense-1.0.0/audiosense/switcher/router.py +128 -0
  40. audiosense-1.0.0/audiosense/switcher/vad.py +40 -0
  41. audiosense-1.0.0/audiosense.egg-info/PKG-INFO +334 -0
  42. audiosense-1.0.0/audiosense.egg-info/SOURCES.txt +48 -0
  43. audiosense-1.0.0/audiosense.egg-info/dependency_links.txt +1 -0
  44. audiosense-1.0.0/audiosense.egg-info/entry_points.txt +4 -0
  45. audiosense-1.0.0/audiosense.egg-info/requires.txt +21 -0
  46. audiosense-1.0.0/audiosense.egg-info/top_level.txt +1 -0
  47. audiosense-1.0.0/pyproject.toml +78 -0
  48. audiosense-1.0.0/requirements.txt +28 -0
  49. audiosense-1.0.0/setup.cfg +4 -0
  50. audiosense-1.0.0/setup.py +3 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 AudioSense Team
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,4 @@
1
+ include README.md
2
+ include LICENSE
3
+ include requirements.txt
4
+ recursive-include audiosense *.py
@@ -0,0 +1,334 @@
1
+ Metadata-Version: 2.4
2
+ Name: audiosense
3
+ Version: 1.0.0
4
+ Summary: Audio intelligence framework for machines, robots and modern applications
5
+ Author-email: Mohan <mohanevs@users.noreply.github.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/mohanevs/AudioSense
8
+ Project-URL: Repository, https://github.com/mohanevs/AudioSense.git
9
+ Project-URL: Issues, https://github.com/mohanevs/AudioSense/issues
10
+ Keywords: audio,sound-classification,speech-recognition,audio-intelligence,robotics,panns,ast,yamnet,whisper,vad
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Analysis
21
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: numpy
26
+ Requires-Dist: scipy
27
+ Requires-Dist: soundfile
28
+ Requires-Dist: sounddevice
29
+ Requires-Dist: librosa
30
+ Requires-Dist: moviepy
31
+ Requires-Dist: torch
32
+ Requires-Dist: torchaudio
33
+ Requires-Dist: transformers
34
+ Requires-Dist: panns_inference
35
+ Requires-Dist: tensorflow
36
+ Requires-Dist: tensorflow-hub
37
+ Requires-Dist: kagglehub
38
+ Requires-Dist: silero-vad
39
+ Requires-Dist: faster-whisper
40
+ Requires-Dist: requests
41
+ Requires-Dist: setuptools>=70.0.0
42
+ Provides-Extra: server
43
+ Requires-Dist: fastapi; extra == "server"
44
+ Requires-Dist: uvicorn; extra == "server"
45
+ Dynamic: license-file
46
+
47
+ <div align="center">
48
+
49
+ # 🎙️ AudioSense SDK
50
+ ### *Audio Intelligence for Machines, Robots, and Modern Applications*
51
+
52
+ [![Python Version](https://img.shields.io/badge/python-3.10%20%7C%203.11%20%7C%203.12-blue.svg?style=for-the-badge&logo=python&logoColor=white)](https://python.org)
53
+ [![Version](https://img.shields.io/badge/version-1.0.0-00C49F.svg?style=for-the-badge)](https://github.com)
54
+ [![Models](https://img.shields.io/badge/Ensemble-AST%20%7C%20PANN%20%7C%20YAMNet-9900EF.svg?style=for-the-badge)](https://github.com)
55
+ [![Speech](https://img.shields.io/badge/Speech-Whisper%20%7C%20Silero%20VAD-FF6F00.svg?style=for-the-badge)](https://github.com)
56
+ [![Platform](https://img.shields.io/badge/Platform-Linux%20%7C%20macOS%20%7C%20Windows-lightgrey.svg?style=for-the-badge)](https://github.com)
57
+ [![License](https://img.shields.io/badge/license-MIT-blueviolet.svg?style=for-the-badge)](LICENSE)
58
+
59
+ <p align="center">
60
+ <a href="#-overview">Overview</a> •
61
+ <a href="#-key-features">Key Features</a> •
62
+ <a href="#-installation">Installation</a> •
63
+ <a href="#-quickstart">Quickstart</a> •
64
+ <a href="#-core-api-guide">Core API</a> •
65
+ <a href="#-7-layer-architecture">Architecture</a> •
66
+ <a href="#-live-streaming">Live Streaming</a>
67
+ </p>
68
+
69
+ ---
70
+
71
+ </div>
72
+
73
+ ## 📌 Overview
74
+
75
+ Computers see and read well, but most still hear poorly. A security camera sees a broken window; a microphone beside it usually outputs nothing but a raw waveform.
76
+
77
+ **AudioSense** bridges that gap. It is a comprehensive audio intelligence framework that transforms raw acoustic signals into:
78
+ 1. **Events** — *What happened and when* (sound classification & detection).
79
+ 2. **Speech Transcripts** — *What was spoken* (automatic VAD gating and transcription).
80
+ 3. **Insights & Context** — *What is happening in the environment* (LLM-driven situational reasoning).
81
+
82
+ Designed with a clean, tiered API: add intelligent hearing to your application or robot with **three lines of Python**, while retaining deep control over every layer underneath.
83
+
84
+ ```python
85
+ from audiosense import AudioSense
86
+
87
+ model = AudioSense()
88
+ result = model.predict("environment.wav")
89
+ print(result)
90
+ # Output: Surroundings currently exhibit sounds of Water tap, faucet, Water, Sink (filling or washing).
91
+ ```
92
+
93
+ ---
94
+
95
+ ## ✨ Key Features
96
+
97
+ | Capability | Description |
98
+ |---|---|
99
+ | **Multi-Model Ensemble** | Combines **AST** (Audio Spectrogram Transformer), **PANNs** (CNN14), and **YAMNet** in a single pass with calibrated consensus scoring. |
100
+ | **Intelligent Routing (Switcher)** | Automatically routes human speech to **Faster-Whisper** transcription and environmental sounds to **Context Intelligence / RAG**. |
101
+ | **Universal Audio Converter** | Ingests `.mp3`, `.flac`, `.ogg`, `.m4a`, `.aac`, `.opus`, `.webm`, `.mp4` and converts to pristine `.wav`. |
102
+ | **Dual Sample-Rate Conditioning** | Internal preprocessing pipeline delivers synchronized 32 kHz (for PANN) and 16 kHz (for AST/YAMNet/Whisper) audio arrays. |
103
+ | **Context Reasoning & LLM** | Synthesizes high-level environmental situation reports via local LLMs (Phi-4, llama-server, Ollama) or custom endpoints. |
104
+ | **Multilingual Translation** | Live translation of context statements and speech transcripts into Hindi (`'hin'`), Spanish, French, German, etc. |
105
+ | **Acoustic Database & Audit** | Built-in JSON event store tracking occurrence counts, timestamps, and confidence histories with `view_db()` and `summary_db()`. |
106
+ | **Real-time Live Mic Loop** | Continuous laptop microphone listening loop (`audiosense.loop()`) with VAD activity filtering and instant callbacks. |
107
+ | **Zero Log Pollution** | Completely silences noisy TensorFlow, oneDNN, and progress-bar outputs, giving clean, predictable returns. |
108
+
109
+ ---
110
+
111
+ ## 🚀 Installation
112
+
113
+ ### 1. Clone & Set Up Environment
114
+ ```bash
115
+ git clone https://github.com/your-org/AudioSense.git
116
+ cd AudioSense
117
+ python -m venv env
118
+ # Windows
119
+ .\env\Scripts\activate
120
+ # Linux/macOS
121
+ source env/bin/activate
122
+ ```
123
+
124
+ ### 2. Install Dependencies
125
+ ```bash
126
+ pip install -r requirements.txt
127
+ ```
128
+
129
+ ### 3. Acquire Pre-Trained Weights
130
+ AudioSense dynamically downloads and verifies model weights:
131
+ ```bash
132
+ # Downloads AST, PANN (CNN14), and YAMNet
133
+ python load_model.py
134
+
135
+ # Optional: Downloads local Phi-4-mini LLM (~2.5 GB)
136
+ python load_llm.py
137
+ ```
138
+
139
+ ---
140
+
141
+ ## ⚡ Quickstart
142
+
143
+ ```python
144
+ from audiosense import AudioSense, view_db
145
+
146
+ # 1. Initialize engine
147
+ model = AudioSense(device="auto", language="en")
148
+
149
+ # 2. Predict on an audio file
150
+ result = model.predict("audio/audio.wav")
151
+ print("Detected Environment:", result.text)
152
+ print("Top Labels:", result.labels)
153
+
154
+ # 3. View persistent sound history
155
+ view_db()
156
+ ```
157
+
158
+ ---
159
+
160
+ ## 📖 Core API Guide
161
+
162
+ ### 1. Unified Prediction & Routing
163
+ The switcher automatically checks whether incoming sound is human speech or environmental:
164
+ ```python
165
+ res = model.predict("audio/sample.wav")
166
+
167
+ if res.route == "transcription":
168
+ print("Speech detected:", res.transcription)
169
+ else:
170
+ print("Context statement:", res.context)
171
+ ```
172
+
173
+ ### 2. Functional Label APIs
174
+ Direct functional access to taxonomy and model predictions:
175
+ ```python
176
+ from audiosense.label import get_label, get_all, get_categories, top_k
177
+
178
+ # Primary label
179
+ primary = get_label("audio/audio.wav")
180
+
181
+ # Top 5 consensus predictions
182
+ rankings = top_k("audio/audio.wav", k=5)
183
+
184
+ # Inspect categories
185
+ cats = get_categories(["Dog", "Car horn", "Speech"])
186
+ # {'Dog': 'Animal sounds', 'Car horn': 'Sounds of things', 'Speech': 'Human sounds'}
187
+
188
+ # Full predictions across all models
189
+ all_preds = get_all("audio/audio.wav", threshold=0.1)
190
+ ```
191
+
192
+ ### 3. Switcher Tuning
193
+ Customize routing rules dynamically for your specific application:
194
+ ```python
195
+ # Move bird chirping to human/speech indicators for this instance
196
+ model.switcher("Birds_chirping", "HUMAN")
197
+
198
+ # Or move a sound to environmental indicators
199
+ model.switcher("Shout", "ENV")
200
+ ```
201
+
202
+ ### 4. Custom LLM Integration
203
+ Plug in your own local or remote language model:
204
+ ```python
205
+ # Custom GGUF file path
206
+ model.llm(path="C:/models/custom_model.gguf")
207
+
208
+ # Or custom Ollama / llama-server HTTP endpoint
209
+ model.llm(url="http://localhost:11434/api/generate")
210
+ ```
211
+
212
+ ### 5. Multilingual Output
213
+ Output context statements and transcripts in your desired language:
214
+ ```python
215
+ model.translation('hin') # Output in Hindi
216
+ res = model.predict("audio/audio.wav")
217
+ print(res.text) # e.g., 'आस-पास पानी के नल और सिंक की आवाजें आ रही हैं।'
218
+ ```
219
+
220
+ ### 6. Database Inspection & Summaries
221
+ ```python
222
+ # Print formatted database table
223
+ model.view_db()
224
+
225
+ # Generate executive summary of auditory history using LLM
226
+ model.summary_db()
227
+ ```
228
+
229
+ ### 7. Universal Audio Conversion
230
+ ```python
231
+ from audiosense import convert_to_wav
232
+
233
+ # Accepts MP3, FLAC, M4A, OGG, WEBM, MP4, etc.
234
+ wav_path = convert_to_wav("recording.m4a", output_path="recording.wav")
235
+ ```
236
+
237
+ ---
238
+
239
+ ## 🎙️ Live Microphone & Robotics
240
+
241
+ Continuous, non-blocking real-time listening through the system microphone:
242
+
243
+ ```python
244
+ from audiosense import AudioSense
245
+
246
+ model = AudioSense()
247
+
248
+ # Start continuous listening loop (Press Ctrl+C to stop)
249
+ model.loop(chunk_duration=3.0)
250
+ ```
251
+
252
+ You can also provide a callback for robot reactive control:
253
+ ```python
254
+ def on_sound(res):
255
+ if "Siren" in [lbl for lbl, _ in res.labels]:
256
+ print("🚨 Warning: Emergency siren heard! Halting robot.")
257
+
258
+ model.loop(chunk_duration=2.0, callback=on_sound)
259
+ ```
260
+
261
+ ---
262
+
263
+ ## 🏛️ 7-Layer Architecture
264
+
265
+ ```
266
+ ┌─────────────────────────────────────────────────────────────┐
267
+ │ 7. Application & Robot Interface │
268
+ │ Python API (AudioSense) • Callbacks • Controller Maps │
269
+ ├─────────────────────────────────────────────────────────────┤
270
+ │ 6. Context Intelligence Layer │
271
+ │ Situational Reasoning • LLM Summaries • Multilingual │
272
+ ├─────────────────────────────────────────────────────────────┤
273
+ │ 5. Ensemble Decision Layer & Switcher │
274
+ │ Weighted Consensus • Speech vs. Environmental Router │
275
+ ├─────────────────────────────────────────────────────────────┤
276
+ │ 4. AI Model Layer │
277
+ │ AST (Transformer) • PANN (CNN14) • YAMNet • Whisper │
278
+ ├─────────────────────────────────────────────────────────────┤
279
+ │ 3. Feature Extraction Layer │
280
+ │ Log-mel Spectrograms • Filterbanks • Embeddings │
281
+ ├─────────────────────────────────────────────────────────────┤
282
+ │ 2. Audio Processing Layer │
283
+ │ Resampling (16 kHz / 32 kHz) • VAD Gating • Normalization│
284
+ ├─────────────────────────────────────────────────────────────┤
285
+ │ 1. Audio Input Layer │
286
+ │ File Decoders (conversion.py) • Microphones • Streams │
287
+ └─────────────────────────────────────────────────────────────┘
288
+ ```
289
+
290
+ ---
291
+
292
+ ## 📦 Project Structure
293
+
294
+ ```
295
+ AudioSense/
296
+ ├── audiosense/
297
+ │ ├── __init__.py # Public framework exports
298
+ │ ├── core/
299
+ │ │ ├── engine.py # AudioSense primary engine
300
+ │ │ ├── config.py # Dynamic relative path resolver
301
+ │ │ └── silence.py # Log & warning suppression engine
302
+ │ ├── audio/
303
+ │ │ ├── conversion.py # Universal audio converter (mp3/flac/m4a -> wav)
304
+ │ │ ├── preprocess.py # Dual-rate (16k/32k) conditioning
305
+ │ │ └── io.py # Audio reader/writer utilities
306
+ │ ├── models/
307
+ │ │ ├── ast_model.py # AST adapter
308
+ │ │ ├── pann_model.py # PANN (CNN14) adapter
309
+ │ │ ├── yamnet_model.py # YAMNet adapter
310
+ │ │ ├── whisper_model.py # Faster-Whisper adapter
311
+ │ │ └── manager.py # Lazy model coordinator
312
+ │ ├── label.py & labels/ # Functional label APIs & AudioSet taxonomy
313
+ │ ├── switcher/
314
+ │ │ ├── router.py # Tunable Speech vs Environment Router
315
+ │ │ └── vad.py # Silero-VAD detector
316
+ │ ├── context/
317
+ │ │ ├── llm.py # LLM interface (Ollama / llama-server / custom)
318
+ │ │ ├── translator.py # Multi-language translation engine
319
+ │ │ └── generator.py # Context & database summarizers
320
+ │ ├── storage/
321
+ │ │ └── database.py # JSON database engine (view_db)
322
+ │ └── live/
323
+ │ └── loop.py # Real-time microphone listening loop
324
+ ├── models/ # Local model weights cache
325
+ ├── load_model.py # Model acquisition script
326
+ ├── load_llm.py # LLM downloader script
327
+ ├── requirements.txt # Python dependencies
328
+ └── README.md # Documentation
329
+ ```
330
+
331
+ ---
332
+
333
+ ## 📄 License
334
+ This project is licensed under the [MIT License](LICENSE).
@@ -0,0 +1,288 @@
1
+ <div align="center">
2
+
3
+ # 🎙️ AudioSense SDK
4
+ ### *Audio Intelligence for Machines, Robots, and Modern Applications*
5
+
6
+ [![Python Version](https://img.shields.io/badge/python-3.10%20%7C%203.11%20%7C%203.12-blue.svg?style=for-the-badge&logo=python&logoColor=white)](https://python.org)
7
+ [![Version](https://img.shields.io/badge/version-1.0.0-00C49F.svg?style=for-the-badge)](https://github.com)
8
+ [![Models](https://img.shields.io/badge/Ensemble-AST%20%7C%20PANN%20%7C%20YAMNet-9900EF.svg?style=for-the-badge)](https://github.com)
9
+ [![Speech](https://img.shields.io/badge/Speech-Whisper%20%7C%20Silero%20VAD-FF6F00.svg?style=for-the-badge)](https://github.com)
10
+ [![Platform](https://img.shields.io/badge/Platform-Linux%20%7C%20macOS%20%7C%20Windows-lightgrey.svg?style=for-the-badge)](https://github.com)
11
+ [![License](https://img.shields.io/badge/license-MIT-blueviolet.svg?style=for-the-badge)](LICENSE)
12
+
13
+ <p align="center">
14
+ <a href="#-overview">Overview</a> •
15
+ <a href="#-key-features">Key Features</a> •
16
+ <a href="#-installation">Installation</a> •
17
+ <a href="#-quickstart">Quickstart</a> •
18
+ <a href="#-core-api-guide">Core API</a> •
19
+ <a href="#-7-layer-architecture">Architecture</a> •
20
+ <a href="#-live-streaming">Live Streaming</a>
21
+ </p>
22
+
23
+ ---
24
+
25
+ </div>
26
+
27
+ ## 📌 Overview
28
+
29
+ Computers see and read well, but most still hear poorly. A security camera sees a broken window; a microphone beside it usually outputs nothing but a raw waveform.
30
+
31
+ **AudioSense** bridges that gap. It is a comprehensive audio intelligence framework that transforms raw acoustic signals into:
32
+ 1. **Events** — *What happened and when* (sound classification & detection).
33
+ 2. **Speech Transcripts** — *What was spoken* (automatic VAD gating and transcription).
34
+ 3. **Insights & Context** — *What is happening in the environment* (LLM-driven situational reasoning).
35
+
36
+ Designed with a clean, tiered API: add intelligent hearing to your application or robot with **three lines of Python**, while retaining deep control over every layer underneath.
37
+
38
+ ```python
39
+ from audiosense import AudioSense
40
+
41
+ model = AudioSense()
42
+ result = model.predict("environment.wav")
43
+ print(result)
44
+ # Output: Surroundings currently exhibit sounds of Water tap, faucet, Water, Sink (filling or washing).
45
+ ```
46
+
47
+ ---
48
+
49
+ ## ✨ Key Features
50
+
51
+ | Capability | Description |
52
+ |---|---|
53
+ | **Multi-Model Ensemble** | Combines **AST** (Audio Spectrogram Transformer), **PANNs** (CNN14), and **YAMNet** in a single pass with calibrated consensus scoring. |
54
+ | **Intelligent Routing (Switcher)** | Automatically routes human speech to **Faster-Whisper** transcription and environmental sounds to **Context Intelligence / RAG**. |
55
+ | **Universal Audio Converter** | Ingests `.mp3`, `.flac`, `.ogg`, `.m4a`, `.aac`, `.opus`, `.webm`, `.mp4` and converts to pristine `.wav`. |
56
+ | **Dual Sample-Rate Conditioning** | Internal preprocessing pipeline delivers synchronized 32 kHz (for PANN) and 16 kHz (for AST/YAMNet/Whisper) audio arrays. |
57
+ | **Context Reasoning & LLM** | Synthesizes high-level environmental situation reports via local LLMs (Phi-4, llama-server, Ollama) or custom endpoints. |
58
+ | **Multilingual Translation** | Live translation of context statements and speech transcripts into Hindi (`'hin'`), Spanish, French, German, etc. |
59
+ | **Acoustic Database & Audit** | Built-in JSON event store tracking occurrence counts, timestamps, and confidence histories with `view_db()` and `summary_db()`. |
60
+ | **Real-time Live Mic Loop** | Continuous laptop microphone listening loop (`audiosense.loop()`) with VAD activity filtering and instant callbacks. |
61
+ | **Zero Log Pollution** | Completely silences noisy TensorFlow, oneDNN, and progress-bar outputs, giving clean, predictable returns. |
62
+
63
+ ---
64
+
65
+ ## 🚀 Installation
66
+
67
+ ### 1. Clone & Set Up Environment
68
+ ```bash
69
+ git clone https://github.com/your-org/AudioSense.git
70
+ cd AudioSense
71
+ python -m venv env
72
+ # Windows
73
+ .\env\Scripts\activate
74
+ # Linux/macOS
75
+ source env/bin/activate
76
+ ```
77
+
78
+ ### 2. Install Dependencies
79
+ ```bash
80
+ pip install -r requirements.txt
81
+ ```
82
+
83
+ ### 3. Acquire Pre-Trained Weights
84
+ AudioSense dynamically downloads and verifies model weights:
85
+ ```bash
86
+ # Downloads AST, PANN (CNN14), and YAMNet
87
+ python load_model.py
88
+
89
+ # Optional: Downloads local Phi-4-mini LLM (~2.5 GB)
90
+ python load_llm.py
91
+ ```
92
+
93
+ ---
94
+
95
+ ## ⚡ Quickstart
96
+
97
+ ```python
98
+ from audiosense import AudioSense, view_db
99
+
100
+ # 1. Initialize engine
101
+ model = AudioSense(device="auto", language="en")
102
+
103
+ # 2. Predict on an audio file
104
+ result = model.predict("audio/audio.wav")
105
+ print("Detected Environment:", result.text)
106
+ print("Top Labels:", result.labels)
107
+
108
+ # 3. View persistent sound history
109
+ view_db()
110
+ ```
111
+
112
+ ---
113
+
114
+ ## 📖 Core API Guide
115
+
116
+ ### 1. Unified Prediction & Routing
117
+ The switcher automatically checks whether incoming sound is human speech or environmental:
118
+ ```python
119
+ res = model.predict("audio/sample.wav")
120
+
121
+ if res.route == "transcription":
122
+ print("Speech detected:", res.transcription)
123
+ else:
124
+ print("Context statement:", res.context)
125
+ ```
126
+
127
+ ### 2. Functional Label APIs
128
+ Direct functional access to taxonomy and model predictions:
129
+ ```python
130
+ from audiosense.label import get_label, get_all, get_categories, top_k
131
+
132
+ # Primary label
133
+ primary = get_label("audio/audio.wav")
134
+
135
+ # Top 5 consensus predictions
136
+ rankings = top_k("audio/audio.wav", k=5)
137
+
138
+ # Inspect categories
139
+ cats = get_categories(["Dog", "Car horn", "Speech"])
140
+ # {'Dog': 'Animal sounds', 'Car horn': 'Sounds of things', 'Speech': 'Human sounds'}
141
+
142
+ # Full predictions across all models
143
+ all_preds = get_all("audio/audio.wav", threshold=0.1)
144
+ ```
145
+
146
+ ### 3. Switcher Tuning
147
+ Customize routing rules dynamically for your specific application:
148
+ ```python
149
+ # Move bird chirping to human/speech indicators for this instance
150
+ model.switcher("Birds_chirping", "HUMAN")
151
+
152
+ # Or move a sound to environmental indicators
153
+ model.switcher("Shout", "ENV")
154
+ ```
155
+
156
+ ### 4. Custom LLM Integration
157
+ Plug in your own local or remote language model:
158
+ ```python
159
+ # Custom GGUF file path
160
+ model.llm(path="C:/models/custom_model.gguf")
161
+
162
+ # Or custom Ollama / llama-server HTTP endpoint
163
+ model.llm(url="http://localhost:11434/api/generate")
164
+ ```
165
+
166
+ ### 5. Multilingual Output
167
+ Output context statements and transcripts in your desired language:
168
+ ```python
169
+ model.translation('hin') # Output in Hindi
170
+ res = model.predict("audio/audio.wav")
171
+ print(res.text) # e.g., 'आस-पास पानी के नल और सिंक की आवाजें आ रही हैं।'
172
+ ```
173
+
174
+ ### 6. Database Inspection & Summaries
175
+ ```python
176
+ # Print formatted database table
177
+ model.view_db()
178
+
179
+ # Generate executive summary of auditory history using LLM
180
+ model.summary_db()
181
+ ```
182
+
183
+ ### 7. Universal Audio Conversion
184
+ ```python
185
+ from audiosense import convert_to_wav
186
+
187
+ # Accepts MP3, FLAC, M4A, OGG, WEBM, MP4, etc.
188
+ wav_path = convert_to_wav("recording.m4a", output_path="recording.wav")
189
+ ```
190
+
191
+ ---
192
+
193
+ ## 🎙️ Live Microphone & Robotics
194
+
195
+ Continuous, non-blocking real-time listening through the system microphone:
196
+
197
+ ```python
198
+ from audiosense import AudioSense
199
+
200
+ model = AudioSense()
201
+
202
+ # Start continuous listening loop (Press Ctrl+C to stop)
203
+ model.loop(chunk_duration=3.0)
204
+ ```
205
+
206
+ You can also provide a callback for robot reactive control:
207
+ ```python
208
+ def on_sound(res):
209
+ if "Siren" in [lbl for lbl, _ in res.labels]:
210
+ print("🚨 Warning: Emergency siren heard! Halting robot.")
211
+
212
+ model.loop(chunk_duration=2.0, callback=on_sound)
213
+ ```
214
+
215
+ ---
216
+
217
+ ## 🏛️ 7-Layer Architecture
218
+
219
+ ```
220
+ ┌─────────────────────────────────────────────────────────────┐
221
+ │ 7. Application & Robot Interface │
222
+ │ Python API (AudioSense) • Callbacks • Controller Maps │
223
+ ├─────────────────────────────────────────────────────────────┤
224
+ │ 6. Context Intelligence Layer │
225
+ │ Situational Reasoning • LLM Summaries • Multilingual │
226
+ ├─────────────────────────────────────────────────────────────┤
227
+ │ 5. Ensemble Decision Layer & Switcher │
228
+ │ Weighted Consensus • Speech vs. Environmental Router │
229
+ ├─────────────────────────────────────────────────────────────┤
230
+ │ 4. AI Model Layer │
231
+ │ AST (Transformer) • PANN (CNN14) • YAMNet • Whisper │
232
+ ├─────────────────────────────────────────────────────────────┤
233
+ │ 3. Feature Extraction Layer │
234
+ │ Log-mel Spectrograms • Filterbanks • Embeddings │
235
+ ├─────────────────────────────────────────────────────────────┤
236
+ │ 2. Audio Processing Layer │
237
+ │ Resampling (16 kHz / 32 kHz) • VAD Gating • Normalization│
238
+ ├─────────────────────────────────────────────────────────────┤
239
+ │ 1. Audio Input Layer │
240
+ │ File Decoders (conversion.py) • Microphones • Streams │
241
+ └─────────────────────────────────────────────────────────────┘
242
+ ```
243
+
244
+ ---
245
+
246
+ ## 📦 Project Structure
247
+
248
+ ```
249
+ AudioSense/
250
+ ├── audiosense/
251
+ │ ├── __init__.py # Public framework exports
252
+ │ ├── core/
253
+ │ │ ├── engine.py # AudioSense primary engine
254
+ │ │ ├── config.py # Dynamic relative path resolver
255
+ │ │ └── silence.py # Log & warning suppression engine
256
+ │ ├── audio/
257
+ │ │ ├── conversion.py # Universal audio converter (mp3/flac/m4a -> wav)
258
+ │ │ ├── preprocess.py # Dual-rate (16k/32k) conditioning
259
+ │ │ └── io.py # Audio reader/writer utilities
260
+ │ ├── models/
261
+ │ │ ├── ast_model.py # AST adapter
262
+ │ │ ├── pann_model.py # PANN (CNN14) adapter
263
+ │ │ ├── yamnet_model.py # YAMNet adapter
264
+ │ │ ├── whisper_model.py # Faster-Whisper adapter
265
+ │ │ └── manager.py # Lazy model coordinator
266
+ │ ├── label.py & labels/ # Functional label APIs & AudioSet taxonomy
267
+ │ ├── switcher/
268
+ │ │ ├── router.py # Tunable Speech vs Environment Router
269
+ │ │ └── vad.py # Silero-VAD detector
270
+ │ ├── context/
271
+ │ │ ├── llm.py # LLM interface (Ollama / llama-server / custom)
272
+ │ │ ├── translator.py # Multi-language translation engine
273
+ │ │ └── generator.py # Context & database summarizers
274
+ │ ├── storage/
275
+ │ │ └── database.py # JSON database engine (view_db)
276
+ │ └── live/
277
+ │ └── loop.py # Real-time microphone listening loop
278
+ ├── models/ # Local model weights cache
279
+ ├── load_model.py # Model acquisition script
280
+ ├── load_llm.py # LLM downloader script
281
+ ├── requirements.txt # Python dependencies
282
+ └── README.md # Documentation
283
+ ```
284
+
285
+ ---
286
+
287
+ ## 📄 License
288
+ This project is licensed under the [MIT License](LICENSE).