ameva-runtime 2.0.2__tar.gz → 2.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ameva_runtime-2.2.0/PKG-INFO +92 -0
- ameva_runtime-2.2.0/README.md +111 -0
- ameva_runtime-2.2.0/README.pypi.md +41 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/pyproject.toml +12 -12
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/_version.py +1 -1
- {ameva_runtime-2.0.2/python/ameva_runtime/vulkan → ameva_runtime-2.2.0/python/ameva_runtime}/adapters/__init__.py +18 -1
- ameva_runtime-2.2.0/python/ameva_runtime/adapters/base.py +247 -0
- {ameva_runtime-2.0.2/python/ameva_runtime/vulkan → ameva_runtime-2.2.0/python/ameva_runtime}/adapters/bitnet.py +49 -16
- ameva_runtime-2.2.0/python/ameva_runtime/adapters/diffusion.py +215 -0
- {ameva_runtime-2.0.2/python/ameva_runtime/vulkan → ameva_runtime-2.2.0/python/ameva_runtime}/adapters/llamacpp.py +51 -15
- ameva_runtime-2.2.0/python/ameva_runtime/adapters/stt.py +164 -0
- ameva_runtime-2.2.0/python/ameva_runtime/adapters/tts.py +232 -0
- {ameva_runtime-2.0.2/python/ameva_runtime/vulkan → ameva_runtime-2.2.0/python/ameva_runtime}/adapters/vision.py +61 -21
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/detector.py +217 -236
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/protocol.py +18 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/router.py +125 -10
- ameva_runtime-2.2.0/python/ameva_runtime/vulkan/adapters/__init__.py +34 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/doctor.py +18 -0
- ameva_runtime-2.2.0/python/ameva_runtime/vulkan/protocol.py +22 -0
- ameva_runtime-2.2.0/python/ameva_runtime.egg-info/PKG-INFO +92 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime.egg-info/SOURCES.txt +15 -12
- ameva_runtime-2.2.0/python/ameva_runtime.egg-info/requires.txt +26 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime.egg-info/top_level.txt +1 -0
- ameva_runtime-2.2.0/python/termux_train/__init__.py +30 -0
- ameva_runtime-2.2.0/python/termux_train/backends/__init__.py +21 -0
- ameva_runtime-2.2.0/python/termux_train/backends/base.py +105 -0
- ameva_runtime-2.2.0/python/termux_train/backends/diff.py +27 -0
- ameva_runtime-2.2.0/python/termux_train/backends/llm.py +175 -0
- ameva_runtime-2.2.0/python/termux_train/backends/stt.py +27 -0
- ameva_runtime-2.2.0/python/termux_train/backends/tts.py +27 -0
- ameva_runtime-2.2.0/python/termux_train/backends/vision.py +27 -0
- ameva_runtime-2.2.0/python/termux_train/cli.py +127 -0
- ameva_runtime-2.2.0/python/termux_train/core.py +57 -0
- ameva_runtime-2.2.0/python/termux_train/exceptions.py +66 -0
- ameva_runtime-2.2.0/python/termux_train/utils/__init__.py +13 -0
- ameva_runtime-2.2.0/python/termux_train/utils/hardware.py +95 -0
- ameva_runtime-2.2.0/python/termux_train/utils/monitor.py +73 -0
- ameva_runtime-2.2.0/tests/test_termux_train.py +143 -0
- ameva_runtime-2.0.2/PKG-INFO +0 -145
- ameva_runtime-2.0.2/README.md +0 -170
- ameva_runtime-2.0.2/README.pypi.md +0 -94
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/__init__.py +0 -23
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/base.py +0 -67
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/bitnet.py +0 -50
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/diffusion.py +0 -50
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/llamacpp.py +0 -88
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/stt.py +0 -67
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/tts.py +0 -50
- ameva_runtime-2.0.2/python/ameva_runtime/adapters/vision.py +0 -51
- ameva_runtime-2.0.2/python/ameva_runtime/vulkan/adapters/base.py +0 -182
- ameva_runtime-2.0.2/python/ameva_runtime/vulkan/adapters/diffusion.py +0 -144
- ameva_runtime-2.0.2/python/ameva_runtime/vulkan/adapters/stt.py +0 -118
- ameva_runtime-2.0.2/python/ameva_runtime/vulkan/adapters/tts.py +0 -72
- ameva_runtime-2.0.2/python/ameva_runtime/vulkan/protocol.py +0 -125
- ameva_runtime-2.0.2/python/ameva_runtime.egg-info/PKG-INFO +0 -145
- ameva_runtime-2.0.2/python/ameva_runtime.egg-info/requires.txt +0 -26
- ameva_runtime-2.0.2/python/tests/test_ameva_runtime.py +0 -171
- ameva_runtime-2.0.2/python/tests/test_core.py +0 -143
- ameva_runtime-2.0.2/python/tests/test_doctor_v0_v11.py +0 -117
- ameva_runtime-2.0.2/python/tests/test_handle_lifecycle.py +0 -62
- ameva_runtime-2.0.2/python/tests/test_modalities.py +0 -165
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/LICENSE +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/MANIFEST.in +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/profiles/validated-vulkan-profiles.json +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/__init__.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/cli.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/core.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/doctor.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/exceptions.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/platform.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/__init__.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/bindings.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/cli.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/core.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/exceptions.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/platform.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/profiles/__init__.py +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/profiles/validated-vulkan-profiles.json +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime/vulkan/py.typed +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime.egg-info/dependency_links.txt +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/python/ameva_runtime.egg-info/entry_points.txt +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/setup.cfg +0 -0
- {ameva_runtime-2.0.2 → ameva_runtime-2.2.0}/setup.py +0 -0
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ameva-runtime
|
|
3
|
+
Version: 2.2.0
|
|
4
|
+
Summary: Unified Next-Gen Hardware Orchestration & AI Acceleration Runtime for Mobile & Edge
|
|
5
|
+
Home-page: https://github.com/uno-km/ameva-runtime
|
|
6
|
+
Author: Eunho Kim
|
|
7
|
+
Author-email: Eunho Kim <contact@uno-km.com>
|
|
8
|
+
License: Apache-2.0
|
|
9
|
+
Project-URL: Homepage, https://uno-km.vercel.app/lib/vulkan/
|
|
10
|
+
Project-URL: Repository, https://github.com/uno-km/ameva-runtime
|
|
11
|
+
Project-URL: Documentation, https://uno-km.vercel.app/lib/vulkan/
|
|
12
|
+
Keywords: vulkan,opencl,npu,cpu-neon,android,termux,acceleration,adreno,mali,hal,ameva
|
|
13
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
16
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
17
|
+
Classifier: Operating System :: Android
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
25
|
+
Requires-Python: >=3.8
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Provides-Extra: stt
|
|
29
|
+
Requires-Dist: termux-stt>=1.2.0; extra == "stt"
|
|
30
|
+
Provides-Extra: diffusion
|
|
31
|
+
Requires-Dist: termux-diffusion>=1.5.0; extra == "diffusion"
|
|
32
|
+
Provides-Extra: bitnet
|
|
33
|
+
Requires-Dist: termux-bitnet>=1.2.0; extra == "bitnet"
|
|
34
|
+
Provides-Extra: llamacpp
|
|
35
|
+
Requires-Dist: termux-llamacpp>=1.3.0; extra == "llamacpp"
|
|
36
|
+
Provides-Extra: tts
|
|
37
|
+
Requires-Dist: termux-tts>=1.4.0; extra == "tts"
|
|
38
|
+
Provides-Extra: vision
|
|
39
|
+
Requires-Dist: termux-vision>=1.2.0; extra == "vision"
|
|
40
|
+
Provides-Extra: all
|
|
41
|
+
Requires-Dist: termux-stt>=1.2.0; extra == "all"
|
|
42
|
+
Requires-Dist: termux-diffusion>=1.5.0; extra == "all"
|
|
43
|
+
Requires-Dist: termux-bitnet>=1.2.0; extra == "all"
|
|
44
|
+
Requires-Dist: termux-llamacpp>=1.3.0; extra == "all"
|
|
45
|
+
Requires-Dist: termux-tts>=1.4.0; extra == "all"
|
|
46
|
+
Requires-Dist: termux-vision>=1.2.0; extra == "all"
|
|
47
|
+
Dynamic: author
|
|
48
|
+
Dynamic: home-page
|
|
49
|
+
Dynamic: license-file
|
|
50
|
+
Dynamic: requires-python
|
|
51
|
+
|
|
52
|
+
# AMEVA-Runtime (Python)
|
|
53
|
+
|
|
54
|
+
[](https://pypi.org/project/ameva-runtime/)
|
|
55
|
+
[](https://pypi.org/project/ameva-runtime/)
|
|
56
|
+
[](https://github.com/uno-km/ameva-runtime)
|
|
57
|
+
|
|
58
|
+
> Next-Gen Unified On-Device Hardware Orchestration & 6-Modality AI Acceleration Runtime for Mobile & Edge
|
|
59
|
+
|
|
60
|
+
## Installation
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install ameva-runtime
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Quickstart
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
import ameva_runtime as ameva
|
|
70
|
+
from ameva_runtime import vulkan
|
|
71
|
+
|
|
72
|
+
# Inspect hardware
|
|
73
|
+
profile = ameva.detect_hardware()
|
|
74
|
+
print(f"Target: {profile.soc_name} | {profile.gpu_vendor}")
|
|
75
|
+
|
|
76
|
+
# Run diagnostics
|
|
77
|
+
doc = vulkan.Doctor()
|
|
78
|
+
report = doc.run_self_test()
|
|
79
|
+
print(f"Passed: {report.passed_stages}/{report.total_stages}")
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Empirical Benchmarks
|
|
83
|
+
|
|
84
|
+
- **Galaxy S25 (Adreno 830)**: LLM 35.80 t/s (VRAM 25/25 layers), Whisper STT 4,401 ms, TTS RTF 0.264x (medium) / 0.993x (high-fp16).
|
|
85
|
+
- **Galaxy A35 (Mali-G68 MP5)**: LLM 4.44 t/s (+26.9% vs NEON), Whisper STT 360.60s (2.26x speedup), TTS RTF 1.146x.
|
|
86
|
+
|
|
87
|
+
## Documentation
|
|
88
|
+
- [Official Documentation](https://uno-km.vercel.app/lib/vulkan/)
|
|
89
|
+
- [GitHub Repository](https://github.com/uno-km/ameva-runtime)
|
|
90
|
+
|
|
91
|
+
## License
|
|
92
|
+
Apache-2.0 License. Copyright (c) 2026 Eunho Kim (@uno-km).
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
# AMEVA-Runtime
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/ameva-runtime/)
|
|
4
|
+
[](https://pypi.org/project/ameva-runtime/)
|
|
5
|
+
[](https://www.npmjs.com/package/@ameva/runtime)
|
|
6
|
+
[](https://github.com/uno-km/ameva-runtime)
|
|
7
|
+
|
|
8
|
+
> Next-Gen Unified On-Device Hardware Orchestration & 6-Modality AI Acceleration Runtime for Mobile & Edge
|
|
9
|
+
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
## Architecture & Overview
|
|
13
|
+
|
|
14
|
+
AMEVA-Runtime is a hardware abstraction layer (HAL) and compute orchestration engine engineered specifically for mobile ARM64 devices (Android Termux, Linux Edge). It continuously inspects underlying silicon topology (/dev/kgsl-3d0, /dev/mali0) to route tensor execution across Qualcomm Adreno, ARM Mali, and ARM Cortex CPU-NEON backends.
|
|
15
|
+
|
|
16
|
+
### 6-Modality Acceleration Matrix
|
|
17
|
+
|
|
18
|
+
| Modality | Engine Integration | Status (v2.1.0) | Hardware Acceleration Mechanism |
|
|
19
|
+
| :--- | :--- | :---: | :--- |
|
|
20
|
+
| **1. LLM (Text)** | Llama.cpp (Qwen2.5, Llama 3.2) | **Production** | Full 25/25 layer VRAM offload (Adreno 35.80 t/s, Mali 4.44 t/s) |
|
|
21
|
+
| **2. STT (Speech)** | Whisper.cpp (Large-v3-Turbo) | **Production** | Vulkan compute shader acceleration (Adreno 4.4s, Mali 2.26x speedup) |
|
|
22
|
+
| **3. TTS (Audio)** | Sherpa-NCNN / Piper | **Production** | Pure Vulkan GPU neural synthesis (Adreno RTF 0.264x, Mali RTF 1.146x) |
|
|
23
|
+
| **4. Vision (VLM)** | CLIP / MobileVLM / LLaVA | **In Development** | GGML Vulkan vision encoder tensor bindings |
|
|
24
|
+
| **5. Diffusion (Image)** | Stable Diffusion v1.5 / FLUX.1 | **In Development** | On-device Vulkan UNet & DiT tensor offload |
|
|
25
|
+
| **6. Train (Training)** | On-Device LoRA / QLoRA | **In Development** | Mobile Vulkan gradient descent backpropagation |
|
|
26
|
+
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
## Empirical Physical Device Benchmarks
|
|
30
|
+
|
|
31
|
+
Tested on physical devices running Android 16 under Termux ARM64:
|
|
32
|
+
|
|
33
|
+
### 1. LLM Generation (Qwen2.5-0.5B-Instruct Q4_K_M)
|
|
34
|
+
| Target Device | Hardware Architecture | Active Backend | Layers in VRAM | Generation Speed | Prompt Processing | Speedup |
|
|
35
|
+
| :--- | :--- | :---: | :---: | :---: | :---: | :---: |
|
|
36
|
+
| **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | Vulkan 1.3 | **25/25 (100%)** | **35.80 t/s** (27.9 ms/t) | 4.53 t/s | **35.8x (vs CPU)** |
|
|
37
|
+
| **Galaxy A35** | Exynos 1380 / ARM Mali-G68 MP5 | Vulkan 1.3 | **25/25 (100%)** | **4.44 t/s** (225 ms/t) | 6.12 t/s | **+26.9% (vs NEON)** |
|
|
38
|
+
| **Galaxy A35** | Cortex-A78 CPU-NEON (3 Threads) | CPU-NEON | 0/25 | 3.55 t/s (281 ms/t) | 8.05 t/s | Baseline |
|
|
39
|
+
|
|
40
|
+
### 2. Speech-to-Text (Whisper Large-v3-Turbo Q5_0, 548MB)
|
|
41
|
+
| Target Device | Hardware Architecture | Backend Mode | Latency (1-min audio) | GPU Load | CPU Load | Speedup |
|
|
42
|
+
| :--- | :--- | :---: | :---: | :---: | :---: | :---: |
|
|
43
|
+
| **Galaxy A35** | Exynos 1380 / Mali-G68 MP5 | Vulkan GPU | **360.60 s (6m 00s)** | **949 MHz (100%)** | 20~30% | **2.26x (56% time saved)** |
|
|
44
|
+
| **Galaxy A35** | Cortex-A78 x4 Cores | CPU-NEON | 816.48 s (13m 36s) | 0% | 291% | Baseline |
|
|
45
|
+
|
|
46
|
+
### 3. Text-to-Speech (Termux-TTS v1.3.0 Vulkan)
|
|
47
|
+
| Target Device | Hardware Architecture | Model Tier | Audio Length | Compute Time | Real-Time Factor (RTF) | Status |
|
|
48
|
+
| :--- | :--- | :--- | :---: | :---: | :---: | :---: |
|
|
49
|
+
| **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | `lessac-high-fp16` | 6.70 s | **6.65 s** | **0.993x** | Real-time Studio |
|
|
50
|
+
| **Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | `lessac-medium` | 4.59 s | **1.21 s** | **0.264x** | 3.79x Faster than RT |
|
|
51
|
+
| **Galaxy A35** | Exynos 1380 / Mali-G68 MP5 | `lessac-medium` | 4.52 s | **5.18 s** | **1.146x** | Validated |
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## Root-Cause Driver Solutions
|
|
56
|
+
|
|
57
|
+
1. **ARM Mali-G68 Valhall Integer Truncation**: Enforced medium tile matmul kernel dispatch (`loadstride_b = 4 > 0`), permanently eliminating shader zero-stride infinite loops on subgroup-16 hardware.
|
|
58
|
+
2. **Qualcomm Adreno 830 JIT Register Bug**: Bounded vector column specialization (`mul_mat_vec_max_cols = 2`), preventing compiler crash `VK_ERROR_UNKNOWN (-13)`.
|
|
59
|
+
3. **Zero-Silent-Fallback**: Guaranteed fail-fast architecture without silent CPU degradation upon GPU driver faults.
|
|
60
|
+
|
|
61
|
+
---
|
|
62
|
+
|
|
63
|
+
## Installation
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
# Python SDK
|
|
67
|
+
pip install ameva-runtime
|
|
68
|
+
|
|
69
|
+
# Node.js / TypeScript
|
|
70
|
+
npm install @ameva/runtime
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
## Quickstart
|
|
76
|
+
|
|
77
|
+
### Python SDK
|
|
78
|
+
```python
|
|
79
|
+
import ameva_runtime as ameva
|
|
80
|
+
from ameva_runtime import vulkan
|
|
81
|
+
|
|
82
|
+
# 1. Inspect on-device silicon topology
|
|
83
|
+
profile = ameva.detect_hardware()
|
|
84
|
+
print(f"SoC: {profile.soc_name} | GPU: {profile.gpu_vendor}")
|
|
85
|
+
|
|
86
|
+
# 2. Run hardware self-test
|
|
87
|
+
doc = vulkan.Doctor()
|
|
88
|
+
report = doc.run_self_test()
|
|
89
|
+
print(f"GPU: {report.device_name} (Passed: {report.passed_stages}/{report.total_stages})")
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
### Node.js / TypeScript
|
|
93
|
+
```typescript
|
|
94
|
+
import { Doctor, createContext } from '@ameva/runtime';
|
|
95
|
+
|
|
96
|
+
const doc = new Doctor();
|
|
97
|
+
const report = await doc.runSelfTest();
|
|
98
|
+
console.log(`Vulkan GPU: ${report.deviceName}`);
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
---
|
|
102
|
+
|
|
103
|
+
## Official Documentation & Benchmarks
|
|
104
|
+
- [Official Architecture & API Reference](https://uno-km.vercel.app/lib/vulkan/)
|
|
105
|
+
- [Ecosystem Metrics & Registry Stats](https://uno-km.vercel.app/foundation/metrics)
|
|
106
|
+
- [AMEVA Open-Source Foundation Portal](https://uno-km.vercel.app/foundation/index.html)
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## License
|
|
111
|
+
Licensed under the Apache-2.0 License. Copyright (c) 2026 Eunho Kim ([@uno-km](https://github.com/uno-km)).
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# AMEVA-Runtime (Python)
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/ameva-runtime/)
|
|
4
|
+
[](https://pypi.org/project/ameva-runtime/)
|
|
5
|
+
[](https://github.com/uno-km/ameva-runtime)
|
|
6
|
+
|
|
7
|
+
> Next-Gen Unified On-Device Hardware Orchestration & 6-Modality AI Acceleration Runtime for Mobile & Edge
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install ameva-runtime
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Quickstart
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
import ameva_runtime as ameva
|
|
19
|
+
from ameva_runtime import vulkan
|
|
20
|
+
|
|
21
|
+
# Inspect hardware
|
|
22
|
+
profile = ameva.detect_hardware()
|
|
23
|
+
print(f"Target: {profile.soc_name} | {profile.gpu_vendor}")
|
|
24
|
+
|
|
25
|
+
# Run diagnostics
|
|
26
|
+
doc = vulkan.Doctor()
|
|
27
|
+
report = doc.run_self_test()
|
|
28
|
+
print(f"Passed: {report.passed_stages}/{report.total_stages}")
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Empirical Benchmarks
|
|
32
|
+
|
|
33
|
+
- **Galaxy S25 (Adreno 830)**: LLM 35.80 t/s (VRAM 25/25 layers), Whisper STT 4,401 ms, TTS RTF 0.264x (medium) / 0.993x (high-fp16).
|
|
34
|
+
- **Galaxy A35 (Mali-G68 MP5)**: LLM 4.44 t/s (+26.9% vs NEON), Whisper STT 360.60s (2.26x speedup), TTS RTF 1.146x.
|
|
35
|
+
|
|
36
|
+
## Documentation
|
|
37
|
+
- [Official Documentation](https://uno-km.vercel.app/lib/vulkan/)
|
|
38
|
+
- [GitHub Repository](https://github.com/uno-km/ameva-runtime)
|
|
39
|
+
|
|
40
|
+
## License
|
|
41
|
+
Apache-2.0 License. Copyright (c) 2026 Eunho Kim (@uno-km).
|
|
@@ -40,19 +40,19 @@ Repository = "https://github.com/uno-km/ameva-runtime"
|
|
|
40
40
|
Documentation = "https://uno-km.vercel.app/lib/vulkan/"
|
|
41
41
|
|
|
42
42
|
[project.optional-dependencies]
|
|
43
|
-
stt = ["termux-stt>=1.
|
|
44
|
-
diffusion = ["termux-diffusion>=1.
|
|
45
|
-
bitnet = ["termux-bitnet>=1.
|
|
46
|
-
llamacpp = ["termux-llamacpp>=1.
|
|
47
|
-
tts = ["termux-tts>=1.
|
|
48
|
-
vision = ["termux-vision>=1.
|
|
43
|
+
stt = ["termux-stt>=1.2.0"]
|
|
44
|
+
diffusion = ["termux-diffusion>=1.5.0"]
|
|
45
|
+
bitnet = ["termux-bitnet>=1.2.0"]
|
|
46
|
+
llamacpp = ["termux-llamacpp>=1.3.0"]
|
|
47
|
+
tts = ["termux-tts>=1.4.0"]
|
|
48
|
+
vision = ["termux-vision>=1.2.0"]
|
|
49
49
|
all = [
|
|
50
|
-
"termux-stt>=1.
|
|
51
|
-
"termux-diffusion>=1.
|
|
52
|
-
"termux-bitnet>=1.
|
|
53
|
-
"termux-llamacpp>=1.
|
|
54
|
-
"termux-tts>=1.
|
|
55
|
-
"termux-vision>=1.
|
|
50
|
+
"termux-stt>=1.2.0",
|
|
51
|
+
"termux-diffusion>=1.5.0",
|
|
52
|
+
"termux-bitnet>=1.2.0",
|
|
53
|
+
"termux-llamacpp>=1.3.0",
|
|
54
|
+
"termux-tts>=1.4.0",
|
|
55
|
+
"termux-vision>=1.2.0",
|
|
56
56
|
]
|
|
57
57
|
|
|
58
58
|
[tool.pytest.ini_options]
|
|
@@ -11,7 +11,19 @@ Re-exports individual modality adapters for:
|
|
|
11
11
|
"""
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
|
-
from .base import
|
|
14
|
+
from .base import (
|
|
15
|
+
_is_vulkan_report,
|
|
16
|
+
_make_cpu_fallback,
|
|
17
|
+
_make_cpu_binding,
|
|
18
|
+
check_vulkan_availability_or_raise,
|
|
19
|
+
_ADRENO_VENDOR_ID,
|
|
20
|
+
_MALI_VENDOR_ID,
|
|
21
|
+
BindingResult,
|
|
22
|
+
BaseAdapter,
|
|
23
|
+
resolve_diagnostic_report,
|
|
24
|
+
get_vulkan_env,
|
|
25
|
+
find_system_vulkan_driver_dir,
|
|
26
|
+
)
|
|
15
27
|
from .bitnet import BitnetAdapter
|
|
16
28
|
from .diffusion import DiffusionAdapter
|
|
17
29
|
from .llamacpp import LlamaCppAdapter
|
|
@@ -20,6 +32,8 @@ from .tts import TtsAdapter
|
|
|
20
32
|
from .vision import VisionAdapter
|
|
21
33
|
|
|
22
34
|
__all__ = [
|
|
35
|
+
"BaseAdapter",
|
|
36
|
+
"resolve_diagnostic_report",
|
|
23
37
|
"SttAdapter",
|
|
24
38
|
"DiffusionAdapter",
|
|
25
39
|
"BitnetAdapter",
|
|
@@ -31,4 +45,7 @@ __all__ = [
|
|
|
31
45
|
"find_system_vulkan_driver_dir",
|
|
32
46
|
"_is_vulkan_report",
|
|
33
47
|
"_make_cpu_fallback",
|
|
48
|
+
"_make_cpu_binding",
|
|
49
|
+
"check_vulkan_availability_or_raise",
|
|
34
50
|
]
|
|
51
|
+
|
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Base Utilities & Common Logic for Ameva Modality Adapters
|
|
3
|
+
"""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import logging
|
|
7
|
+
from typing import Any, Optional
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
from ..doctor import DiagnosticReport
|
|
11
|
+
except ImportError:
|
|
12
|
+
DiagnosticReport = Any
|
|
13
|
+
|
|
14
|
+
from ameva_runtime.protocol import BindingResult
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger("ameva_vulkan_runtime.adapters")
|
|
17
|
+
|
|
18
|
+
_ADRENO_VENDOR_ID = 0x5143
|
|
19
|
+
_MALI_VENDOR_ID = 0x13B5
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def resolve_diagnostic_report(report: Any = None, profile: Any = None) -> DiagnosticReport:
|
|
23
|
+
"""Resolves DiagnosticReport from either an existing report, a HardwareProfile, or Doctor auto-probe."""
|
|
24
|
+
if report is not None:
|
|
25
|
+
return report
|
|
26
|
+
if profile is not None:
|
|
27
|
+
is_vk = (getattr(profile, "recommended_backend", "") == "vulkan" and not getattr(profile, "hardware_hazard", None))
|
|
28
|
+
gpu_family = getattr(profile, "gpu_family", "generic")
|
|
29
|
+
return DiagnosticReport(
|
|
30
|
+
device_name=getattr(profile, "soc_model", None) or gpu_family,
|
|
31
|
+
vendor_id=_MALI_VENDOR_ID if gpu_family == "mali" else _ADRENO_VENDOR_ID,
|
|
32
|
+
overall_success=is_vk,
|
|
33
|
+
recommended_backend=getattr(profile, "recommended_backend", "cpu_neon"),
|
|
34
|
+
passed_stages=12 if is_vk else 0,
|
|
35
|
+
total_stages=12,
|
|
36
|
+
loader_path="/system/lib64/libvulkan.so",
|
|
37
|
+
hazard=getattr(profile, "hardware_hazard", None),
|
|
38
|
+
allowed_cpus=sorted(list(getattr(profile, "allowed_cpu_set", []))),
|
|
39
|
+
diagnosis_reason=getattr(profile, "diagnosis_reason", ""),
|
|
40
|
+
)
|
|
41
|
+
try:
|
|
42
|
+
from ..doctor import Doctor
|
|
43
|
+
return Doctor().run_diagnostics()
|
|
44
|
+
except Exception as err:
|
|
45
|
+
logger.debug("[ameva-runtime:adapters] Doctor probe fallback to minimal report: %s", err)
|
|
46
|
+
return DiagnosticReport(
|
|
47
|
+
device_name="Unknown",
|
|
48
|
+
vendor_id=0,
|
|
49
|
+
overall_success=False,
|
|
50
|
+
recommended_backend="cpu_neon",
|
|
51
|
+
passed_stages=0,
|
|
52
|
+
total_stages=12,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class BaseAdapter:
|
|
57
|
+
"""Base class for all ameva modality adapters providing common diagnostic helpers."""
|
|
58
|
+
|
|
59
|
+
@staticmethod
|
|
60
|
+
def resolve_diagnostic_report(report: Any = None, profile: Any = None) -> DiagnosticReport:
|
|
61
|
+
return resolve_diagnostic_report(report, profile)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _is_vulkan_report(report: Optional[DiagnosticReport]) -> bool:
|
|
65
|
+
if report is None:
|
|
66
|
+
return False
|
|
67
|
+
if not report.device_name or report.device_name in ("Unknown", "None", ""):
|
|
68
|
+
return False
|
|
69
|
+
return bool(
|
|
70
|
+
report.overall_success
|
|
71
|
+
or report.recommended_backend in ("vulkan", "vulkan_driver_only")
|
|
72
|
+
or report.passed_stages >= 7
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def check_vulkan_availability_or_raise(
|
|
77
|
+
module: str,
|
|
78
|
+
report: Any,
|
|
79
|
+
is_vk: bool,
|
|
80
|
+
requested_backend: Optional[str] = None,
|
|
81
|
+
) -> None:
|
|
82
|
+
"""Enforces Fail-Fast: If Vulkan was explicitly requested but is unavailable, raise PlatformNotSupportedError."""
|
|
83
|
+
if requested_backend == "vulkan" and not is_vk:
|
|
84
|
+
device = getattr(report, "device_name", None) or "Unknown"
|
|
85
|
+
diag_reason = getattr(report, "diagnosis_reason", None) or "Vulkan ICD / driver missing or self-test validation failed"
|
|
86
|
+
from ..exceptions import PlatformNotSupportedError
|
|
87
|
+
raise PlatformNotSupportedError(
|
|
88
|
+
f"[{module}] Vulkan acceleration backend explicitly requested, but no valid Vulkan driver/environment is available.\n"
|
|
89
|
+
f"Target Device: {device}\n"
|
|
90
|
+
f"Failure Cause: {diag_reason}\n"
|
|
91
|
+
f"Remediation: Verify Vulkan driver installation or explicitly specify '--device cpu'."
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _make_cpu_binding(
|
|
96
|
+
module: str,
|
|
97
|
+
report: DiagnosticReport,
|
|
98
|
+
config: dict,
|
|
99
|
+
reason: str = "",
|
|
100
|
+
) -> BindingResult:
|
|
101
|
+
"""Creates a standardized CPU NEON BindingResult when CPU backend is explicitly requested or routed."""
|
|
102
|
+
device = getattr(report, "device_name", None) or "Generic CPU"
|
|
103
|
+
msg = f"[ameva-runtime:{module}] Binding CPU NEON backend (Device: {device}"
|
|
104
|
+
if reason:
|
|
105
|
+
msg += f", Reason: {reason}"
|
|
106
|
+
msg += ")"
|
|
107
|
+
logger.info(msg)
|
|
108
|
+
config["backend"] = "cpu_neon"
|
|
109
|
+
return BindingResult(
|
|
110
|
+
module=module,
|
|
111
|
+
backend="cpu_neon",
|
|
112
|
+
is_vulkan=False,
|
|
113
|
+
device_name=getattr(report, "device_name", "CPU"),
|
|
114
|
+
vendor_id=getattr(report, "vendor_id", 0),
|
|
115
|
+
config=config,
|
|
116
|
+
status="BOUND_CPU_NEON",
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# Backward compatibility alias
|
|
121
|
+
_make_cpu_fallback = _make_cpu_binding
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _get_optimal_threads() -> int:
|
|
125
|
+
"""Returns optimal threads count for big/performance cores."""
|
|
126
|
+
import os
|
|
127
|
+
cpu_count = os.cpu_count() or 8
|
|
128
|
+
return max(1, min(4, cpu_count // 2 if cpu_count > 4 else cpu_count))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def find_system_vulkan_driver_dir() -> Optional[str]:
|
|
132
|
+
"""
|
|
133
|
+
Dynamically probes and returns the absolute directory path of the valid libvulkan.so driver installed on the system.
|
|
134
|
+
Tier 1: Directly inspect /proc/self/maps Linux kernel virtual memory mapping (Ground Truth, 0% root required).
|
|
135
|
+
Tier 2: Probe standard 64-bit / 32-bit Android HAL directories.
|
|
136
|
+
"""
|
|
137
|
+
import sys
|
|
138
|
+
import os
|
|
139
|
+
import ctypes
|
|
140
|
+
from pathlib import Path
|
|
141
|
+
|
|
142
|
+
# Tier 1: Linux kernel process virtual memory map inspection (Ground Truth)
|
|
143
|
+
try:
|
|
144
|
+
handle = ctypes.CDLL("libvulkan.so")
|
|
145
|
+
if hasattr(handle, "vkCreateInstance"):
|
|
146
|
+
maps_path = Path("/proc/self/maps")
|
|
147
|
+
if maps_path.is_file():
|
|
148
|
+
with open(maps_path, "r", encoding="utf-8", errors="ignore") as f:
|
|
149
|
+
for line in f:
|
|
150
|
+
if "libvulkan.so" in line:
|
|
151
|
+
parts = line.strip().split()
|
|
152
|
+
if len(parts) >= 6:
|
|
153
|
+
real_path = parts[-1]
|
|
154
|
+
if os.path.isabs(real_path) and os.path.exists(real_path):
|
|
155
|
+
return str(Path(real_path).parent.resolve())
|
|
156
|
+
except Exception:
|
|
157
|
+
pass
|
|
158
|
+
|
|
159
|
+
# Tier 2: Search standard 64-bit vs 32-bit Android HAL directories
|
|
160
|
+
is_64bit = sys.maxsize > 2**32
|
|
161
|
+
|
|
162
|
+
if is_64bit:
|
|
163
|
+
candidate_dirs = [
|
|
164
|
+
Path("/system/lib64"),
|
|
165
|
+
Path("/vendor/lib64"),
|
|
166
|
+
Path("/apex/com.android.runtime/lib64"),
|
|
167
|
+
Path("/system/vendor/lib64"),
|
|
168
|
+
]
|
|
169
|
+
else:
|
|
170
|
+
candidate_dirs = [
|
|
171
|
+
Path("/system/lib"),
|
|
172
|
+
Path("/vendor/lib"),
|
|
173
|
+
Path("/apex/com.android.runtime/lib"),
|
|
174
|
+
Path("/system/vendor/lib"),
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
for root in candidate_dirs:
|
|
178
|
+
so_path = root / "libvulkan.so"
|
|
179
|
+
if so_path.is_file():
|
|
180
|
+
try:
|
|
181
|
+
h = ctypes.CDLL(str(so_path))
|
|
182
|
+
if hasattr(h, "vkCreateInstance"):
|
|
183
|
+
return str(root.resolve())
|
|
184
|
+
except Exception:
|
|
185
|
+
continue
|
|
186
|
+
|
|
187
|
+
# Tier 3: Simple file existence probe
|
|
188
|
+
for root in candidate_dirs:
|
|
189
|
+
if (root / "libvulkan.so").is_file():
|
|
190
|
+
return str(root.resolve())
|
|
191
|
+
|
|
192
|
+
return None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def get_vulkan_env(base_env: Optional[dict[str, str]] = None) -> dict[str, str]:
|
|
196
|
+
"""
|
|
197
|
+
Returns environment variable dictionary for the AMEVA Vulkan runtime.
|
|
198
|
+
Golden Link Order:
|
|
199
|
+
1: System Vulkan driver directory (/system/lib64)
|
|
200
|
+
2: Termux native C++ runtime (/data/data/com.termux/files/usr/lib)
|
|
201
|
+
3: Bundled runtime library (~/.termux-llama/current/lib)
|
|
202
|
+
4: Existing system and user environment paths
|
|
203
|
+
"""
|
|
204
|
+
import os
|
|
205
|
+
from pathlib import Path
|
|
206
|
+
|
|
207
|
+
env = dict(base_env or os.environ)
|
|
208
|
+
current_ld = env.get("LD_LIBRARY_PATH", "")
|
|
209
|
+
|
|
210
|
+
# Dynamically probe verified Vulkan driver directory
|
|
211
|
+
discovered_driver_dir = find_system_vulkan_driver_dir()
|
|
212
|
+
|
|
213
|
+
ordered_dirs = []
|
|
214
|
+
|
|
215
|
+
# Priority 1: Smartphone system Vulkan driver directory (/system/lib64)
|
|
216
|
+
# Prevents Android 15 Bionic libunwindstack symbol collision (system liblzma binds first)
|
|
217
|
+
if discovered_driver_dir and discovered_driver_dir not in ordered_dirs:
|
|
218
|
+
ordered_dirs.append(discovered_driver_dir)
|
|
219
|
+
elif os.path.isdir("/system/lib64") and "/system/lib64" not in ordered_dirs:
|
|
220
|
+
ordered_dirs.append("/system/lib64")
|
|
221
|
+
|
|
222
|
+
# Priority 2: Termux native C++ runtime
|
|
223
|
+
termux_usr_lib = "/data/data/com.termux/files/usr/lib"
|
|
224
|
+
if os.path.isdir(termux_usr_lib) and termux_usr_lib not in ordered_dirs:
|
|
225
|
+
ordered_dirs.append(termux_usr_lib)
|
|
226
|
+
|
|
227
|
+
# Priority 3: Bundled llama/runtime libraries
|
|
228
|
+
llama_lib = str(Path.home() / ".termux-llama/current/lib")
|
|
229
|
+
if os.path.isdir(llama_lib) and llama_lib not in ordered_dirs:
|
|
230
|
+
ordered_dirs.append(llama_lib)
|
|
231
|
+
|
|
232
|
+
# Priority 4: Append remaining paths without duplication
|
|
233
|
+
existing_parts = [p for p in current_ld.split(":") if p]
|
|
234
|
+
final_dirs = ordered_dirs + [p for p in existing_parts if p not in ordered_dirs]
|
|
235
|
+
env["LD_LIBRARY_PATH"] = ":".join(final_dirs)
|
|
236
|
+
|
|
237
|
+
# Mali GPU Quirks: Avoid infinite GEMM quantization loops on ARM Mali Valhall architectures
|
|
238
|
+
try:
|
|
239
|
+
from ..platform import detect_soc_environment
|
|
240
|
+
soc = detect_soc_environment()
|
|
241
|
+
if "mali" in str(soc.gpu_family).lower() or soc.vendor == "samsung":
|
|
242
|
+
env.setdefault("GGML_VK_FORCE_MEDIUM_MATMUL", "1")
|
|
243
|
+
env.setdefault("GGML_VK_DISABLE_F16", "1")
|
|
244
|
+
except Exception:
|
|
245
|
+
pass
|
|
246
|
+
|
|
247
|
+
return env
|