termux-vision 1.3.1 → 1.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -16
- package/README.pypi.md +11 -10
- package/bin/cli.js +36 -372
- package/lib/vlm.js +7 -5
- package/package.json +3 -5
package/README.md
CHANGED
|
@@ -60,8 +60,8 @@ pkg install -y python nodejs clang make cmake git termux-api wget vulkan-loader
|
|
|
60
60
|
|
|
61
61
|
* **Option B: Direct GitHub Releases Wheel Asset (SSOT Verified)**:
|
|
62
62
|
```bash
|
|
63
|
-
# Download and install the prebuilt v1.
|
|
64
|
-
pip install https://github.com/uno-km/termux-vision/releases/download/v1.
|
|
63
|
+
# Download and install the prebuilt v1.4.1 release wheel
|
|
64
|
+
pip install https://github.com/uno-km/termux-vision/releases/download/v1.4.1/termux_vision-1.4.1-py3-none-any.whl
|
|
65
65
|
```
|
|
66
66
|
|
|
67
67
|
### 2.3 Node.js / TypeScript CLI Installation
|
|
@@ -98,8 +98,8 @@ npm install -g termux-vision @ameva/runtime
|
|
|
98
98
|
|
|
99
99
|
| GPU Microarchitecture | Silicon / SoC Reference | Status | Optimization Mechanics |
|
|
100
100
|
| :--- | :--- | :--- | :--- |
|
|
101
|
+
| **Qualcomm Adreno GPU** | Snapdragon 8 Elite (Adreno 830)<br>Snapdragon 8 Gen 1/2/3 (Adreno 730-750) | 🟢 **Production Verified** | Direct Bionic ICD binding, SPIR-V JIT patch (`mul_mat_vec_max_cols = 2`), KGSL Watchdog defense (`GGML_VULKAN_SKIP_CHECKS="999999999"`), micro-batch prefill chunking (`-b 64 -ub 64`). |
|
|
101
102
|
| **ARM Mali GPU** | Exynos 2100 (Mali-G78 MP14)<br>Exynos 1380 (Mali-G68 MP5) | 🟢 **Production Verified** | Bionic Vulkan ICD binding, Tile-Based Deferred Rendering (TBDR) memory isolation, MMVQ matrix-vector kernel dispatch (`--tune-mali`). |
|
|
102
|
-
| **Qualcomm Adreno GPU** | Snapdragon 8 Gen 1/2/3/Elite<br>(Adreno 730 / 740 / 750 / 830) | 🟡 **In Development (개발 진행 중)** | Direct Bionic ICD & Freedreno/Turnip dispatch layers under active engineering. |
|
|
103
103
|
| **Samsung Xclipse GPU** | Exynos 2200 / 2400<br>(Xclipse 920 / 940 - AMD RDNA) | 🟡 **In Development (개발 진행 중)** | SPIR-V instruction scheduling and RDNA mobile shader alignment under active engineering. |
|
|
104
104
|
|
|
105
105
|
Verify GPU driver detection and hardware readiness:
|
|
@@ -223,19 +223,48 @@ termux-vision doctor
|
|
|
223
223
|
|
|
224
224
|
## 7. Real-World Benchmarks & Hardware Scorecard
|
|
225
225
|
|
|
226
|
-
All metrics represent deterministic ground-truth measurements obtained on physical test devices running Android Termux unrooted,
|
|
227
|
-
|
|
228
|
-
| Target Device | SoC
|
|
229
|
-
| :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- |
|
|
230
|
-
| **Samsung Galaxy
|
|
231
|
-
| Samsung Galaxy S21 5G | Exynos 2100<br>
|
|
232
|
-
|
|
|
233
|
-
| Samsung Galaxy A35 5G | Exynos 1380<br>
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
1
|
|
237
|
-
|
|
238
|
-
|
|
226
|
+
All metrics represent deterministic ground-truth measurements obtained on physical test devices running Android Termux unrooted, comparing **Moondream2 1.8B f16** and **SmolVLM-500M Instruct (Q4_K_M + Q8_0 mmproj)**.
|
|
227
|
+
|
|
228
|
+
| Target Device | SoC & GPU Architecture | Model Architecture | Precision & Weights | Mode | Prompt Processing | Token Generation | Total Latency | Mapped CPU VRAM | Vulkan GPU VRAM | Status / Speedup |
|
|
229
|
+
| :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- |
|
|
230
|
+
| **Samsung Galaxy S25** | Snapdragon 8 Elite<br>Qualcomm Adreno 830 | **Moondream2 1.8B** | Text f16 (2.7GB)<br>+ ViT f16 (868MB) | **GPU (Vulkan 25/25)** | **19.84 tok/s** | **15.00 tok/s** | **39.20 s** | **0.00 MiB** | **2,706.00 MiB** | **Production Verified** |
|
|
231
|
+
| **Samsung Galaxy S21 5G** | Exynos 2100<br>ARM Mali-G78 MP14 | SmolVLM-500M-Instruct | Q4_K_M (350MB)<br>+ Q8_0 mmproj (200MB) | **GPU (Vulkan)** | **14.28 tok/s** | **12.65 tok/s** | **21.26 s** | **0.00 MiB** | **1,059.02 MiB** | **+58.9% vs CPU** |
|
|
232
|
+
| Samsung Galaxy S21 5G | Exynos 2100<br>8-Core ARMv8.2-A CPU | SmolVLM-500M-Instruct | Q4_K_M (350MB)<br>+ Q8_0 mmproj (200MB) | CPU (NEON) | 8.84 tok/s | 7.96 tok/s | 34.22 s | 1,059.02 MiB | 0.00 MiB | Baseline |
|
|
233
|
+
| **Samsung Galaxy A35 5G** | Exynos 1380<br>ARM Mali-G68 MP5 | SmolVLM-500M-Instruct | Q4_K_M (350MB)<br>+ Q8_0 mmproj (200MB) | **GPU (Vulkan)** | **5.67 tok/s** | **5.47 tok/s** | **52.57 s** | **0.00 MiB** | **1,059.02 MiB** | **+55.8% vs CPU** |
|
|
234
|
+
| Samsung Galaxy A35 5G | Exynos 1380<br>8-Core ARMv8.2-A CPU | SmolVLM-500M-Instruct | Q4_K_M (350MB)<br>+ Q8_0 mmproj (200MB) | CPU (NEON) | 4.88 tok/s | 3.51 tok/s | 65.34 s | 1,059.02 MiB | 0.00 MiB | Baseline |
|
|
235
|
+
|
|
236
|
+
### 7.1 Empirical Visual Question Answering Verification (Galaxy S25 Adreno 830)
|
|
237
|
+
|
|
238
|
+
#### Test Case A: Geometric & Spatial Reasoning (`test_shapes_224x224.png`)
|
|
239
|
+
* **Input Image**: Clean canvas with primary geometric primitives (triangles, rectangle).
|
|
240
|
+
* **Prompt**: `"Describe the colors and geometric shapes visible in this image."`
|
|
241
|
+
* **Model**: `Moondream2 1.8B` (f16 text + f16 ViT mmproj, 2,706 MiB VRAM)
|
|
242
|
+
* **Exact Ground-Truth Output**:
|
|
243
|
+
> *"The image features a white background with three distinct geometric shapes: two triangles and one rectangle..."*
|
|
244
|
+
* **Execution Metrics**:
|
|
245
|
+
- GPU Layers Offloaded: **25 / 25 (100% Full GPU)**
|
|
246
|
+
- Vulkan VRAM Allocated: **2,706.00 MiB** (CPU Mapped VRAM: **0.00 MiB**)
|
|
247
|
+
- Prompt Evaluation: **19.60 tokens/sec** (749 tokens in 38,211 ms)
|
|
248
|
+
- Token Generation: **14.97 tokens/sec** (39 tokens in 2,605 ms, 66.80 ms/tok)
|
|
249
|
+
- Total Execution Time: **51.24s** (Cold weights loading: 10.4s, CPU ViT projection: 14.1s, GPU decoding: 2.6s)
|
|
250
|
+
- Exit Status: `Exit Code 0`
|
|
251
|
+
|
|
252
|
+
#### Test Case B: Photorealistic Real-World Scene (`test_elephant_gpu_step2.png`)
|
|
253
|
+
* **Input Image**: Diffusion-synthesized high-detail photorealistic scene.
|
|
254
|
+
* **Prompt**: `"What animal is this and what is it doing?"`
|
|
255
|
+
* **Model**: `Moondream2 1.8B` (f16 text + f16 ViT mmproj)
|
|
256
|
+
* **Exact Ground-Truth Output**:
|
|
257
|
+
> *"The image shows a large elephant riding on top of a surfboard in the ocean."*
|
|
258
|
+
* **Execution Metrics**:
|
|
259
|
+
- Prompt Processing: **19.84 tokens/sec** (747 tokens in 37,651 ms)
|
|
260
|
+
- Generation Speed: **15.00 tokens/sec** (24 tokens in 1,600 ms, 66.67 ms/tok)
|
|
261
|
+
- KGSL Watchdog State: Fully stabilized via `GGML_VULKAN_SKIP_CHECKS="999999999"` (No `ErrorDeviceLost`)
|
|
262
|
+
- Exit Status: `Exit Code 0`
|
|
263
|
+
|
|
264
|
+
### 7.2 Key Architectural Discoveries
|
|
265
|
+
1. **Adreno 830 SPIR-V JIT Fix**: Qualcomm's new compiler fails during unrolled vector compilation with `mul_mat_vec_max_cols = 8` (`VK_ERROR_UNKNOWN`). Reducing this parameter to `2` eliminates register spilling and enables full 25/25 layer GPU offloading.
|
|
266
|
+
2. **Android KGSL Watchdog Defense**: Prefilling 729 vision tokens in mobile Vulkan exceeds the 5-second kernel watchdog timer unless micro-batched. Configuring `-b 64 -ub 64` and injecting `GGML_VULKAN_SKIP_CHECKS="999999999"` slices prefill into 1.8s units, preventing `ErrorDeviceLost`.
|
|
267
|
+
3. **Pure GPU Isolation (0.00 MiB CPU VRAM)**: Across both Qualcomm Adreno 830 and ARM Mali (G78/G68), all tensor weights and KV cache reside strictly in Vulkan GPU memory.
|
|
239
268
|
|
|
240
269
|
---
|
|
241
270
|
|
package/README.pypi.md
CHANGED
|
@@ -31,7 +31,7 @@ pip install termux-vision
|
|
|
31
31
|
|
|
32
32
|
### 2.2 Direct GitHub Releases Wheel Asset
|
|
33
33
|
```bash
|
|
34
|
-
pip install https://github.com/uno-km/termux-vision/releases/download/v1.
|
|
34
|
+
pip install https://github.com/uno-km/termux-vision/releases/download/v1.4.1/termux_vision-1.4.1-py3-none-any.whl
|
|
35
35
|
```
|
|
36
36
|
|
|
37
37
|
### 2.3 One-Line Bootstrap Installer
|
|
@@ -48,9 +48,9 @@ pip install termux-vision ameva-runtime termux-llamacpp
|
|
|
48
48
|
```
|
|
49
49
|
|
|
50
50
|
### Silicon Architecture Support Status
|
|
51
|
+
* **Qualcomm Adreno GPU (Snapdragon 8 Elite / Adreno 830, Adreno 7xx)**: Production Verified & Supported (Full 25/25 layer GPU offloading, 15.00 tok/s on Moondream2 1.8B f16, SPIR-V JIT patch, KGSL watchdog defense via `GGML_VULKAN_SKIP_CHECKS="999999999"`).
|
|
51
52
|
* **ARM Mali GPU (Mali-G78, Mali-G68, etc.)**: Production Verified & Supported (Pure GPU offloading, 0.00 MiB CPU Mapped VRAM, MMVQ tuning via `--tune-mali`).
|
|
52
|
-
* **
|
|
53
|
-
* **Samsung Xclipse GPU (Xclipse 920 / 940 - AMD RDNA)**: Under Active Development (In Progress / 개발 진행 중).
|
|
53
|
+
* **Samsung Xclipse GPU (Xclipse 920 / 940 - AMD RDNA)**: Under Active Engineering (In Progress / 개발 진행 중).
|
|
54
54
|
|
|
55
55
|
Run hardware diagnostics:
|
|
56
56
|
```bash
|
|
@@ -105,14 +105,15 @@ with tv.vlm.load("smolvlm-500m", device="gpu") as engine:
|
|
|
105
105
|
|
|
106
106
|
---
|
|
107
107
|
|
|
108
|
-
## 6. Real-World Benchmarks
|
|
108
|
+
## 6. Real-World Benchmarks & Hardware Scorecard
|
|
109
109
|
|
|
110
|
-
| Target Device | SoC
|
|
111
|
-
| :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- |
|
|
112
|
-
| **Samsung Galaxy
|
|
113
|
-
| Samsung Galaxy S21 5G | Exynos 2100 /
|
|
114
|
-
|
|
|
115
|
-
| Samsung Galaxy A35 5G | Exynos 1380 /
|
|
110
|
+
| Target Device | SoC & GPU | Model Architecture | Mode | Prompt Processing | Token Generation | Mapped CPU VRAM | Vulkan GPU VRAM | Status / Speedup |
|
|
111
|
+
| :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- | :--- |
|
|
112
|
+
| **Samsung Galaxy S25** | Snapdragon 8 Elite / Adreno 830 | **Moondream2 1.8B f16** | **GPU (Vulkan 25/25)** | **19.84 tok/s** | **15.00 tok/s** | **0.00 MiB** | **2,706.00 MiB** | **Verified (Full GPU)** |
|
|
113
|
+
| **Samsung Galaxy S21 5G** | Exynos 2100 / Mali-G78 | SmolVLM-500M-Instruct | **GPU (Vulkan)** | **14.28 tok/s** | **12.65 tok/s** | **0.00 MiB** | **1,059.02 MiB** | **+58.9% vs CPU** |
|
|
114
|
+
| Samsung Galaxy S21 5G | Exynos 2100 / 8-Core CPU | SmolVLM-500M-Instruct | CPU (NEON) | 8.84 tok/s | 7.96 tok/s | 1,059.02 MiB | 0.00 MiB | Baseline |
|
|
115
|
+
| **Samsung Galaxy A35 5G** | Exynos 1380 / Mali-G68 | SmolVLM-500M-Instruct | **GPU (Vulkan)** | **5.67 tok/s** | **5.47 tok/s** | **0.00 MiB** | **1,059.02 MiB** | **+55.8% vs CPU** |
|
|
116
|
+
| Samsung Galaxy A35 5G | Exynos 1380 / 8-Core CPU | SmolVLM-500M-Instruct | CPU (NEON) | 4.88 tok/s | 3.51 tok/s | 1,059.02 MiB | 0.00 MiB | Baseline |
|
|
116
117
|
|
|
117
118
|
---
|
|
118
119
|
|
package/bin/cli.js
CHANGED
|
@@ -1,387 +1,51 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
|
|
3
2
|
/**
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
* Open-Source under Apache License 2.0.
|
|
3
|
+
* AMEVA Standard Node.js CLI Runner for termux_vision.
|
|
4
|
+
* Automatically resolves Python 3 environment and dispatches to python -m termux_vision.
|
|
7
5
|
*/
|
|
8
|
-
|
|
9
|
-
'use strict';
|
|
10
|
-
|
|
11
6
|
const fs = require('fs');
|
|
12
|
-
const
|
|
13
|
-
const os = require('os');
|
|
14
|
-
const readline = require('readline');
|
|
15
|
-
|
|
16
|
-
const { version } = require('../package.json');
|
|
17
|
-
const { runDoctor } = require('../lib/doctor');
|
|
18
|
-
const { ModelCacheManager, CATALOG } = require('../lib/cache');
|
|
19
|
-
const { load } = require('../lib/vlm');
|
|
20
|
-
const { canny } = require('../lib/cv');
|
|
21
|
-
const {
|
|
22
|
-
ModelNotFoundError,
|
|
23
|
-
NoInstalledModelsError,
|
|
24
|
-
ModelSelectionRequiredError,
|
|
25
|
-
VulkanNotAvailableError,
|
|
26
|
-
RuntimeNotFoundError,
|
|
27
|
-
ModelDownloadError
|
|
28
|
-
} = require('../lib/errors');
|
|
29
|
-
|
|
30
|
-
const args = process.argv.slice(2);
|
|
31
|
-
const command = args[0] || '--help';
|
|
7
|
+
const { spawn, execSync } = require('child_process');
|
|
32
8
|
|
|
33
|
-
function
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
return idx !== -1 && args[idx + 1] ? args[idx + 1] : null;
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
function hasFlag(flag, alias = null) {
|
|
40
|
-
return args.includes(flag) || (alias && args.includes(alias));
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
function printHelp() {
|
|
44
|
-
console.log(`termux-vision CLI v${version} (Node.js Engine)`);
|
|
45
|
-
console.log('Usage: termux-vision <command> [options]\n');
|
|
46
|
-
console.log('Commands:');
|
|
47
|
-
console.log(' doctor Inspect device hardware, RAM, and Vulkan GPU');
|
|
48
|
-
console.log(' model list List installed VLM models');
|
|
49
|
-
console.log(' model install <model_id> Download and install official VLM model');
|
|
50
|
-
console.log(' model download <url_or_repo> Freely download model from Hugging Face or direct URL');
|
|
51
|
-
console.log(' model remove <model_id> Remove model from cache');
|
|
52
|
-
console.log(' vlm <image_path> [options] Execute multimodal image description/chat');
|
|
53
|
-
console.log(' canny <image_path> [options] Run Canny edge detection');
|
|
54
|
-
console.log(' benchmark Run on-device vision latency benchmark\n');
|
|
55
|
-
console.log('Options for VLM:');
|
|
56
|
-
console.log(' -p, --prompt <text> Prompt query');
|
|
57
|
-
console.log(' -m, --model <model_id_or_path> Model ID or direct path to .gguf file');
|
|
58
|
-
console.log(' --mmproj <path> Vision projector path (mmproj-*.gguf)');
|
|
59
|
-
console.log(' --device <auto|cpu|vulkan|gpu> Device backend (auto: Vulkan with CPU fallback; gpu: strict Vulkan)');
|
|
60
|
-
console.log(' --runtime <path> Explicit path to llama-cli executable');
|
|
61
|
-
console.log(' --allow-download Automatically download model if missing');
|
|
62
|
-
console.log(' -t, --threads <num> Inference threads (default: 4)');
|
|
63
|
-
console.log(' -n, --max-tokens <num> Maximum generated tokens (default: 150)');
|
|
64
|
-
console.log(' --temp, --temperature <val> Sampling temperature (default: 0.2)');
|
|
65
|
-
console.log(' --top-p <val> Top-p nucleus sampling');
|
|
66
|
-
console.log(' --top-k <num> Top-k sampling threshold');
|
|
67
|
-
console.log(' --repeat-penalty <val> Repetition penalty (default: 1.2)');
|
|
68
|
-
console.log(' --seed <num> Random RNG seed');
|
|
69
|
-
console.log(' --system-prompt <text> System prompt context');
|
|
70
|
-
console.log(' --ngl <num> Number of GPU offload layers');
|
|
71
|
-
console.log(' --json Output full metrics in JSON format');
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
async function promptUser(question) {
|
|
75
|
-
const rl = readline.createInterface({ input: process.stdin, output: process.stderr });
|
|
76
|
-
return new Promise((resolve) => {
|
|
77
|
-
rl.question(question, (ans) => {
|
|
78
|
-
rl.close();
|
|
79
|
-
resolve(ans.trim().toLowerCase());
|
|
80
|
-
});
|
|
81
|
-
});
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
async function main() {
|
|
85
|
-
if (hasFlag('-v') || hasFlag('--version')) {
|
|
86
|
-
console.log(`termux-vision ${version}`);
|
|
87
|
-
process.exit(0);
|
|
9
|
+
function findPython() {
|
|
10
|
+
if (process.env.PYTHON && fs.existsSync(process.env.PYTHON)) {
|
|
11
|
+
return process.env.PYTHON;
|
|
88
12
|
}
|
|
89
|
-
|
|
90
|
-
if (
|
|
91
|
-
|
|
92
|
-
process.exit(0);
|
|
13
|
+
const termuxBin = '/data/data/com.termux/files/usr/bin/python3';
|
|
14
|
+
if (fs.existsSync(termuxBin)) {
|
|
15
|
+
return termuxBin;
|
|
93
16
|
}
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
if (command === 'doctor') {
|
|
98
|
-
const probeVulkan = hasFlag('--probe-vulkan');
|
|
99
|
-
const isJson = hasFlag('--json');
|
|
100
|
-
const rep = runDoctor(probeVulkan);
|
|
101
|
-
|
|
102
|
-
if (isJson) {
|
|
103
|
-
console.log(JSON.stringify(rep, null, 2));
|
|
104
|
-
} else {
|
|
105
|
-
console.log('=== termux-vision Diagnostic Doctor (Node.js Engine) ===');
|
|
106
|
-
console.log(` Platform : ${rep.platform.system} (${rep.platform.machine}) | Android: ${rep.platform.isAndroid}`);
|
|
107
|
-
console.log(` RAM : Total ${rep.hardware.totalRamMb}MB | Available ${rep.hardware.availableRamMb}MB`);
|
|
108
|
-
console.log(` CPU Cores: ${rep.hardware.cpuCores}`);
|
|
109
|
-
console.log(` Vulkan : Loader=${rep.vulkan.loaderDetected} | Driver=${rep.vulkan.driverDetected} | Status=${rep.vulkan.status}`);
|
|
110
|
-
console.log(` Models : ${rep.vlmRuntime.installedModelsCount} installed in ${rep.vlmRuntime.cacheDir}`);
|
|
111
|
-
console.log(` Preset : ${rep.recommendedPreset}`);
|
|
112
|
-
}
|
|
113
|
-
process.exit(0);
|
|
114
|
-
}
|
|
115
|
-
|
|
116
|
-
if (command === 'model') {
|
|
117
|
-
const action = args[1] || 'list';
|
|
118
|
-
if (action === 'list') {
|
|
119
|
-
const isJson = hasFlag('--json');
|
|
120
|
-
const installed = cache.listInstalled();
|
|
121
|
-
if (isJson) {
|
|
122
|
-
console.log(JSON.stringify(installed, null, 2));
|
|
123
|
-
} else {
|
|
124
|
-
console.log(`=== Installed VLM Models (${installed.length}) ===`);
|
|
125
|
-
if (installed.length === 0) {
|
|
126
|
-
console.log(' (none in ~/.cache/termux-vision/models)');
|
|
127
|
-
} else {
|
|
128
|
-
for (const m of installed) {
|
|
129
|
-
console.log(` - ${m.modelId.padEnd(20)} | Tier: ${m.tier} | State: ${m.state} | Size: ${m.sizeMb}MB`);
|
|
130
|
-
}
|
|
131
|
-
}
|
|
132
|
-
console.log('\nAvailable Official Presets:');
|
|
133
|
-
for (const [k, v] of Object.entries(CATALOG)) {
|
|
134
|
-
console.log(` * ${k.padEnd(20)} | Tier: ${v.tier} | Est. RAM: ${v.estimatedMemoryMb}MB`);
|
|
135
|
-
}
|
|
136
|
-
}
|
|
137
|
-
process.exit(0);
|
|
138
|
-
}
|
|
139
|
-
|
|
140
|
-
if (action === 'install') {
|
|
141
|
-
const modelId = args[2];
|
|
142
|
-
if (!modelId) {
|
|
143
|
-
console.error('[ERROR] Please specify a model ID to install (e.g. smolvlm-500m-q4).');
|
|
144
|
-
process.exit(2);
|
|
145
|
-
}
|
|
146
|
-
console.log(`[*] Installing model: ${modelId}...`);
|
|
147
|
-
try {
|
|
148
|
-
await cache.install(modelId, (fname, downloaded, total) => {
|
|
149
|
-
const pct = total > 0 ? (downloaded / total * 100).toFixed(1) : '0.0';
|
|
150
|
-
process.stdout.write(`\r Downloading ${fname}: ${(downloaded / 1048576).toFixed(1)}/${(total / 1048576).toFixed(1)}MB (${pct}%)`);
|
|
151
|
-
});
|
|
152
|
-
console.log(`\n[+] Successfully installed '${modelId}'.`);
|
|
153
|
-
process.exit(0);
|
|
154
|
-
} catch (err) {
|
|
155
|
-
console.error(`\n[-] Model installation failed: ${err.message}`);
|
|
156
|
-
process.exit(11);
|
|
157
|
-
}
|
|
158
|
-
}
|
|
159
|
-
|
|
160
|
-
if (action === 'download') {
|
|
161
|
-
const source = args[2];
|
|
162
|
-
if (!source) {
|
|
163
|
-
console.error('[ERROR] Please provide a Hugging Face repo or direct URL.');
|
|
164
|
-
process.exit(2);
|
|
165
|
-
}
|
|
166
|
-
const outDir = getArg('-o', '--output') || cache.modelsDir;
|
|
167
|
-
console.log(`[*] Downloading model from: ${source}...`);
|
|
168
|
-
try {
|
|
169
|
-
let dest = path.join(outDir, path.basename(source.split('?')[0]));
|
|
170
|
-
if (source.startsWith('hf:') || source.includes(':')) {
|
|
171
|
-
const parts = (source.startsWith('hf:') ? source.slice(3) : source).split(':');
|
|
172
|
-
if (parts.length === 2) {
|
|
173
|
-
const url = `https://huggingface.co/${parts[0].trim()}/resolve/main/${parts[1].trim()}`;
|
|
174
|
-
dest = path.join(outDir, parts[1].trim());
|
|
175
|
-
await cache.downloadFile(url, dest, (fname, d, t) => {
|
|
176
|
-
const pct = t > 0 ? (d / t * 100).toFixed(1) : '0.0';
|
|
177
|
-
process.stdout.write(`\r Downloading ${fname}: ${(d / 1048576).toFixed(1)}/${(t / 1048576).toFixed(1)}MB (${pct}%)`);
|
|
178
|
-
});
|
|
179
|
-
}
|
|
180
|
-
} else if (source.startsWith('http')) {
|
|
181
|
-
await cache.downloadFile(source, dest, (fname, d, t) => {
|
|
182
|
-
const pct = t > 0 ? (d / t * 100).toFixed(1) : '0.0';
|
|
183
|
-
process.stdout.write(`\r Downloading ${fname}: ${(d / 1048576).toFixed(1)}/${(t / 1048576).toFixed(1)}MB (${pct}%)`);
|
|
184
|
-
});
|
|
185
|
-
}
|
|
186
|
-
console.log(`\n[+] Model downloaded successfully to: ${dest}`);
|
|
187
|
-
process.exit(0);
|
|
188
|
-
} catch (err) {
|
|
189
|
-
console.error(`\n[-] Model download failed: ${err.message}`);
|
|
190
|
-
process.exit(11);
|
|
191
|
-
}
|
|
192
|
-
}
|
|
193
|
-
|
|
194
|
-
if (action === 'remove') {
|
|
195
|
-
const modelId = args[2];
|
|
196
|
-
if (!modelId) {
|
|
197
|
-
console.error('[ERROR] Please specify a model ID to remove.');
|
|
198
|
-
process.exit(2);
|
|
199
|
-
}
|
|
200
|
-
if (cache.remove(modelId)) {
|
|
201
|
-
console.log(`[+] Model '${modelId}' removed from cache.`);
|
|
202
|
-
process.exit(0);
|
|
203
|
-
} else {
|
|
204
|
-
console.error(`[-] Model '${modelId}' was not found in cache.`);
|
|
205
|
-
process.exit(10);
|
|
206
|
-
}
|
|
207
|
-
}
|
|
17
|
+
const termuxBinAlt = '/data/data/com.termux/files/usr/bin/python';
|
|
18
|
+
if (fs.existsSync(termuxBinAlt)) {
|
|
19
|
+
return termuxBinAlt;
|
|
208
20
|
}
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
const imagePath = args[1];
|
|
212
|
-
if (!imagePath || imagePath.startsWith('-')) {
|
|
213
|
-
console.error('[ERROR] Missing input image path for VLM inference.');
|
|
214
|
-
console.error('Usage: termux-vision vlm <image_path> [options]');
|
|
215
|
-
process.exit(2);
|
|
216
|
-
}
|
|
217
|
-
|
|
218
|
-
const prompt = getArg('-p', '--prompt');
|
|
219
|
-
const model = getArg('-m', '--model');
|
|
220
|
-
const mmproj = getArg('--mmproj', null);
|
|
221
|
-
const device = getArg('--device', null) || 'auto';
|
|
222
|
-
const runtime = getArg('--runtime', null);
|
|
223
|
-
const allowDownload = hasFlag('--allow-download');
|
|
224
|
-
const threads = getArg('-t', '--threads');
|
|
225
|
-
const maxTokens = getArg('-n', '--max-tokens');
|
|
226
|
-
const temp = getArg('--temp', '--temperature');
|
|
227
|
-
const topP = getArg('--top-p', null);
|
|
228
|
-
const topK = getArg('--top-k', null);
|
|
229
|
-
const repeatPenalty = getArg('--repeat-penalty', null);
|
|
230
|
-
const seed = getArg('--seed', null);
|
|
231
|
-
const systemPrompt = getArg('--system-prompt', null);
|
|
232
|
-
const ctxSize = getArg('-c', '--ctx-size');
|
|
233
|
-
const quality = getArg('-q', '--quality') || 'optimal';
|
|
234
|
-
const maxDim = getArg('--max-dim', null);
|
|
235
|
-
const ngl = getArg('--ngl', null);
|
|
236
|
-
const isJson = hasFlag('--json');
|
|
237
|
-
|
|
21
|
+
const candidates = ['python3', 'python'];
|
|
22
|
+
for (const cmd of candidates) {
|
|
238
23
|
try {
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
if (allowDownload) {
|
|
244
|
-
targetModel = 'smolvlm-500m-q4';
|
|
245
|
-
} else if (process.stdin.isTTY) {
|
|
246
|
-
console.error('\n---------------------------------------------------------');
|
|
247
|
-
console.error(' [Notice] No local VLM model is currently installed.');
|
|
248
|
-
console.error(' Default Model : smolvlm-500m-q4 (SmolVLM 500M Instruct)');
|
|
249
|
-
console.error(' Download Size : ~550 MB');
|
|
250
|
-
console.error(' Target Path : ~/.cache/termux-vision/models/smolvlm-500m-q4/');
|
|
251
|
-
console.error('---------------------------------------------------------');
|
|
252
|
-
const ans = await promptUser('Do you want to download and install this model now? [y/N]: ');
|
|
253
|
-
if (ans === 'y' || ans === 'yes') {
|
|
254
|
-
console.error('[*] Downloading smolvlm-500m-q4 (~550MB)...');
|
|
255
|
-
await cache.install('smolvlm-500m-q4');
|
|
256
|
-
console.error('[+] Successfully installed smolvlm-500m-q4.');
|
|
257
|
-
targetModel = 'smolvlm-500m-q4';
|
|
258
|
-
} else {
|
|
259
|
-
throw new NoInstalledModelsError(cache.getAvailableCatalogModels());
|
|
260
|
-
}
|
|
261
|
-
} else {
|
|
262
|
-
throw new NoInstalledModelsError(cache.getAvailableCatalogModels());
|
|
263
|
-
}
|
|
264
|
-
} else if (installed.length === 1) {
|
|
265
|
-
targetModel = installed[0].modelId;
|
|
266
|
-
console.error(`[INFO] Selected installed model: ${targetModel}`);
|
|
267
|
-
} else {
|
|
268
|
-
throw new ModelSelectionRequiredError(installed.map(m => m.modelId));
|
|
269
|
-
}
|
|
270
|
-
}
|
|
271
|
-
|
|
272
|
-
const engine = await load({
|
|
273
|
-
modelId: targetModel,
|
|
274
|
-
mmprojPath: mmproj,
|
|
275
|
-
device: device,
|
|
276
|
-
runtimePath: runtime,
|
|
277
|
-
allowDownload: allowDownload,
|
|
278
|
-
threads: threads ? parseInt(threads, 10) : 4,
|
|
279
|
-
contextLimit: ctxSize ? parseInt(ctxSize, 10) : null,
|
|
280
|
-
ngl: ngl ? parseInt(ngl, 10) : null
|
|
281
|
-
});
|
|
282
|
-
|
|
283
|
-
const res = await engine.describe(imagePath, {
|
|
284
|
-
prompt: prompt,
|
|
285
|
-
maxTokens: maxTokens ? parseInt(maxTokens, 10) : 150,
|
|
286
|
-
temperature: temp ? parseFloat(temp) : 0.2,
|
|
287
|
-
topP: topP ? parseFloat(topP) : undefined,
|
|
288
|
-
topK: topK ? parseInt(topK, 10) : undefined,
|
|
289
|
-
repeatPenalty: repeatPenalty ? parseFloat(repeatPenalty) : undefined,
|
|
290
|
-
seed: seed ? parseInt(seed, 10) : undefined,
|
|
291
|
-
systemPrompt: systemPrompt || undefined,
|
|
292
|
-
quality: quality,
|
|
293
|
-
maxDim: maxDim
|
|
294
|
-
});
|
|
295
|
-
|
|
296
|
-
if (isJson) {
|
|
297
|
-
console.log(JSON.stringify(res, null, 2));
|
|
298
|
-
} else {
|
|
299
|
-
if (res.warnings && res.warnings.length > 0) {
|
|
300
|
-
for (const w of res.warnings) {
|
|
301
|
-
console.error(`[WARNING] ${w}`);
|
|
302
|
-
}
|
|
303
|
-
}
|
|
304
|
-
const tpsStr = res.metrics.tokensPerSecond ? ` | ${res.metrics.tokensPerSecond.toFixed(1)} t/s` : '';
|
|
305
|
-
console.log(`\n[VLM Result | backend=${res.metrics.backend}${tpsStr}]`);
|
|
306
|
-
console.log(res.text);
|
|
307
|
-
}
|
|
308
|
-
process.exit(0);
|
|
309
|
-
} catch (err) {
|
|
310
|
-
if (err instanceof VulkanNotAvailableError) {
|
|
311
|
-
console.error(`[ERROR] ${err.message}`);
|
|
312
|
-
process.exit(24);
|
|
313
|
-
} else if (err instanceof RuntimeNotFoundError) {
|
|
314
|
-
console.error(`[ERROR] ${err.message}`);
|
|
315
|
-
process.exit(20);
|
|
316
|
-
} else if (err instanceof NoInstalledModelsError) {
|
|
317
|
-
console.error(`[ERROR] ${err.message}`);
|
|
318
|
-
process.exit(21);
|
|
319
|
-
} else if (err instanceof ModelSelectionRequiredError) {
|
|
320
|
-
console.error(`[ERROR] ${err.message}`);
|
|
321
|
-
process.exit(23);
|
|
322
|
-
} else if (err instanceof ModelNotFoundError) {
|
|
323
|
-
console.error(`[ERROR] ${err.message}`);
|
|
324
|
-
process.exit(10);
|
|
325
|
-
} else if (err instanceof ModelDownloadError) {
|
|
326
|
-
console.error(`[ERROR] ${err.message}`);
|
|
327
|
-
process.exit(11);
|
|
328
|
-
} else {
|
|
329
|
-
console.error(`[ERROR] VLM execution failed: ${err.message}`);
|
|
330
|
-
process.exit(15);
|
|
331
|
-
}
|
|
332
|
-
}
|
|
333
|
-
}
|
|
334
|
-
|
|
335
|
-
if (command === 'canny') {
|
|
336
|
-
const imagePath = args[1];
|
|
337
|
-
if (!imagePath || imagePath.startsWith('-')) {
|
|
338
|
-
console.error('[ERROR] Missing input image path for Canny edge detection.');
|
|
339
|
-
console.error('Usage: termux-vision canny <image_path> [options]');
|
|
340
|
-
process.exit(2);
|
|
341
|
-
}
|
|
342
|
-
const resolvedImg = path.resolve(imagePath.replace(/^~(?=$|\/|\\)/, os.homedir()));
|
|
343
|
-
if (!fs.existsSync(resolvedImg)) {
|
|
344
|
-
console.error(`[ERROR] Image file not found: '${resolvedImg}'`);
|
|
345
|
-
process.exit(2);
|
|
346
|
-
}
|
|
347
|
-
const outPath = getArg('-o', '--output') || 'edges.png';
|
|
348
|
-
const low = parseFloat(getArg('--low') || '40.0');
|
|
349
|
-
const high = parseFloat(getArg('--high') || '120.0');
|
|
350
|
-
|
|
351
|
-
// Execute Canny via Python bridge or fallback
|
|
352
|
-
const { spawnSync } = require('child_process');
|
|
353
|
-
const pyRes = spawnSync('python3', [
|
|
354
|
-
'-m', 'termux_vision.cli.main', 'canny', resolvedImg, '-o', outPath, '--low', String(low), '--high', String(high)
|
|
355
|
-
], { stdio: 'inherit' });
|
|
356
|
-
|
|
357
|
-
if (pyRes.status === 0) {
|
|
358
|
-
process.exit(0);
|
|
359
|
-
} else {
|
|
360
|
-
process.exit(pyRes.status || 1);
|
|
361
|
-
}
|
|
24
|
+
const checkCmd = process.platform === 'win32' ? `where ${cmd}` : `command -v ${cmd}`;
|
|
25
|
+
const res = execSync(checkCmd, { stdio: ['ignore', 'pipe', 'ignore'] }).toString().trim();
|
|
26
|
+
if (res) return cmd;
|
|
27
|
+
} catch (_) {}
|
|
362
28
|
}
|
|
29
|
+
return 'python3';
|
|
30
|
+
}
|
|
363
31
|
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
const width = 256;
|
|
367
|
-
const height = 256;
|
|
368
|
-
const dummy = new Uint8Array(width * height);
|
|
369
|
-
for (let i = 0; i < dummy.length; i++) dummy[i] = Math.floor(Math.random() * 256);
|
|
32
|
+
const pythonBin = findPython();
|
|
33
|
+
const args = ['-m', 'termux_vision', ...process.argv.slice(2)];
|
|
370
34
|
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
console.log('[+] Benchmark Complete.');
|
|
376
|
-
process.exit(0);
|
|
377
|
-
}
|
|
378
|
-
|
|
379
|
-
console.error(`[ERROR] Unknown command: '${command}'`);
|
|
380
|
-
printHelp();
|
|
381
|
-
process.exit(2);
|
|
382
|
-
}
|
|
35
|
+
const child = spawn(pythonBin, args, {
|
|
36
|
+
stdio: 'inherit',
|
|
37
|
+
env: process.env
|
|
38
|
+
});
|
|
383
39
|
|
|
384
|
-
|
|
385
|
-
console.error(`[
|
|
40
|
+
child.on('error', (err) => {
|
|
41
|
+
console.error(`[${'termux_vision'}] Failed to spawn python process (${pythonBin}):`, err.message);
|
|
386
42
|
process.exit(1);
|
|
387
43
|
});
|
|
44
|
+
|
|
45
|
+
child.on('exit', (code, signal) => {
|
|
46
|
+
if (signal) {
|
|
47
|
+
process.kill(process.pid, signal);
|
|
48
|
+
} else {
|
|
49
|
+
process.exit(code || 0);
|
|
50
|
+
}
|
|
51
|
+
});
|
package/lib/vlm.js
CHANGED
|
@@ -45,15 +45,17 @@ function resolveLlamaCli(explicitPath = null) {
|
|
|
45
45
|
}
|
|
46
46
|
|
|
47
47
|
const prefix = process.env.PREFIX || '/data/data/com.termux/files/usr';
|
|
48
|
+
const xdgCache = process.env.XDG_CACHE_HOME || path.join(os.homedir(), '.cache');
|
|
48
49
|
const candidates = [
|
|
49
|
-
path.join(
|
|
50
|
-
path.join(os.homedir(), '.
|
|
51
|
-
'llama-cli',
|
|
50
|
+
path.join(xdgCache, 'termux-llamacpp', 'bin', 'llama-cli'),
|
|
51
|
+
path.join(os.homedir(), '.local', 'bin', 'llama-cli'),
|
|
52
52
|
path.join(prefix, 'bin', 'llama-cli'),
|
|
53
53
|
path.join(prefix, 'bin', 'termux-llama-cli'),
|
|
54
54
|
path.join(prefix, 'bin', 'llama-mtmd-cli'),
|
|
55
|
-
path.join(os.homedir(), '.
|
|
56
|
-
path.join(os.homedir(), 'bin', 'llama-cli')
|
|
55
|
+
path.join(os.homedir(), '.termux-llamacpp', 'current', 'bin', 'llama-cli'),
|
|
56
|
+
path.join(os.homedir(), '.termux-llama', 'current', 'bin', 'llama-cli'),
|
|
57
|
+
path.join(os.homedir(), 'bin', 'llama-cli'),
|
|
58
|
+
'llama-cli'
|
|
57
59
|
];
|
|
58
60
|
|
|
59
61
|
const pathDirs = (process.env.PATH || '').split(path.delimiter);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "termux-vision",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.4.1",
|
|
4
4
|
"description": "On-device computer vision & VLM multimodal inference framework utilizing device resources for Android Termux & ARM64",
|
|
5
5
|
"main": "index.js",
|
|
6
6
|
"types": "index.d.ts",
|
|
@@ -49,10 +49,8 @@
|
|
|
49
49
|
"bugs": {
|
|
50
50
|
"url": "https://github.com/uno-km/termux-vision/issues"
|
|
51
51
|
},
|
|
52
|
-
"dependencies": {
|
|
53
|
-
"@ameva/runtime": ">=2.0.0"
|
|
54
|
-
},
|
|
52
|
+
"dependencies": {},
|
|
55
53
|
"engines": {
|
|
56
54
|
"node": ">=16.0.0"
|
|
57
55
|
}
|
|
58
|
-
}
|
|
56
|
+
}
|