omnius 1.0.418 → 1.0.420
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +64 -12
- package/dist/scripts/live-whisper.py +39 -15
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -603059,6 +603059,21 @@ async function resolveMicSource() {
|
|
|
603059
603059
|
const best = candidates[0];
|
|
603060
603060
|
return { pulseSource: best.name, channels: best.channels, label: `pulse:${best.name}` };
|
|
603061
603061
|
}
|
|
603062
|
+
async function ensureSourceCaptureVolume(source) {
|
|
603063
|
+
try {
|
|
603064
|
+
await execFileText4("pactl", ["set-source-mute", source, "0"], { timeout: 3e3 });
|
|
603065
|
+
} catch {
|
|
603066
|
+
}
|
|
603067
|
+
try {
|
|
603068
|
+
const raw = await execFileText4("pactl", ["get-source-volume", source], { timeout: 3e3 });
|
|
603069
|
+
const percents = [...raw.matchAll(/(\d+)%/g)].map((m2) => Number(m2[1])).filter((n2) => Number.isFinite(n2));
|
|
603070
|
+
const current = percents.length ? Math.max(...percents) : null;
|
|
603071
|
+
if (current !== null && current < 100) {
|
|
603072
|
+
await execFileText4("pactl", ["set-source-volume", source, "100%"], { timeout: 3e3 });
|
|
603073
|
+
}
|
|
603074
|
+
} catch {
|
|
603075
|
+
}
|
|
603076
|
+
}
|
|
603062
603077
|
function micFfmpegFilter(channels) {
|
|
603063
603078
|
const parts = [];
|
|
603064
603079
|
if ((channels ?? 1) > 2) parts.push("pan=mono|c0=c0");
|
|
@@ -603079,6 +603094,7 @@ async function findMicCaptureCommand() {
|
|
|
603079
603094
|
};
|
|
603080
603095
|
}
|
|
603081
603096
|
if (resolved?.pulseSource && await commandExists2("ffmpeg")) {
|
|
603097
|
+
await ensureSourceCaptureVolume(resolved.pulseSource);
|
|
603082
603098
|
return {
|
|
603083
603099
|
cmd: "ffmpeg",
|
|
603084
603100
|
args: [
|
|
@@ -603687,7 +603703,7 @@ var init_listen = __esm({
|
|
|
603687
603703
|
else if (levelDb < this.micNoiseFloorDb) this.micNoiseFloorDb = 0.3 * levelDb + 0.7 * this.micNoiseFloorDb;
|
|
603688
603704
|
else this.micNoiseFloorDb = 0.02 * levelDb + 0.98 * this.micNoiseFloorDb;
|
|
603689
603705
|
const floor = this.micNoiseFloorDb;
|
|
603690
|
-
const speech = this.micLevelDb > Math.max(-
|
|
603706
|
+
const speech = this.micLevelDb > Math.max(-58, floor + 6);
|
|
603691
603707
|
const changed = speech !== this.micSpeechActive;
|
|
603692
603708
|
this.micSpeechActive = speech;
|
|
603693
603709
|
const now2 = Date.now();
|
|
@@ -604575,7 +604591,7 @@ function readPcm16WavActivity(filePath, bins = 36) {
|
|
|
604575
604591
|
const abs = Math.abs(sample);
|
|
604576
604592
|
peak = Math.max(peak, abs);
|
|
604577
604593
|
sumSquares += sample * sample;
|
|
604578
|
-
if (abs >
|
|
604594
|
+
if (abs > 8e-3) active++;
|
|
604579
604595
|
const b = Math.min(bucketSquares.length - 1, Math.floor(i2 / frames * bucketSquares.length));
|
|
604580
604596
|
bucketSquares[b] += sample * sample;
|
|
604581
604597
|
bucketCounts[b] += 1;
|
|
@@ -605411,16 +605427,22 @@ function formatLiveSensorContextFromSnapshot(snapshot, opts = {}) {
|
|
|
605411
605427
|
if (!snapshot) return "";
|
|
605412
605428
|
const now2 = opts.now ?? Date.now();
|
|
605413
605429
|
const maxAgeMs = opts.maxAgeMs ?? 9e4;
|
|
605430
|
+
const toolCapable = opts.toolCapable !== false;
|
|
605414
605431
|
const lines = [];
|
|
605415
605432
|
const activeVideo = snapshot.config.videoEnabled || Boolean(snapshot.video);
|
|
605416
605433
|
const activeAudio = snapshot.config.audioEnabled || snapshot.config.audioOutputEnabled || Boolean(snapshot.audio);
|
|
605417
605434
|
if (!activeVideo && !activeAudio) return "";
|
|
605418
605435
|
lines.push("<live-sensor-context>");
|
|
605419
605436
|
lines.push("## LIVE SENSOR STATUS");
|
|
605420
|
-
|
|
605421
|
-
|
|
605422
|
-
|
|
605423
|
-
|
|
605437
|
+
if (toolCapable) {
|
|
605438
|
+
lines.push("Live sensor context is operator-enabled ambient input. Treat it as current perceptual evidence for this turn; if it is stale, missing, or insufficient, use camera/audio tools to refresh before answering what you see or hear.");
|
|
605439
|
+
lines.push('Live world-model rule: when the user asks what is around, who is present, what changed, what is heard, or asks a deictic question like "what do you see", treat the live camera/audio fields as first-class context. Be curious and tool-capable: refresh stale sensors, inspect frame paths, query vision/CLIP/visual_memory/audio tools, and combine metadata into a concise answer. Do not rely only on summaries when the task requires current perception.');
|
|
605440
|
+
lines.push("Live speech rule: in low-latency voice/live modes, short spoken replies are appropriate for direct address, identity clarification, or important visual/audio triggers. Do not narrate every frame.");
|
|
605441
|
+
lines.push("Identity rule: never invent names for observed people. If a person is unknown and identity matters, ask the user or the person for their name, then store the association with visual_memory(action='enroll', image='<face crop>', name='<name>') when a face crop is available, or multimodal_memory(action='meet', person_name='<name>') when live voice/name context is available.");
|
|
605442
|
+
} else {
|
|
605443
|
+
lines.push("This is your current live perception: what the camera sees and what is heard right now. Answer questions about sight/sound directly from these fields.");
|
|
605444
|
+
lines.push("If the data below is stale or missing, say you cannot see/hear clearly right now. Never invent observations or names for people. To have the main agent investigate further, relay a request with the voice_to_main tool.");
|
|
605445
|
+
}
|
|
605424
605446
|
lines.push(`Snapshot file: ${liveSnapshotPath(snapshot.repoRoot)}`);
|
|
605425
605447
|
lines.push(`Streams: video=${snapshot.config.videoEnabled ? "on" : "off"} infer=${snapshot.config.inferEnabled ? "on" : "off"} clip=${snapshot.config.clipEnabled ? "on" : "off"} audio=${snapshot.config.audioEnabled ? "on" : "off"} asr=${snapshot.config.asrEnabled ? "on" : "off"} sounds=${snapshot.config.audioAnalysisEnabled ? "on" : "off"} output-monitor=${snapshot.config.audioOutputEnabled ? "on" : "off"}`);
|
|
605426
605448
|
if (snapshot.config.selectedCamera) {
|
|
@@ -605563,8 +605585,8 @@ function formatLiveSensorContextFromSnapshot(snapshot, opts = {}) {
|
|
|
605563
605585
|
... live sensor context truncated ...
|
|
605564
605586
|
</live-sensor-context>` : text2;
|
|
605565
605587
|
}
|
|
605566
|
-
function formatLiveSensorContext(repoRoot) {
|
|
605567
|
-
return formatLiveSensorContextFromSnapshot(readLiveSensorSnapshot(repoRoot));
|
|
605588
|
+
function formatLiveSensorContext(repoRoot, opts = {}) {
|
|
605589
|
+
return formatLiveSensorContextFromSnapshot(readLiveSensorSnapshot(repoRoot), opts);
|
|
605568
605590
|
}
|
|
605569
605591
|
function isLiveSensorSnapshotActive(snapshot) {
|
|
605570
605592
|
const cfg = snapshot?.config;
|
|
@@ -605737,7 +605759,7 @@ async function tryRecordAudioDevice(device, durationSec, outputPath3, channels)
|
|
|
605737
605759
|
}
|
|
605738
605760
|
function hasLiveAudioSignal(activity) {
|
|
605739
605761
|
if (!activity) return false;
|
|
605740
|
-
return activity.peak >= 0.
|
|
605762
|
+
return activity.peak >= 0.012 || activity.rmsDb > -55 || activity.activeRatio >= 0.015;
|
|
605741
605763
|
}
|
|
605742
605764
|
function audioCaptureMethod(device) {
|
|
605743
605765
|
return device.startsWith("pulse:") ? "pulse/ffmpeg" : "arecord";
|
|
@@ -696617,9 +696639,23 @@ function buildRealtimeVoiceMessages(turns, maxMessages = MAX_REALTIME_MODEL_MESS
|
|
|
696617
696639
|
dynamic.push(turn);
|
|
696618
696640
|
}
|
|
696619
696641
|
}
|
|
696620
|
-
if (latestSnapshot)
|
|
696642
|
+
if (latestSnapshot) {
|
|
696643
|
+
let lastUserIdx = -1;
|
|
696644
|
+
for (let i2 = dynamic.length - 1; i2 >= 0; i2--) {
|
|
696645
|
+
if (dynamic[i2].role === "user") {
|
|
696646
|
+
lastUserIdx = i2;
|
|
696647
|
+
break;
|
|
696648
|
+
}
|
|
696649
|
+
}
|
|
696650
|
+
if (lastUserIdx >= 0) dynamic.splice(lastUserIdx, 0, latestSnapshot);
|
|
696651
|
+
else dynamic.push(latestSnapshot);
|
|
696652
|
+
}
|
|
696621
696653
|
const available = Math.max(0, maxMessages - pinned.length);
|
|
696622
|
-
|
|
696654
|
+
let windowed = dynamic.slice(-available);
|
|
696655
|
+
if (latestSnapshot && available > 0 && !windowed.includes(latestSnapshot)) {
|
|
696656
|
+
windowed = [latestSnapshot, ...windowed.slice(-(available - 1))];
|
|
696657
|
+
}
|
|
696658
|
+
return [...pinned, ...windowed];
|
|
696623
696659
|
}
|
|
696624
696660
|
var VAD_SILENCE_MS, MAX_SEGMENT_MS, SUMMARY_INJECTION_INTERVAL, MAX_CONTEXT_TURNS, MAX_REALTIME_MODEL_MESSAGES, CONTEXT_SNAPSHOT_PREFIX, MAX_VOICE_REPLY_CHARS, AGENT_ECHO_WINDOW_MS, AGENT_ECHO_SIMILARITY, SYSTEM_PROMPT2, MIN_SIGNAL_SCORE, NOISE_ONLY_RE, VoiceChatSession;
|
|
696625
696661
|
var init_voicechat = __esm({
|
|
@@ -696637,6 +696673,9 @@ var init_voicechat = __esm({
|
|
|
696637
696673
|
SYSTEM_PROMPT2 = `You are a voice assistant having a live spoken conversation. Keep responses extremely brief — 1-2 sentences max. You're speaking aloud, not writing. Be conversational, direct, and helpful. Don't use markdown or formatting — just natural speech.
|
|
696638
696674
|
|
|
696639
696675
|
Rules:
|
|
696676
|
+
- ALWAYS answer the user's most recent message directly. The live context snapshot is background perception — never the topic. Do not describe cameras, feeds, or surroundings unless the user's latest message asks about them.
|
|
696677
|
+
- Never repeat your previous answer. If you already described the scene, don't describe it again unless explicitly asked again — respond to what the user just said.
|
|
696678
|
+
- Live perception: each turn you may receive a read-only "Context snapshot" system message containing what the camera currently sees (objects, people, classifications, recent visual events) and what is currently heard (sound scene, recent sounds, transcripts). Treat it as your own live sight and hearing. When the user asks what you SEE, answer from the camera fields; when the user asks what you HEAR, answer from the audio/sound fields. If the snapshot is missing or stale, say you can't see/hear clearly right now — do not invent observations.
|
|
696640
696679
|
- Never invent environment facts (cwd, OS, specs, repo state). If you need a precise fact from the main agent, request a tool by outputting on a single line EXACTLY one JSON object: {"tool": string, "args": object} and nothing else. Then wait for the tool result before answering.
|
|
696641
696680
|
- You may also request to relay a user task to the main agent by emitting {"tool":"voice_to_main","args":{"message":"...","start":true}}.
|
|
696642
696681
|
- Prefer tools for factual queries; otherwise, answer directly with a short reply.
|
|
@@ -696992,6 +697031,19 @@ ${toolOutput}` });
|
|
|
696992
697031
|
this.context.push({ role: "system", content: `You have tools. Use them. ${this.toolCatalogNote}` });
|
|
696993
697032
|
response = await this.streamOllamaInference(this.abortController.signal);
|
|
696994
697033
|
}
|
|
697034
|
+
if (this.lastAgentSpeech?.text && response.trim()) {
|
|
697035
|
+
const prevTokens = comparableTokens(this.lastAgentSpeech.text);
|
|
697036
|
+
const nextTokens = comparableTokens(stripToolJsonLines(response));
|
|
697037
|
+
const similarity3 = jaccardSimilarity(prevTokens, nextTokens);
|
|
697038
|
+
if (similarity3 >= 0.8 && nextTokens.length >= 8) {
|
|
697039
|
+
this.context.push({
|
|
697040
|
+
role: "system",
|
|
697041
|
+
content: `Your draft reply repeats your previous answer. Do NOT describe the scene or cameras again. Respond directly and briefly to the user's latest message: "${truncateForLog(lastUser ?? "", 200)}"`
|
|
697042
|
+
});
|
|
697043
|
+
const retry = await this.streamOllamaInference(this.abortController.signal);
|
|
697044
|
+
if (retry.trim()) response = retry;
|
|
697045
|
+
}
|
|
697046
|
+
}
|
|
696995
697047
|
if (response.trim()) {
|
|
696996
697048
|
const reply = extractVoiceModelReply(stripToolJsonLines(response.trim()));
|
|
696997
697049
|
const finalSpoken = reply.text;
|
|
@@ -736343,7 +736395,7 @@ Respond concisely and safely. Remember: you are talking to the general public.`;
|
|
|
736343
736395
|
} else {
|
|
736344
736396
|
parts.push("active: no");
|
|
736345
736397
|
}
|
|
736346
|
-
const liveContext = formatLiveSensorContext(repoRoot);
|
|
736398
|
+
const liveContext = formatLiveSensorContext(repoRoot, { toolCapable: false, maxChars: 7e3 });
|
|
736347
736399
|
if (liveContext) parts.push(liveContext);
|
|
736348
736400
|
return parts.join("\n");
|
|
736349
736401
|
}
|
|
@@ -165,6 +165,24 @@ def main():
|
|
|
165
165
|
window_samples = int(args.window_seconds * SAMPLE_RATE)
|
|
166
166
|
last_text = ""
|
|
167
167
|
running = True
|
|
168
|
+
noise_floor = None # EMA of the quietest 0.5s block RMS — adapts to the capture chain
|
|
169
|
+
|
|
170
|
+
def block_energies(samples):
|
|
171
|
+
"""RMS of consecutive 0.5s blocks — lets speech anywhere in the
|
|
172
|
+
window register even when most of the window is silence."""
|
|
173
|
+
block = SAMPLE_RATE // 2
|
|
174
|
+
n = max(1, len(samples) // block)
|
|
175
|
+
return [float(np.sqrt(np.mean(samples[i * block:(i + 1) * block] ** 2))) for i in range(n)]
|
|
176
|
+
|
|
177
|
+
def amplify(samples):
|
|
178
|
+
"""Bring quiet-but-real speech toward full scale. Whisper performs far
|
|
179
|
+
better near normal levels; quiet capture chains (distant speakers,
|
|
180
|
+
low source gain, beamformed arrays) otherwise force users to shout
|
|
181
|
+
into the mic. Gain capped at 40x (~32dB), output clipped to [-1, 1]."""
|
|
182
|
+
peak = float(np.max(np.abs(samples))) if len(samples) else 0.0
|
|
183
|
+
if 0 < peak < 0.5:
|
|
184
|
+
samples = np.clip(samples * min(40.0, 0.9 / max(peak, 1e-4)), -1.0, 1.0)
|
|
185
|
+
return samples
|
|
168
186
|
|
|
169
187
|
def read_stdin():
|
|
170
188
|
"""Read PCM16 from stdin in a background thread."""
|
|
@@ -197,18 +215,25 @@ def main():
|
|
|
197
215
|
# Take the last window_seconds of audio
|
|
198
216
|
window = audio_buf[-window_samples:].copy() if len(audio_buf) > window_samples else audio_buf.copy()
|
|
199
217
|
|
|
200
|
-
#
|
|
201
|
-
#
|
|
202
|
-
#
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
218
|
+
# Adaptive silence gate: whisper hallucinates repetitive garbage
|
|
219
|
+
# on silent input, but absolute thresholds reject real speech on
|
|
220
|
+
# quiet capture chains. Instead track the noise floor (quietest
|
|
221
|
+
# 0.5s block) and require the loudest block to rise above it.
|
|
222
|
+
energies = block_energies(window)
|
|
223
|
+
quiet = min(energies)
|
|
224
|
+
loud = max(energies)
|
|
225
|
+
if noise_floor is None:
|
|
226
|
+
noise_floor = quiet
|
|
227
|
+
elif quiet < noise_floor:
|
|
228
|
+
noise_floor = 0.7 * noise_floor + 0.3 * quiet # drop fast
|
|
229
|
+
else:
|
|
230
|
+
noise_floor = 0.995 * noise_floor + 0.005 * quiet # creep up slowly
|
|
231
|
+
gate = max(1.5e-4, noise_floor * 2.5) # ≥ ~8dB above floor, ~-76dB min
|
|
232
|
+
if loud < gate:
|
|
206
233
|
continue
|
|
207
234
|
|
|
208
|
-
#
|
|
209
|
-
|
|
210
|
-
if peak < 0.3:
|
|
211
|
-
window = window * min(20.0, 0.9 / max(peak, 1e-4))
|
|
235
|
+
# Amplify toward full scale before transcription
|
|
236
|
+
window = amplify(window)
|
|
212
237
|
|
|
213
238
|
# Transcribe
|
|
214
239
|
try:
|
|
@@ -234,12 +259,11 @@ def main():
|
|
|
234
259
|
with buf_lock:
|
|
235
260
|
full_audio = audio_buf.copy()
|
|
236
261
|
|
|
237
|
-
|
|
238
|
-
|
|
262
|
+
final_loud = max(block_energies(full_audio)) if len(full_audio) >= SAMPLE_RATE else 0.0
|
|
263
|
+
final_gate = max(1.5e-4, (noise_floor or 0.0) * 2.5)
|
|
264
|
+
if len(full_audio) >= SAMPLE_RATE and final_loud >= final_gate:
|
|
239
265
|
try:
|
|
240
|
-
|
|
241
|
-
if 0 < final_peak < 0.3:
|
|
242
|
-
full_audio = full_audio * min(20.0, 0.9 / max(final_peak, 1e-4))
|
|
266
|
+
full_audio = amplify(full_audio)
|
|
243
267
|
fp16 = (device == "cuda")
|
|
244
268
|
result = model.transcribe(
|
|
245
269
|
full_audio,
|
package/npm-shrinkwrap.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "omnius",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.420",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "omnius",
|
|
9
|
-
"version": "1.0.
|
|
9
|
+
"version": "1.0.420",
|
|
10
10
|
"bundleDependencies": [
|
|
11
11
|
"image-to-ascii"
|
|
12
12
|
],
|
package/package.json
CHANGED