osborn 0.9.177 → 0.9.179

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -167,7 +167,7 @@ ensureCompactionSettings();
167
167
  // (1M) window; without it opus runs at its 200K default. NOTE: 1M activation
168
168
  // also depends on account entitlement (auto on Team seats; else usage credits).
169
169
  process.env.ENABLE_1M_CONTEXT = '1';
170
- process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '92';
170
+ process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '75'; // auto-compact at 75% of 1M context (750k tokens) — prevents compaction at extreme tail
171
171
  // Research mode tools — full research capabilities
172
172
  // Named sub-agents — the orchestrator delegates to these specialists. Each has
173
173
  // a specific role, model, and tool set. Module-level + exported so the HTTP
@@ -276,13 +276,13 @@ export const NAMED_AGENTS = {
276
276
  },
277
277
  writer: {
278
278
  description: [
279
- 'Execution agent with file write/edit permissions (Sonnet).',
279
+ 'Execution agent with file write/edit permissions (Opus).',
280
280
  'Handles ALL file operations: code, config, docs, scripts, data files.',
281
281
  'VERIFY-FIRST workflow: checks assumptions before making changes, runs tests after.',
282
282
  'If anything is unclear, asks the main agent for clarification before touching files.',
283
283
  ].join(' '),
284
284
  tools: ['Read', 'Write', 'Edit', 'MultiEdit', 'Bash', 'Glob', 'Grep', 'NotebookRead', 'NotebookEdit'],
285
- model: 'sonnet',
285
+ model: 'opus',
286
286
  prompt: [
287
287
  'You are Osborn\'s writer agent. You execute file changes with a verify-first approach.',
288
288
  '',
@@ -329,12 +329,12 @@ export const NAMED_AGENTS = {
329
329
  },
330
330
  tester: {
331
331
  description: [
332
- 'Test-runner agent (Sonnet). Use for: running test suites, executing builds, interpreting',
332
+ 'Test-runner agent (Opus). Use for: running test suites, executing builds, interpreting',
333
333
  'CI failures, checking compilation errors, verifying that a change did not break anything.',
334
334
  'Returns structured pass/fail results with exact output — does NOT edit files.',
335
335
  ].join(' '),
336
336
  tools: ['Bash', 'Read', 'Glob', 'Grep', 'Write', 'Edit'],
337
- model: 'sonnet',
337
+ model: 'opus',
338
338
  prompt: [
339
339
  'You are Osborn\'s tester agent. Your job is running tests and builds, then reporting results.',
340
340
  '',
package/dist/voice-io.js CHANGED
@@ -131,8 +131,8 @@ export const DEFAULT_VOICE_IO_CONFIG = {
131
131
  export const DIRECT_MODE_STT = {
132
132
  // provider: 'groq-whisper', model: 'whisper-large-v3-turbo', // Batch — needs VAD
133
133
  // provider: 'openai-whisper', model: 'whisper-1', // Batch — needs VAD
134
- provider: 'deepgram', model: 'nova-3', language: 'en', // Streaming, silence-based endpointing
135
- // provider: 'deepgram-flux', model: 'flux-general-en', language: 'en', // Streaming, ML-based turn detection — Flux V2 has silent keepalive timeout bug (~30s silence kills connection)
134
+ // provider: 'deepgram', model: 'nova-3', language: 'en', // Streaming, silence-based endpointing
135
+ provider: 'deepgram-flux', model: 'flux-general-en', language: 'en', // Streaming, ML-based turn detection (requires Deepgram V2 access)
136
136
  };
137
137
  export const DIRECT_MODE_TTS = {
138
138
  // provider: 'deepgram', model: 'aura-2-asteria-en', // WebSocket-based: handles TTS abort cleanly (no unrecoverable crash on interruption)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.177",
3
+ "version": "0.9.179",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {
@@ -0,0 +1,77 @@
1
+ /**
2
+ * Regression tests for CLAUDE_AUTOCOMPACT_PCT_OVERRIDE in claude-llm.ts
3
+ *
4
+ * Scope: verifies that the compaction threshold is set to a value that
5
+ * leaves meaningful context headroom before compaction fires.
6
+ *
7
+ * Derived from: requirements documented in CLAUDE.md (ENABLE_1M_CONTEXT,
8
+ * compaction rationale comments in claude-llm.ts) and the 0.9.177 → 0.9.178
9
+ * change that lowered the threshold from 92 → 60.
10
+ * NOT derived from the implementation diff.
11
+ */
12
+
13
+ import { strict as assert } from 'node:assert'
14
+ import { readFileSync } from 'node:fs'
15
+ import { dirname, join } from 'node:path'
16
+ import { fileURLToPath } from 'node:url'
17
+
18
+ const here = dirname(fileURLToPath(import.meta.url))
19
+ const src = readFileSync(join(here, '../src/claude-llm.ts'), 'utf8')
20
+
21
+ let passed = 0
22
+ let failed = 0
23
+
24
+ function test(name: string, fn: () => void) {
25
+ try {
26
+ fn()
27
+ console.log(` ✅ ${name}`)
28
+ passed++
29
+ } catch (err: any) {
30
+ console.error(` ❌ ${name}`)
31
+ console.error(` ${err.message}`)
32
+ failed++
33
+ }
34
+ }
35
+
36
+ // Extract the numeric value of CLAUDE_AUTOCOMPACT_PCT_OVERRIDE from source
37
+ const match = src.match(/process\.env\.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE\s*=\s*'(\d+)'/)
38
+ const pctValue = match ? parseInt(match[1], 10) : NaN
39
+
40
+ console.log('\n=== autocompact PCT override regression tests ===\n')
41
+ console.log(` (detected value: CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '${pctValue}')`)
42
+ console.log()
43
+
44
+ test('CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is set in claude-llm.ts', () => {
45
+ assert.ok(
46
+ !isNaN(pctValue),
47
+ 'CLAUDE_AUTOCOMPACT_PCT_OVERRIDE must be explicitly set in claude-llm.ts'
48
+ )
49
+ })
50
+
51
+ test('CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is a valid percentage (1–99)', () => {
52
+ assert.ok(
53
+ pctValue >= 1 && pctValue <= 99,
54
+ `CLAUDE_AUTOCOMPACT_PCT_OVERRIDE must be between 1 and 99, got ${pctValue}`
55
+ )
56
+ })
57
+
58
+ test('CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is ≤ 92 (prevents compaction at extreme tail)', () => {
59
+ // Values above 92 leave almost no headroom before hitting the token limit.
60
+ // The long-standing default was 92; anything higher than that risks the
61
+ // "compacting every message" bug described in memory/osborn-1m-context-and-worker-restart.md
62
+ assert.ok(
63
+ pctValue <= 92,
64
+ `CLAUDE_AUTOCOMPACT_PCT_OVERRIDE is ${pctValue} which is above 92 — ` +
65
+ 'values this high risk compaction firing on every message at the context ceiling'
66
+ )
67
+ })
68
+
69
+ test('ENABLE_1M_CONTEXT is also set in claude-llm.ts', () => {
70
+ assert.ok(
71
+ src.includes("process.env.ENABLE_1M_CONTEXT = '1'"),
72
+ "ENABLE_1M_CONTEXT = '1' must be set alongside CLAUDE_AUTOCOMPACT_PCT_OVERRIDE"
73
+ )
74
+ })
75
+
76
+ console.log(`\n--- ${passed + failed} tests: ${passed} passed, ${failed} failed ---\n`)
77
+ if (failed > 0) process.exit(1)
@@ -0,0 +1,177 @@
1
+ /**
2
+ * Regression tests for voice-io.ts STT configuration
3
+ *
4
+ * Scope: covers the 0.9.175 → 0.9.177 change that switches DIRECT_MODE_STT
5
+ * from deepgram-flux (ML turn detection) back to deepgram (silence-based
6
+ * endpointing), due to Flux V2 keepalive timeout bug (~30s of silence kills
7
+ * the connection).
8
+ *
9
+ * Derived from: requirements in CLAUDE.md (Three Voice Modes), voice-io.ts
10
+ * interface definitions, and the documented bug. NOT derived from the diff.
11
+ */
12
+
13
+ import { strict as assert } from 'node:assert'
14
+
15
+ // ──────────────────────────────────────────────────────────────────────────────
16
+ // Inline the minimal types — avoid importing from the module which requires
17
+ // live LK plugins (can't instantiate in a unit test without credentials).
18
+ // ──────────────────────────────────────────────────────────────────────────────
19
+ interface STTConfig {
20
+ provider: 'deepgram' | 'deepgram-flux' | 'groq-whisper' | 'openai-whisper'
21
+ model?: string
22
+ language?: string
23
+ eotThreshold?: number
24
+ eotTimeoutMs?: number
25
+ }
26
+
27
+ // ──────────────────────────────────────────────────────────────────────────────
28
+ // Read the compiled output so we get the real runtime values, not just TS types
29
+ // ──────────────────────────────────────────────────────────────────────────────
30
+ // We import from the transpiled dist to avoid plugin side-effects at import time.
31
+ // If dist is missing, fall back to a direct static-analysis of the source file.
32
+
33
+ let DIRECT_MODE_STT_PROVIDER: string | undefined
34
+ let DIRECT_MODE_STT_MODEL: string | undefined
35
+ let DIRECT_MODE_STT_LANG: string | undefined
36
+
37
+ try {
38
+ // dist/voice-io.js is the compiled output of the build step
39
+ const mod = await import('../dist/voice-io.js')
40
+ const cfg: STTConfig = mod.DIRECT_MODE_STT
41
+ DIRECT_MODE_STT_PROVIDER = cfg.provider
42
+ DIRECT_MODE_STT_MODEL = cfg.model
43
+ DIRECT_MODE_STT_LANG = cfg.language
44
+ } catch (e) {
45
+ // Fallback: parse source file statically to extract the active provider line
46
+ const { readFileSync } = await import('node:fs')
47
+ const { fileURLToPath } = await import('node:url')
48
+ const { dirname, join } = await import('node:path')
49
+ const here = dirname(fileURLToPath(import.meta.url))
50
+ const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
51
+
52
+ // Find the non-commented active provider line in DIRECT_MODE_STT
53
+ const providerMatch = src.match(
54
+ /export const DIRECT_MODE_STT[\s\S]*?^\s+provider:\s*'([^']+)'/m
55
+ )
56
+ const modelMatch = src.match(
57
+ /export const DIRECT_MODE_STT[\s\S]*?model:\s*'([^']+)'/m
58
+ )
59
+ const langMatch = src.match(
60
+ /export const DIRECT_MODE_STT[\s\S]*?language:\s*'([^']+)'/m
61
+ )
62
+ DIRECT_MODE_STT_PROVIDER = providerMatch?.[1]
63
+ DIRECT_MODE_STT_MODEL = modelMatch?.[1]
64
+ DIRECT_MODE_STT_LANG = langMatch?.[1]
65
+ }
66
+
67
+ // ──────────────────────────────────────────────────────────────────────────────
68
+ // Tests
69
+ // ──────────────────────────────────────────────────────────────────────────────
70
+
71
+ let passed = 0
72
+ let failed = 0
73
+
74
+ function test(name: string, fn: () => void) {
75
+ try {
76
+ fn()
77
+ console.log(` ✅ ${name}`)
78
+ passed++
79
+ } catch (err: any) {
80
+ console.error(` ❌ ${name}`)
81
+ console.error(` ${err.message}`)
82
+ failed++
83
+ }
84
+ }
85
+
86
+ console.log('\n=== voice-io STT config regression tests ===\n')
87
+
88
+ // ── 1. Provider must be deepgram (not deepgram-flux) ──────────────────────────
89
+ test('DIRECT_MODE_STT.provider is "deepgram" (not deepgram-flux)', () => {
90
+ assert.equal(
91
+ DIRECT_MODE_STT_PROVIDER,
92
+ 'deepgram',
93
+ `Expected provider "deepgram" but got "${DIRECT_MODE_STT_PROVIDER}". ` +
94
+ 'deepgram-flux has a known keepalive timeout bug (silent ~30s kills the connection).'
95
+ )
96
+ })
97
+
98
+ test('DIRECT_MODE_STT.provider is NOT "deepgram-flux"', () => {
99
+ assert.notEqual(
100
+ DIRECT_MODE_STT_PROVIDER,
101
+ 'deepgram-flux',
102
+ 'deepgram-flux must not be the active provider — ' +
103
+ 'Flux V2 has silent keepalive timeout bug (~30s silence kills connection).'
104
+ )
105
+ })
106
+
107
+ // ── 2. Model and language ──────────────────────────────────────────────────────
108
+ test('DIRECT_MODE_STT.model is "nova-3"', () => {
109
+ assert.equal(
110
+ DIRECT_MODE_STT_MODEL,
111
+ 'nova-3',
112
+ `Expected model "nova-3" but got "${DIRECT_MODE_STT_MODEL}".`
113
+ )
114
+ })
115
+
116
+ test('DIRECT_MODE_STT.language is "en"', () => {
117
+ assert.equal(
118
+ DIRECT_MODE_STT_LANG,
119
+ 'en',
120
+ `Expected language "en" but got "${DIRECT_MODE_STT_LANG}".`
121
+ )
122
+ })
123
+
124
+ // ── 3. STTConfig type-level: provider values are the expected set ─────────────
125
+ // (These are compile-time guarantees verified at build time, but we assert
126
+ // them at runtime too so future changes surface here.)
127
+ test('DIRECT_MODE_STT provider is one of the valid union members', () => {
128
+ const validProviders = ['deepgram', 'deepgram-flux', 'groq-whisper', 'openai-whisper']
129
+ assert.ok(
130
+ validProviders.includes(DIRECT_MODE_STT_PROVIDER!),
131
+ `provider "${DIRECT_MODE_STT_PROVIDER}" is not in valid set: ${validProviders.join(', ')}`
132
+ )
133
+ })
134
+
135
+ // ── 4. Backward-compat: createSTT deepgram path shape ────────────────────────
136
+ // We verify the source still has the deepgram case in createSTT so the switch
137
+ // won't hit the `default: throw` path at runtime.
138
+ test('createSTT source still contains deepgram case (backward compat)', async () => {
139
+ const { readFileSync } = await import('node:fs')
140
+ const { fileURLToPath } = await import('node:url')
141
+ const { dirname, join } = await import('node:path')
142
+ const here = dirname(fileURLToPath(import.meta.url))
143
+ const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
144
+ assert.ok(
145
+ src.includes("case 'deepgram':"),
146
+ "createSTT source must still contain case 'deepgram' — removing it would throw at runtime"
147
+ )
148
+ })
149
+
150
+ test('createSTT source still contains deepgram-flux case (backward compat)', async () => {
151
+ const { readFileSync } = await import('node:fs')
152
+ const { fileURLToPath } = await import('node:url')
153
+ const { dirname, join } = await import('node:path')
154
+ const here = dirname(fileURLToPath(import.meta.url))
155
+ const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
156
+ assert.ok(
157
+ src.includes("case 'deepgram-flux':"),
158
+ "createSTT source must still contain case 'deepgram-flux' — it must remain selectable via config"
159
+ )
160
+ })
161
+
162
+ // ── 5. deepgram endpointing value in createSTT ────────────────────────────────
163
+ test('createSTT deepgram case uses 550ms endpointing (mid-sentence fragment prevention)', async () => {
164
+ const { readFileSync } = await import('node:fs')
165
+ const { fileURLToPath } = await import('node:url')
166
+ const { dirname, join } = await import('node:path')
167
+ const here = dirname(fileURLToPath(import.meta.url))
168
+ const src = readFileSync(join(here, '../src/voice-io.ts'), 'utf8')
169
+ assert.ok(
170
+ src.includes('endpointing: 550'),
171
+ 'createSTT deepgram case must use endpointing: 550ms to prevent mid-sentence transcript fragments'
172
+ )
173
+ })
174
+
175
+ // ── Summary ───────────────────────────────────────────────────────────────────
176
+ console.log(`\n--- ${passed + failed} tests: ${passed} passed, ${failed} failed ---\n`)
177
+ if (failed > 0) process.exit(1)