@fastagent-sh/voicenote 0.22.1 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/cli.ts DELETED
@@ -1,2698 +0,0 @@
1
- #!/usr/bin/env bun
2
- import { cac } from 'cac'
3
- import packageJson from '../package.json' with { type: 'json' }
4
- import { parseLockOwner } from './runLock'
5
- import { tosObject, type TosConfig as VolcanoTosConfig } from './tos'
6
- import { applyOutcome, buildJobsView, classify, emptyState, localIso, MAX_ATTEMPTS, migrateLegacyState, ownsOutput, parseJobsLimit, parseStateFile, parseStrictJson, patchJob, pruneUnseen, reconcileInterrupted, requeueFailed, startAttempt, SUMMARY_FAILED_STATUS, type CurrentJob, type JobRecord, type StateFile } from './jobs'
7
- import { createHash, randomUUID } from 'node:crypto'
8
- import { appendFile, chmod, mkdir, readFile, writeFile, copyFile, rename, unlink, stat, readdir } from 'node:fs/promises'
9
- import { existsSync, readFileSync, readdirSync, mkdirSync, writeFileSync, appendFileSync, openSync, closeSync, statSync, readSync, unlinkSync, renameSync } from 'node:fs'
10
- import { dlopen, FFIType, suffix } from 'bun:ffi'
11
- import { basename, dirname, extname, join, resolve } from 'node:path'
12
- import { fileURLToPath, pathToFileURL } from 'node:url'
13
- import { spawn, spawnSync } from 'node:child_process'
14
- import os from 'node:os'
15
-
16
- const VERSION = packageJson.version
17
- const LAUNCH_AGENT_LABEL = 'sh.fastagent.voicenote'
18
- const LAUNCH_AGENT_LABEL_LEGACY = 'com.kid7st.voicenote' // pre-fastagent installs; cleaned up on install
19
- const TASK_NAME = 'VoiceNote' // Windows Task Scheduler name (mac uses LAUNCH_AGENT_LABEL)
20
-
21
- // Single switch every platform branch routes through. Declared before the path
22
- // consts so they can read it.
23
- const IS_WINDOWS = process.platform === 'win32'
24
- const IS_MAC = process.platform === 'darwin'
25
-
26
- // Per-OS base dirs. Windows -> native AppData (Roaming for config, Local for
27
- // logs/lock/state); mac/Linux -> ~/.config and ~/.local/state. appConfigDir /
28
- // appStateDir are hoisted function decls (defined just below).
29
- const CONFIG_DIR = appConfigDir()
30
- const STATE_DIR = appStateDir()
31
- const LOG_DIR = join(STATE_DIR, 'logs')
32
- const LOCK_PATH = join(STATE_DIR, 'run.lock')
33
- const CONFIG_ENV_PATH = join(CONFIG_DIR, 'config.json')
34
-
35
- const AUDIO_EXTENSIONS = new Set(['.mp3', '.wav', '.m4a', '.wma', '.aac', '.flac'])
36
-
37
- // Per-OS base directory resolution (see CONFIG_DIR / STATE_DIR above).
38
- function appConfigDir(): string {
39
- if (IS_WINDOWS) return join(process.env.APPDATA || join(os.homedir(), 'AppData', 'Roaming'), 'voicenote')
40
- return join(os.homedir(), '.config', 'voicenote')
41
- }
42
- function appStateDir(): string {
43
- if (IS_WINDOWS) return join(process.env.LOCALAPPDATA || join(os.homedir(), 'AppData', 'Local'), 'voicenote')
44
- return join(os.homedir(), '.local', 'state', 'voicenote')
45
- }
46
-
47
- type Json = Record<string, any>
48
-
49
- type Recording = {
50
- sourcePath: string
51
- sizeBytes: number
52
- modifiedAt: string
53
- durationSeconds: number | null
54
- sourceId: string
55
- recordedAt: Date
56
- }
57
-
58
- type LocalFiles = {
59
- audio: string
60
- transcript: string
61
- notes: string
62
- metadata: string
63
- }
64
-
65
- type SpeakerSelf = { name: string | null; aliases: string[] }
66
- type SpeakerKnown = { name: string; aliases: string[]; relationship?: string | null }
67
- type SpeakersConfig = { self: SpeakerSelf; known: SpeakerKnown[] }
68
-
69
-
70
- type VolcanoConfig = {
71
- apiKey: string // X-Api-Key (new Volcano console)
72
- resourceId: string
73
- language?: string
74
- tos: VolcanoTosConfig
75
- }
76
-
77
- /** How this install runs pi: which binary, which model, what it may read. */
78
- type PiConfig = {
79
- bin: string
80
- /** Set when pi ships as plain JS next to a bundled bun: `<bin> <cli> <args>`. */
81
- cli: string | null
82
- model: string | null
83
- thinking: string
84
- /** Comma-separated tool list; empty = run the summary without tools. */
85
- tools: string
86
- contextDir: string
87
- retries: number
88
- authPath: string
89
- }
90
-
91
- type Config = {
92
- recordDir: string
93
- workspace: string
94
- minBytes: number
95
- minDurationSeconds: number
96
- maxAgeHours: number
97
- speakers: SpeakersConfig
98
- volcano: VolcanoConfig | null
99
- ffprobeBin: string
100
- pi: PiConfig
101
- /** Added to the environment of every process vn spawns. */
102
- childEnv: Record<string, string>
103
- }
104
-
105
- // ────────────────────────────────────────────────────────────────────────────
106
- // Settings → Config
107
- //
108
- // config.json is the only persisted source; the inherited environment overrides
109
- // it for this process only. Everything the program needs is resolved once, in
110
- // getConfig(), and passed down as a frozen Config — no code below reads a
111
- // business setting out of process.env, so behaviour can never depend on whether
112
- // some earlier call happened to hydrate it.
113
- // ────────────────────────────────────────────────────────────────────────────
114
-
115
- // Config keys accepted by `vn config set` and loaded from config.json when the
116
- // inherited environment does not already define them.
117
- const ENV_KEYS = [
118
- 'VOICENOTE_DEVICE_VOLUME',
119
- 'VOICENOTE_RECORD_DIR',
120
- 'VOICENOTE_WORKSPACE',
121
- 'VOICENOTE_MIN_BYTES',
122
- 'VOICENOTE_MIN_DURATION_SECONDS',
123
- 'VOICENOTE_MAX_AGE_HOURS',
124
- 'VOLCANO_ASR_KEY',
125
- 'VOLCANO_ASR_RESOURCE_ID',
126
- 'VOLCANO_ASR_LANGUAGE',
127
- 'VOLCANO_TOS_REGION',
128
- 'VOLCANO_TOS_ENDPOINT',
129
- 'VOLCANO_TOS_BUCKET',
130
- 'VOLCANO_TOS_ACCESS_KEY',
131
- 'VOLCANO_TOS_SECRET_KEY',
132
- 'VOLCANO_TOS_KEEP',
133
- 'VOICENOTE_PI_BIN',
134
- 'VOICENOTE_PI_CLI',
135
- 'PI_CODING_AGENT_DIR',
136
- 'VOICENOTE_FFPROBE_BIN',
137
- 'VOICENOTE_PI_MODEL',
138
- 'VOICENOTE_PI_RETRIES',
139
- 'VOICENOTE_PI_THINKING',
140
- 'VOICENOTE_PI_SUMMARY_TOOLS',
141
- 'VOICENOTE_CONTEXT_DIR',
142
- 'http_proxy', 'https_proxy', 'all_proxy', 'no_proxy',
143
- 'HTTP_PROXY', 'HTTPS_PROXY', 'ALL_PROXY', 'NO_PROXY',
144
- 'LOCAL_PROXY_HOST', 'LOCAL_PROXY_PORT', 'LOCAL_NO_PROXY',
145
- 'OPENAI_API_KEY',
146
- 'DEEPSEEK_API_KEY',
147
- ]
148
-
149
- // Volcano endpoints (TOS object storage + openspeech ASR) should NEVER go through
150
- // the SOCKS/HTTP proxy that pi (ChatGPT Codex OAuth) may need:
151
- // 1) the proxy bandwidth often chokes on multi-megabyte PUTs to TOS
152
- // 2) routing China-mainland Volcano APIs through an overseas proxy is slower / unreliable
153
- const VOLCANO_NO_PROXY_HOSTS = ['.volces.com', '.volcengineapi.com', 'openspeech.bytedance.com']
154
-
155
- function systemProxyUrl(): string | null {
156
- if (process.platform !== 'darwin') return null
157
- try {
158
- const out = spawnSync('scutil', ['--proxy'], { encoding: 'utf8', timeout: 3000 })
159
- if (out.status !== 0 || !out.stdout) return null
160
- const get = (k: string) => out.stdout.match(new RegExp(`\\b${k}\\s*:\\s*(\\S+)`))?.[1]
161
- if (get('HTTPSEnable') === '1' && get('HTTPSProxy') && get('HTTPSPort')) return `http://${get('HTTPSProxy')}:${get('HTTPSPort')}`
162
- if (get('HTTPEnable') === '1' && get('HTTPProxy') && get('HTTPPort')) return `http://${get('HTTPProxy')}:${get('HTTPPort')}`
163
- return null
164
- } catch { return null }
165
- }
166
-
167
- type Settings = Record<string, string>
168
-
169
- /** This run's settings: config.json, overridden by the inherited environment. */
170
- function readSettings(file: Record<string, unknown>): Settings {
171
- const settings: Settings = {}
172
- for (const key of ENV_KEYS) {
173
- const inherited = process.env[key]
174
- if (inherited !== undefined) settings[key] = inherited
175
- else if (typeof file[key] === 'string') settings[key] = file[key] as string
176
- }
177
- return settings
178
- }
179
-
180
- /**
181
- * Proxy variables, resolved from settings or the macOS system proxy. Returned as
182
- * a map instead of being pushed onto process.env alone because Bun does not hand
183
- * a child the variables this process added after startup — every spawn site
184
- * passes them explicitly (covered by summary.test.ts).
185
- */
186
- function proxyEnv(s: Settings): Record<string, string> {
187
- const url = s.https_proxy || s.HTTPS_PROXY || s.http_proxy || s.HTTP_PROXY || s.all_proxy || s.ALL_PROXY
188
- || (s.LOCAL_PROXY_HOST && s.LOCAL_PROXY_PORT ? `http://${s.LOCAL_PROXY_HOST}:${s.LOCAL_PROXY_PORT}` : '')
189
- || systemProxyUrl()
190
- if (!url) return {}
191
- const env: Record<string, string> = {}
192
- for (const key of ['http_proxy', 'https_proxy', 'all_proxy', 'HTTP_PROXY', 'HTTPS_PROXY', 'ALL_PROXY']) env[key] = s[key] || url
193
- const base = s.LOCAL_NO_PROXY || s.no_proxy || s.NO_PROXY || 'localhost,127.0.0.1,::1'
194
- const bypass = [...new Set([...base.split(',').map(v => v.trim()).filter(Boolean), ...VOLCANO_NO_PROXY_HOSTS])].join(',')
195
- env.no_proxy = bypass
196
- env.NO_PROXY = bypass
197
- return env
198
- }
199
-
200
- function volcanoFrom(s: Settings): VolcanoConfig | null {
201
- const apiKey = s.VOLCANO_ASR_KEY || ''
202
- const tosAccess = s.VOLCANO_TOS_ACCESS_KEY
203
- const tosSecret = s.VOLCANO_TOS_SECRET_KEY
204
- const bucket = s.VOLCANO_TOS_BUCKET
205
- if (!apiKey || !tosAccess || !tosSecret || !bucket) return null
206
- const region = s.VOLCANO_TOS_REGION || 'cn-guangzhou'
207
- const endpoint = s.VOLCANO_TOS_ENDPOINT || `tos-s3-${region}.volces.com`
208
- const keep = ['1', 'true', 'yes'].includes((s.VOLCANO_TOS_KEEP || '0').toLowerCase())
209
- return {
210
- apiKey,
211
- resourceId: s.VOLCANO_ASR_RESOURCE_ID || 'volc.seedasr.auc',
212
- language: s.VOLCANO_ASR_LANGUAGE || undefined,
213
- tos: { endpoint, region, bucket, accessKey: tosAccess, secretKey: tosSecret, keep },
214
- }
215
- }
216
-
217
- function volcanoAuthHeaders(volc: VolcanoConfig, taskId: string, includeSequence: boolean): Record<string, string> {
218
- const base: Record<string, string> = {
219
- 'X-Api-Resource-Id': volc.resourceId,
220
- 'X-Api-Request-Id': taskId,
221
- 'Content-Type': 'application/json',
222
- }
223
- if (includeSequence) base['X-Api-Sequence'] = '-1'
224
- base['X-Api-Key'] = volc.apiKey
225
- return base
226
- }
227
-
228
- function settingNumber(s: Settings, key: string, fallback: number): number {
229
- const raw = s[key]
230
- const value = raw === undefined || raw === '' ? fallback : Number(raw)
231
- if (!Number.isFinite(value) || value < 0) throw new Error(`Invalid ${key}: expected a non-negative number, got '${raw}'`)
232
- return value
233
- }
234
-
235
- let configCache: Config | null = null
236
-
237
- function getConfig(): Config {
238
- if (configCache) return configCache
239
- const file = loadConfigJson()
240
- const s = readSettings(file)
241
- const proxy = proxyEnv(s)
242
- // vn's own fetch (the ChatGPT OAuth flow) reads the proxy from the process
243
- // environment, so the derived values have to land there as well.
244
- for (const [key, value] of Object.entries(proxy)) process.env[key] = value
245
- // Passed to every child: the proxy, plus the credentials and config dir that
246
- // pi — not vn — resolves for itself.
247
- const childEnv = { ...proxy }
248
- for (const key of ['PI_CODING_AGENT_DIR', 'OPENAI_API_KEY', 'DEEPSEEK_API_KEY']) if (s[key]) childEnv[key] = s[key]!
249
-
250
- const deviceVolume = s.VOICENOTE_DEVICE_VOLUME || 'VTR6500'
251
- const workspace = expandHome(s.VOICENOTE_WORKSPACE || '~/Documents/meetings')
252
- // pi keeps credentials in its config dir, which PI_CODING_AGENT_DIR relocates.
253
- // Point it at a voicenote-owned directory to get an auth.json that only the
254
- // pipeline reads and refreshes: an interactive pi session rewrites its own
255
- // auth.json wholesale on exit and has already dropped entries that way.
256
- const piAgentDir = expandHome(s.PI_CODING_AGENT_DIR || join(os.homedir(), '.pi', 'agent'))
257
- configCache = Object.freeze({
258
- recordDir: expandHome(s.VOICENOTE_RECORD_DIR || `/Volumes/${deviceVolume}/RECORD`),
259
- workspace,
260
- minBytes: settingNumber(s, 'VOICENOTE_MIN_BYTES', 100000),
261
- minDurationSeconds: settingNumber(s, 'VOICENOTE_MIN_DURATION_SECONDS', 60),
262
- // Only recordings from the last N hours are picked up (0 = no limit), so a
263
- // fresh install doesn't drain the recorder's entire history.
264
- maxAgeHours: settingNumber(s, 'VOICENOTE_MAX_AGE_HOURS', 48),
265
- volcano: volcanoFrom(s),
266
- speakers: normalizeSpeakers(file.speakers ?? DEFAULT_SPEAKERS),
267
- // ffprobe is the only ffmpeg-suite binary the pipeline uses (duration
268
- // detection); a configurable path lets the GUI point at its bundled copy.
269
- ffprobeBin: expandHome(s.VOICENOTE_FFPROBE_BIN || 'ffprobe'),
270
- pi: {
271
- bin: expandHome(s.VOICENOTE_PI_BIN || 'pi'),
272
- cli: s.VOICENOTE_PI_CLI ? expandHome(s.VOICENOTE_PI_CLI) : null,
273
- // pi's --model accepts "provider/id" (e.g. openai-codex/gpt-5.6-sol), so
274
- // this one setting pins both. Null = whatever pi is configured to use.
275
- model: (s.VOICENOTE_PI_MODEL || '').trim() || null,
276
- thinking: s.VOICENOTE_PI_THINKING || 'high',
277
- // Default ON: let the summary model read/grep prior notes for cross-reference
278
- // consistency. Set VOICENOTE_PI_SUMMARY_TOOLS='' to disable.
279
- tools: s.VOICENOTE_PI_SUMMARY_TOOLS === undefined ? 'read,grep' : s.VOICENOTE_PI_SUMMARY_TOOLS.trim(),
280
- // Directory the summary model may read/grep. The published default must not
281
- // reach outside the configured workspace.
282
- contextDir: expandHome(s.VOICENOTE_CONTEXT_DIR || workspace),
283
- retries: Math.max(1, Math.floor(settingNumber(s, 'VOICENOTE_PI_RETRIES', 3))),
284
- authPath: join(piAgentDir, 'auth.json'),
285
- },
286
- childEnv,
287
- })
288
- return configCache
289
- }
290
-
291
- // ────────────────────────────────────────────────────────────────────────────
292
- // Config files (~/.config/voicenote)
293
- // ────────────────────────────────────────────────────────────────────────────
294
-
295
- const DEFAULT_SPEAKERS: SpeakersConfig = { self: { name: null, aliases: [] }, known: [] }
296
-
297
- function normalizeSpeakers(data: unknown): SpeakersConfig {
298
- const raw = (data && typeof data === 'object') ? data as Partial<SpeakersConfig> : {}
299
- return {
300
- self: {
301
- name: typeof raw.self?.name === 'string' ? raw.self.name : null,
302
- aliases: Array.isArray(raw.self?.aliases) ? raw.self!.aliases.filter((a): a is string => typeof a === 'string') : [],
303
- },
304
- known: Array.isArray(raw.known)
305
- ? raw.known
306
- .filter((k): k is SpeakerKnown => !!k && typeof k === 'object' && typeof (k as SpeakerKnown).name === 'string')
307
- .map(k => ({ name: k.name, aliases: Array.isArray(k.aliases) ? k.aliases.filter((a): a is string => typeof a === 'string') : [], relationship: k.relationship ?? null }))
308
- : [],
309
- }
310
- }
311
-
312
- function loadConfigJson(): Record<string, unknown> {
313
- if (!existsSync(CONFIG_ENV_PATH)) return {}
314
- let value: unknown
315
- try { value = JSON.parse(readFileSync(CONFIG_ENV_PATH, 'utf8')) } catch (e: any) {
316
- throw new Error(`${CONFIG_ENV_PATH} is invalid JSON: ${e?.message || e}`)
317
- }
318
- if (!value || typeof value !== 'object' || Array.isArray(value)) throw new Error(`${CONFIG_ENV_PATH} must contain a JSON object`)
319
- return value as Record<string, unknown>
320
- }
321
-
322
- // ────────────────────────────────────────────────────────────────────────────
323
- // Misc helpers
324
- // ────────────────────────────────────────────────────────────────────────────
325
-
326
- function expandHome(path: string): string {
327
- return path.replace(/^(?:~|\$\{?HOME\}?)(?=\/|$)/, os.homedir())
328
- }
329
-
330
- function nowIso(): string { return new Date().toISOString() }
331
- function pad(n: number): string { return String(n).padStart(2, '0') }
332
-
333
- function dateParts(d: Date): { month: string; prefix: string } {
334
- const month = `${d.getFullYear()}-${pad(d.getMonth() + 1)}`
335
- // Local time in filenames uses HH-MM only (the recorder cannot produce two recordings within the same minute)
336
- const prefix = `${month}-${pad(d.getDate())}-${pad(d.getHours())}-${pad(d.getMinutes())}`
337
- return { month, prefix }
338
- }
339
-
340
- function safeSlug(text: string, maxLen = 48): string {
341
- const cleaned = (text || '').trim().replace(/[\\/:*?"<>|\n\r\t]+/g, '-').replace(/\s+/g, '-').replace(/^-+|-+$/g, '')
342
- return cleaned.slice(0, maxLen).replace(/-+$/g, '') || 'note'
343
- }
344
-
345
- function formatSeconds(seconds: number | null | undefined): string {
346
- const total = Math.max(0, Math.round(seconds || 0))
347
- const h = Math.floor(total / 3600)
348
- const m = Math.floor((total % 3600) / 60)
349
- const s = total % 60
350
- return h ? `${pad(h)}:${pad(m)}:${pad(s)}` : `${pad(m)}:${pad(s)}`
351
- }
352
-
353
- // ────────────────────────────────────────────────────────────────────────────
354
- // File state IO
355
- // ────────────────────────────────────────────────────────────────────────────
356
-
357
- async function ensureDirs(config: Config): Promise<void> {
358
- for (const dir of ['_state', '_index', '_audio', '_transcripts', '_metadata']) {
359
- await mkdir(join(config.workspace, dir), { recursive: true })
360
- }
361
- }
362
-
363
- // Write via tmp+rename so readers only ever see a complete file. Anything whose
364
- // mere existence is later treated as a signal MUST go through this: a half
365
- // written file that still parses is worse than no file at all.
366
- async function writeFileAtomic(path: string, body: string): Promise<void> {
367
- await mkdir(dirname(path), { recursive: true })
368
- const tmp = `${path}.tmp`
369
- await writeFile(tmp, body, 'utf8')
370
- await rename(tmp, path)
371
- }
372
-
373
- const writeJson = (path: string, data: any) => writeFileAtomic(path, JSON.stringify(data, null, 2))
374
-
375
- async function appendJsonl(path: string, data: any): Promise<void> {
376
- await mkdir(dirname(path), { recursive: true })
377
- await appendFile(path, JSON.stringify(data) + '\n', 'utf8')
378
- }
379
-
380
- const RAW_TRANSCRIPT_MARKER = '## Raw transcript (no lossy cleanup)\n\n'
381
- const RAW_TRANSCRIPT_MARKER_LEGACY = '## 原始 transcript(不做 lossy 清洗)\n\n' // pre-0.18 files on disk
382
-
383
-
384
-
385
- // ────────────────────────────────────────────────────────────────────────────
386
- // Logging (rolling daily log)
387
- // ────────────────────────────────────────────────────────────────────────────
388
-
389
- function dailyLogPath(): string {
390
- const d = new Date()
391
- return join(LOG_DIR, `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}.log`)
392
- }
393
-
394
- // Best-effort side effects (logging, idle-state) must not crash the pipeline, but
395
- // failures should still be observable. Write directly to stderr (not console.error,
396
- // which wireDailyLog wraps and would recurse into the same failing file) once per tag.
397
- const sideEffectWarned = new Set<string>()
398
- function warnSideEffect(where: string, e: unknown): void {
399
- if (sideEffectWarned.has(where)) return
400
- sideEffectWarned.add(where)
401
- process.stderr.write(`[voicenote] non-fatal: ${where} failed: ${e instanceof Error ? e.message : String(e)}\n`)
402
- }
403
-
404
- let logWired = false
405
- function wireDailyLog(): void {
406
- if (logWired) return
407
- logWired = true
408
- try { mkdirSync(LOG_DIR, { recursive: true }) } catch (e) { warnSideEffect('log dir mkdir', e) }
409
- const path = dailyLogPath()
410
- const append = (level: 'INFO' | 'ERROR', args: any[]) => {
411
- const line = args.map(a => (typeof a === 'string' ? a : JSON.stringify(a))).join(' ')
412
- const stamped = `${nowIso()} [${level}] ${line}\n`
413
- try { appendFileSync(path, stamped, 'utf8') } catch (e) { warnSideEffect('daily log append', e) }
414
- }
415
- const origLog = console.log.bind(console)
416
- const origErr = console.error.bind(console)
417
- console.log = (...a: any[]) => { append('INFO', a); origLog(...a) }
418
- console.error = (...a: any[]) => { append('ERROR', a); origErr(...a) }
419
- }
420
-
421
- function formatBytes(bytes: number): string {
422
- if (!Number.isFinite(bytes)) return 'unknown size'
423
- const units = ['B', 'KB', 'MB', 'GB']
424
- let value = bytes
425
- let unit = 0
426
- while (value >= 1024 && unit < units.length - 1) { value /= 1024; unit++ }
427
- return `${value.toFixed(unit === 0 ? 0 : 1)} ${units[unit]}`
428
- }
429
-
430
- function formatElapsed(ms: number): string {
431
- const total = Math.max(0, Math.round(ms / 1000))
432
- const h = Math.floor(total / 3600)
433
- const m = Math.floor((total % 3600) / 60)
434
- const s = total % 60
435
- if (h) return `${h}h ${m}m ${s}s`
436
- if (m) return `${m}m ${s}s`
437
- return `${s}s`
438
- }
439
-
440
- function progressStep(step: number, total: number, title: string, detail?: string): void {
441
- console.log(`▶ Step ${step}/${total}: ${title}${detail ? ` — ${detail}` : ''}`)
442
- // Single hook for live progress: the dashboard shows the same string the log
443
- // does, instead of regex-guessing the step from log text.
444
- reportStep(title)
445
- }
446
-
447
- async function withHeartbeat<T>(label: string, work: () => Promise<T>, heartbeatSeconds = 60): Promise<T> {
448
- const started = Date.now()
449
- const timer = setInterval(() => {
450
- console.log(`… Still working: ${label} (${formatElapsed(Date.now() - started)} elapsed)`)
451
- }, Math.max(10, heartbeatSeconds) * 1000)
452
- ;(timer as any).unref?.()
453
- try {
454
- const result = await work()
455
- console.log(`✓ Done: ${label} (${formatElapsed(Date.now() - started)})`)
456
- return result
457
- } catch (e) {
458
- console.error(`✗ Failed: ${label} after ${formatElapsed(Date.now() - started)}`)
459
- throw e
460
- } finally {
461
- clearInterval(timer)
462
- }
463
- }
464
-
465
- function shouldLogIdleStatus(key: string, intervalMs = 30 * 60 * 1000): boolean {
466
- const path = join(LOG_DIR, 'idle-status.json')
467
- const now = Date.now()
468
- let prev: any = null
469
- try { prev = JSON.parse(readFileSync(path, 'utf8')) } catch {}
470
- const should = prev?.key !== key || now - Number(prev?.at || 0) >= intervalMs
471
- if (should) {
472
- try {
473
- mkdirSync(LOG_DIR, { recursive: true })
474
- writeFileSync(path, JSON.stringify({ key, at: now, iso: nowIso() }, null, 2) + '\n', 'utf8')
475
- } catch (e) { warnSideEffect('idle-status write', e) }
476
- }
477
- return should
478
- }
479
-
480
- type RunMode = 'notes' | 'transcript'
481
- function normalizeRunMode(opts: any): RunMode {
482
- const raw = String(opts.mode || 'notes').toLowerCase()
483
- if (raw === 'note') return 'notes'
484
- if (raw === 'notes' || raw === 'transcript') return raw
485
- throw new Error(`Invalid --mode "${raw}". Use: notes|transcript`)
486
- }
487
-
488
- // ────────────────────────────────────────────────────────────────────────────
489
- // Cross-process lock
490
- // ────────────────────────────────────────────────────────────────────────────
491
-
492
- // Single-instance mutual exclusion via an OS advisory lock (flock) held on an open
493
- // fd. The kernel releases it automatically when the process exits — including
494
- // SIGKILL/crash — so there is NO pid / mtime / heartbeat / stale-steal logic to
495
- // race on. flock is loaded from libSystem, so it is macOS-only; every other
496
- // platform uses the pid+timestamp lockfile below.
497
- const flockFn = (() => {
498
- try {
499
- const lib = dlopen(`libSystem.${suffix}`, { flock: { args: [FFIType.i32, FFIType.i32], returns: FFIType.i32 } })
500
- return lib.symbols.flock as (fd: number, op: number) => number
501
- } catch { return null }
502
- })()
503
- const FLOCK_EX_NB = 2 | 4 // LOCK_EX | LOCK_NB
504
- const FLOCK_UN = 8
505
-
506
- // Lockfile used wherever flock is not available (Windows, Linux). A pid+timestamp
507
- // file, created atomically with 'wx'. We only reclaim an existing lock when its
508
- // owner pid is dead OR the lock is stale (older than STALE_MS). The holder refreshes its timestamp every 5 minutes
509
- // (heartbeat below), so a legitimately long RUNNING job — ASR on a multi-hour
510
- // recording — never looks stale. The staleness escape exists for the pid-reuse
511
- // false positive (owner died, an unrelated process now has its pid, the aliveness
512
- // probe lies); its known cost: a machine asleep >30min can lose the lock on wake
513
- // (timers don't fire while asleep), so the heartbeat verifies ownership before
514
- // each refresh and, if the lock was reclaimed, stops touching it and warns — the
515
- // old run finishes unprotected rather than corrupting the new holder's record.
516
- // Task Scheduler's IgnoreNew already blocks the common 60s overlap; this only has
517
- // to cover a manual `vn run` racing the scheduled one. The tiny create/reclaim
518
- // window is acceptable: its failure mode is conservatively skipping one run (same
519
- // as mac when flock is already held).
520
- async function acquireRunLockFile(): Promise<{ release: () => Promise<void> } | null> {
521
- await mkdir(dirname(LOCK_PATH), { recursive: true })
522
- const STALE_MS = 30 * 60 * 1000
523
- const tryCreate = (): number | null => {
524
- try { return openSync(LOCK_PATH, 'wx') }
525
- catch (e: any) { if (e?.code === 'EEXIST') return null; throw e }
526
- }
527
- let fd = tryCreate()
528
- if (fd === null) {
529
- let reclaim = false
530
- try {
531
- const data = JSON.parse(readFileSync(LOCK_PATH, 'utf8'))
532
- const pid = Number(data.pid), ts = Number(data.ts)
533
- const alive = pidAlive(pid)
534
- const fresh = Number.isFinite(ts) && (Date.now() - ts) < STALE_MS
535
- reclaim = !alive || !fresh
536
- } catch { reclaim = true } // unreadable/corrupt lock -> reclaim
537
- if (!reclaim) return null // another live run holds it
538
- try { unlinkSync(LOCK_PATH) } catch {}
539
- fd = tryCreate()
540
- if (fd === null) return null // someone grabbed it in the gap
541
- }
542
- writeFileSync(fd, JSON.stringify({ pid: process.pid, ts: Date.now() }))
543
- closeSync(fd)
544
- // Tri-state ownership (parse logic + its 'unknown'-on-read-failure invariant
545
- // are pure + tested in runLock.ts). A transient read failure must NOT be
546
- // treated as loss of ownership: that would kill the heartbeat and hand the
547
- // lock away over a momentary glitch — the overlap the heartbeat prevents.
548
- const lockOwnership = () => {
549
- let raw: string | null
550
- try { raw = readFileSync(LOCK_PATH, 'utf8') } catch { raw = null }
551
- return parseLockOwner(raw, process.pid)
552
- }
553
- // Heartbeat: keep ts fresh while we hold the lock; verify ownership first
554
- // (see header comment — the lock can be reclaimed after a long sleep).
555
- // The refresh writes a temp file and renames it into place: a plain
556
- // truncate+write would open a window where a concurrent acquire reads
557
- // empty/partial JSON, treats the lock as corrupt, and reclaims a LIVE lock.
558
- const heartbeat = setInterval(() => {
559
- const owner = lockOwnership()
560
- if (owner === 'reclaimed') {
561
- clearInterval(heartbeat)
562
- console.error('Run lock was reclaimed by another process (machine slept >30min?); this run continues but is no longer protected against overlap.')
563
- return
564
- }
565
- if (owner === 'unknown') { warnSideEffect('run lock heartbeat read', new Error('lock unreadable this tick; will retry')); return }
566
- try {
567
- const tmp = `${LOCK_PATH}.hb-${process.pid}`
568
- writeFileSync(tmp, JSON.stringify({ pid: process.pid, ts: Date.now() }))
569
- renameSync(tmp, LOCK_PATH) // atomic replace, also on Windows
570
- } catch (e) { warnSideEffect('run lock heartbeat', e) }
571
- }, 5 * 60 * 1000)
572
- ;(heartbeat as any).unref?.()
573
- let released = false
574
- const release = async () => {
575
- if (released) return
576
- released = true
577
- clearInterval(heartbeat)
578
- // Only remove the lock when it is provably still OURS. Not 'reclaimed'
579
- // (that's the new holder's lock) and not 'unknown' either: a transient
580
- // read failure could be a reclaimer mid-swap, and the heartbeat does NOT
581
- // rewrite on 'unknown', so unlinking here would leave a real vacuum until
582
- // STALE_MS. Leaking our own lock on a rare transient failure is the lesser
583
- // evil — it self-heals after STALE_MS via the staleness check.
584
- try { if (lockOwnership() === 'mine') unlinkSync(LOCK_PATH) } catch {}
585
- }
586
- process.once('exit', () => { void release() })
587
- process.once('SIGINT', () => { void release(); process.exit(130) })
588
- process.once('SIGTERM', () => { void release(); process.exit(143) })
589
- return { release }
590
- }
591
-
592
- async function acquireRunLock(): Promise<{ release: () => Promise<void> } | null> {
593
- if (!flockFn) return acquireRunLockFile()
594
- await mkdir(dirname(LOCK_PATH), { recursive: true })
595
- // The lock is a regular file we keep open. Builds ≤ 0.15.2 used a *directory*
596
- // here, held purely by its existence, with no pid or refreshed mtime inside — so
597
- // a leftover legacy dir carries NO reliable signal about whether an old `vn run`
598
- // still holds it. Rather than guess (and risk deleting a live lock → concurrent
599
- // double-processing), refuse to auto-reclaim it: warn and skip. Normal upgrades
600
- // don't hit this (≤ 0.15.2 removes its own dir lock on SIGTERM/exit); it only
601
- // appears after a hard crash of an old build, where a one-time manual cleanup is
602
- // the safe move.
603
- let fd: number
604
- try { fd = openSync(LOCK_PATH, 'w') }
605
- catch (e: any) {
606
- if (e?.code !== 'EISDIR') throw e
607
- console.error(`Found a legacy (≤ 0.15.2) lock directory at ${LOCK_PATH}; it carries no liveness info and can't be auto-reclaimed safely. If no 'vn run' is active, remove it once: rm -rf "${LOCK_PATH}" — skipping this run.`)
608
- return null
609
- }
610
- if (flockFn(fd, FLOCK_EX_NB) !== 0) { closeSync(fd); return null } // another run holds it
611
- let released = false
612
- const release = async () => {
613
- if (released) return
614
- released = true
615
- try { flockFn(fd, FLOCK_UN) } catch {}
616
- try { closeSync(fd) } catch {}
617
- }
618
- process.once('exit', () => { void release() })
619
- process.once('SIGINT', () => { void release(); process.exit(130) })
620
- process.once('SIGTERM', () => { void release(); process.exit(143) })
621
- return { release }
622
- }
623
-
624
- // ────────────────────────────────────────────────────────────────────────────
625
- // Recording scan
626
- // ────────────────────────────────────────────────────────────────────────────
627
-
628
- function parseRecordedAt(path: string): Date {
629
- const stem = basename(path, extname(path))
630
- const m = stem.match(/(20\d{12})/)
631
- if (m?.[1]) {
632
- const s = m[1]
633
- return new Date(Number(s.slice(0, 4)), Number(s.slice(4, 6)) - 1, Number(s.slice(6, 8)), Number(s.slice(8, 10)), Number(s.slice(10, 12)), Number(s.slice(12, 14)))
634
- }
635
- return new Date()
636
- }
637
-
638
- async function sha256File(path: string): Promise<string> {
639
- const h = createHash('sha256')
640
- const reader = Bun.file(path).stream().getReader()
641
- while (true) {
642
- const { done, value } = await reader.read()
643
- if (done) break
644
- h.update(value)
645
- }
646
- return h.digest('hex')
647
- }
648
-
649
- async function sourceIdFor(path: string): Promise<string> {
650
- const st = await stat(path)
651
- const digest = await sha256File(path)
652
- return createHash('sha256').update(`${path}|${st.size}|${Math.floor(st.mtimeMs / 1000)}|${digest}`).digest('hex')
653
- }
654
-
655
- function runCommand(command: string, args: string[], timeoutMs = 20000): Promise<{ stdout: string; stderr: string; code: number }> {
656
- return new Promise((res) => {
657
- const child = spawn(command, args, { stdio: ['ignore', 'pipe', 'pipe'], windowsHide: true })
658
- let stdout = '', stderr = ''
659
- const timer = setTimeout(() => child.kill('SIGKILL'), timeoutMs)
660
- child.stdout.on('data', d => stdout += String(d))
661
- child.stderr.on('data', d => stderr += String(d))
662
- child.on('close', code => { clearTimeout(timer); res({ stdout, stderr, code: code ?? 1 }) })
663
- child.on('error', err => { clearTimeout(timer); res({ stdout, stderr: String(err), code: 1 }) })
664
- })
665
- }
666
-
667
- // Cross-platform "reveal in file manager / open URL in default app".
668
- // macOS `open`, Linux `xdg-open`, Windows `start` (a cmd builtin, so via `cmd /c`;
669
- // the empty "" is start's title arg so a quoted path/URL isn't swallowed as title).
670
- function openPath(target: string, timeoutMs = 5000): Promise<{ stdout: string; stderr: string; code: number }> {
671
- if (IS_WINDOWS) return runCommand('cmd', ['/c', 'start', '', target], timeoutMs)
672
- if (IS_MAC) return runCommand('open', [target], timeoutMs)
673
- return runCommand('xdg-open', [target], timeoutMs)
674
- }
675
-
676
- // Cross-platform replacement for `tail -n N [-F] files`. Windows ships no `tail`,
677
- // so even a non-follow `vn log` would break; a pure-JS implementation also drops a
678
- // process dependency on mac/Linux. Follow mode polls appended bytes every second.
679
- async function tailFiles(files: string[], lines: number, follow: boolean): Promise<void> {
680
- const header = files.length > 1
681
- const lastLines = (text: string, n: number) => {
682
- const arr = text.split('\n')
683
- if (arr.length && arr[arr.length - 1] === '') arr.pop()
684
- return arr.slice(-n).join('\n')
685
- }
686
- const sizes = new Map<string, number>()
687
- for (const f of files) {
688
- const text = await readFile(f, 'utf8').catch(() => '')
689
- if (header) process.stdout.write(`==> ${f} <==\n`)
690
- const tail = lastLines(text, lines)
691
- if (tail) process.stdout.write(tail + '\n')
692
- sizes.set(f, Buffer.byteLength(text))
693
- }
694
- if (!follow) return
695
- await new Promise<void>((resolve) => {
696
- let stop = false
697
- process.once('SIGINT', () => { stop = true; resolve() })
698
- const poll = () => {
699
- if (stop) return
700
- for (const f of files) {
701
- try {
702
- const size = statSync(f).size
703
- const prev = sizes.get(f) ?? 0
704
- if (size > prev) {
705
- const fd = openSync(f, 'r')
706
- try {
707
- const buf = Buffer.alloc(size - prev)
708
- readSync(fd, buf, 0, buf.length, prev)
709
- if (header) process.stdout.write(`==> ${f} <==\n`)
710
- process.stdout.write(buf.toString('utf8'))
711
- } finally { closeSync(fd) }
712
- sizes.set(f, size)
713
- } else if (size < prev) {
714
- sizes.set(f, size) // rotated/truncated
715
- }
716
- } catch (e) { warnSideEffect(`follow ${f}`, e) }
717
- }
718
- if (!stop) setTimeout(poll, 1000)
719
- }
720
- setTimeout(poll, 1000)
721
- })
722
- }
723
-
724
- async function ffprobeDuration(config: Config, path: string): Promise<number | null> {
725
- const result = await runCommand(config.ffprobeBin, ['-v', 'error', '-show_entries', 'format=duration', '-of', 'default=noprint_wrappers=1:nokey=1', path])
726
- if (result.code !== 0) return null
727
- const v = Number(result.stdout.trim())
728
- return Number.isFinite(v) ? v : null
729
- }
730
-
731
- function isCandidateFile(path: string): boolean {
732
- const name = basename(path)
733
- if (name.startsWith('._') || name.startsWith('.')) return false
734
- if (!AUDIO_EXTENSIONS.has(extname(path).toLowerCase())) return false
735
- const parts = path.split(/[/\\]/)
736
- if (parts.includes('.Spotlight-V100') || parts.includes('.fseventsd') || parts.includes('System Volume Information')) return false
737
- return true
738
- }
739
-
740
- /**
741
- * `complete` is false when any part of the listing was lost — the glob threw, or
742
- * a file we had just seen could not be read. It gates pruning: "not in the scan"
743
- * only means "gone from the recorder" if the scan actually saw everything, and
744
- * treating a half-read device as authoritative would delete live queue entries
745
- * along with their retry counters.
746
- */
747
- async function toRecording(config: Config, file: string): Promise<Recording> {
748
- const st = await stat(file)
749
- return {
750
- sourcePath: file,
751
- sizeBytes: st.size,
752
- modifiedAt: st.mtime.toISOString(),
753
- durationSeconds: await ffprobeDuration(config, file),
754
- sourceId: await sourceIdFor(file),
755
- recordedAt: parseRecordedAt(file),
756
- }
757
- }
758
-
759
- async function scanRecordings(config: Config): Promise<{ recordings: Recording[]; complete: boolean }> {
760
- if (!existsSync(config.recordDir)) return { recordings: [], complete: false }
761
- const recordings: Recording[] = []
762
- let complete = true
763
- try {
764
- for await (const file of new Bun.Glob('**/*').scan({ cwd: config.recordDir, absolute: true, dot: true })) {
765
- if (!isCandidateFile(file)) continue
766
- const st = await stat(file).catch(() => null)
767
- // Listed a moment ago but unreadable now: the device is going away, or
768
- // this file is. Either way the listing is no longer trustworthy.
769
- if (!st) { complete = false; continue }
770
- if (!st.isFile()) continue
771
- try {
772
- recordings.push(await toRecording(config, file))
773
- } catch (e) { complete = false; warnSideEffect(`read ${basename(file)} during scan`, e) }
774
- }
775
- } catch (e) {
776
- complete = false
777
- warnSideEffect('scan recorder', e)
778
- }
779
- // Oldest first: backlog is drained in chronological order, so every file is
780
- // guaranteed a turn before newer arrivals jump the queue.
781
- recordings.sort((a, b) => a.recordedAt.getTime() - b.recordedAt.getTime())
782
- return { recordings, complete }
783
- }
784
-
785
- // ────────────────────────────────────────────────────────────────────────────
786
- // File path planning
787
- // ────────────────────────────────────────────────────────────────────────────
788
-
789
- /**
790
- * The one place the output layout is written down. A job starts out untitled
791
- * (timestamp only) and moves to its titled names once the summary produces a
792
- * title; pass `title` — including a null/empty one — for the titled form.
793
- */
794
- function layout(config: Config, rec: Recording, title?: string | null): LocalFiles {
795
- const { month, prefix } = dateParts(rec.recordedAt)
796
- const untitled = title === undefined
797
- const base = untitled ? prefix : `${prefix}-${safeSlug(title || 'note')}`
798
- return {
799
- audio: join(config.workspace, '_audio', month, `${base}-original${extname(rec.sourcePath).toLowerCase()}`),
800
- transcript: join(config.workspace, '_transcripts', month, `${base}-transcript.md`),
801
- notes: join(config.workspace, month, untitled ? `${base}-note.md` : `${base}.md`),
802
- metadata: join(config.workspace, '_metadata', month, `${base}-metadata.json`),
803
- }
804
- }
805
-
806
- // Resume on the evidence, not on a state label: if the transcript is on disk,
807
- // re-running ASR is money spent for nothing. Keying this off `notes_failed`
808
- // instead meant `vn forget` (which drops the record) silently re-paid for ASR,
809
- // even though the transcript was still sitting there.
810
- function resumableTranscriptFiles(config: Config, rec: Recording, store: StateFile, mode: RunMode, force: boolean): LocalFiles | null {
811
- if (force || mode !== 'notes') return null
812
- // Paths recorded by an earlier attempt win: that attempt may already have
813
- // moved its outputs to titled names.
814
- const fallback = layout(config, rec)
815
- const recorded = store.jobs[rec.sourceId]?.paths || {}
816
- const files = Object.fromEntries(
817
- Object.entries(fallback).map(([key, path]) => [key, typeof recorded[key] === 'string' ? recorded[key] : path]),
818
- ) as LocalFiles
819
- return existsSync(files.transcript) ? files : null
820
- }
821
-
822
- async function readSavedTranscript(path: string): Promise<string> {
823
- const markdown = await readFile(path, 'utf8')
824
- const marker = [RAW_TRANSCRIPT_MARKER, RAW_TRANSCRIPT_MARKER_LEGACY].find(m => markdown.includes(m))
825
- if (!marker) throw new Error(`Cannot resume summary: saved transcript is missing raw transcript marker: ${path}`)
826
- const transcript = markdown.slice(markdown.indexOf(marker) + marker.length).trim()
827
- if (!transcript) throw new Error(`Cannot resume summary: saved transcript is empty: ${path}`)
828
- return transcript
829
- }
830
-
831
- async function removeFailedSummaryStub(path: string): Promise<void> {
832
- if (!existsSync(path)) return
833
- try {
834
- const body = await readFile(path, 'utf8')
835
- if (body.startsWith('# Pending summary: ') || body.startsWith('# 待补纪要:')) await unlink(path)
836
- } catch (e) { warnSideEffect(`remove failed-summary stub ${path}`, e) }
837
- }
838
-
839
- /**
840
- * Move a job's existing outputs onto their titled paths. Audio and the
841
- * transcript written before the summary ran move together — they used to be
842
- * renamed in two different places, and the one left behind became an orphan.
843
- * Notes and metadata are rewritten by the caller, so their stale copies from a
844
- * failed attempt are dropped instead of moved.
845
- */
846
- async function promoteOutputs(from: LocalFiles, to: LocalFiles): Promise<void> {
847
- for (const key of ['audio', 'transcript'] as const) {
848
- if (from[key] === to[key] || !existsSync(from[key])) continue
849
- await mkdir(dirname(to[key]), { recursive: true })
850
- if (existsSync(to[key])) await unlink(to[key])
851
- await rename(from[key], to[key])
852
- }
853
- if (from.metadata !== to.metadata && existsSync(from.metadata)) {
854
- await unlink(from.metadata).catch(e => warnSideEffect(`remove orphaned metadata ${from.metadata}`, e))
855
- }
856
- }
857
-
858
- // ───────────────────────────────────────────────────────────────────────
859
- // Volcano (Doubao ASR + TOS upload)
860
- // ───────────────────────────────────────────────────────────────────────
861
-
862
- function volcanoContentTypeFromExt(ext: string): string {
863
- const e = ext.replace(/^\./, '').toLowerCase()
864
- switch (e) {
865
- case 'mp3': return 'audio/mpeg'
866
- case 'wav': return 'audio/wav'
867
- case 'm4a': return 'audio/mp4'
868
- case 'aac': return 'audio/aac'
869
- case 'ogg': return 'audio/ogg'
870
- case 'flac': return 'audio/flac'
871
- default: return 'application/octet-stream'
872
- }
873
- }
874
-
875
- async function volcanoSubmitTask(volc: VolcanoConfig, taskId: string, audioUrl: string, format: string): Promise<void> {
876
- const body = {
877
- user: { uid: 'voicenote' },
878
- audio: { url: audioUrl, format },
879
- request: {
880
- model_name: 'bigmodel',
881
- enable_itn: true,
882
- enable_punc: true,
883
- enable_ddc: true,
884
- enable_speaker_info: true,
885
- show_utterances: true,
886
- ...(volc.language ? { language: volc.language } : {}),
887
- },
888
- }
889
- const res = await fetch('https://openspeech.bytedance.com/api/v3/auc/bigmodel/submit', {
890
- method: 'POST',
891
- headers: volcanoAuthHeaders(volc, taskId, true),
892
- body: JSON.stringify(body),
893
- })
894
- const status = res.headers.get('X-Api-Status-Code') || ''
895
- const message = res.headers.get('X-Api-Message') || ''
896
- if (status !== '20000000') {
897
- const text = await res.text().catch(() => '')
898
- throw new Error(`Volcano submit failed: status=${status} message=${message} body=${text.slice(0, 500)}`)
899
- }
900
- }
901
-
902
- type VolcanoUtterance = {
903
- text?: string
904
- start_time?: number
905
- end_time?: number
906
- speaker_id?: number | string
907
- additions?: { speaker_id?: number | string; speaker?: string | number }
908
- }
909
-
910
- type VolcanoQueryResult = {
911
- status: string
912
- message: string
913
- result?: { text?: string; utterances?: VolcanoUtterance[] }
914
- audio_info?: { duration?: number }
915
- }
916
-
917
- async function volcanoQueryResult(volc: VolcanoConfig, taskId: string): Promise<VolcanoQueryResult> {
918
- const res = await fetch('https://openspeech.bytedance.com/api/v3/auc/bigmodel/query', {
919
- method: 'POST',
920
- headers: volcanoAuthHeaders(volc, taskId, false),
921
- body: '{}',
922
- })
923
- const status = res.headers.get('X-Api-Status-Code') || ''
924
- const message = res.headers.get('X-Api-Message') || ''
925
- const text = await res.text().catch(() => '')
926
- let parsed: any = null
927
- if (text) { try { parsed = JSON.parse(text) } catch { parsed = null } }
928
- return { status, message, result: parsed?.result, audio_info: parsed?.audio_info }
929
- }
930
-
931
- function volcanoSpeakerLabel(u: VolcanoUtterance): string {
932
- const id = u.speaker_id ?? u.additions?.speaker_id ?? u.additions?.speaker
933
- if (id == null || id === '') return 'Speaker A'
934
- const n = Number(id)
935
- if (Number.isFinite(n) && n >= 0 && n < 26) return `Speaker ${String.fromCharCode(65 + n)}`
936
- return `Speaker ${String(id)}`
937
- }
938
-
939
- function volcanoFormatTranscript(result: { text?: string; utterances?: VolcanoUtterance[] }): string {
940
- const utterances = result.utterances || []
941
- if (!utterances.length) return (result.text || '').trim()
942
- const lines = utterances
943
- .map(u => {
944
- const text = String(u.text || '').trim()
945
- if (!text) return ''
946
- const start = formatSeconds(Math.round((u.start_time || 0) / 1000))
947
- const end = formatSeconds(Math.round((u.end_time || 0) / 1000))
948
- return `[${start}-${end}] ${volcanoSpeakerLabel(u)}: ${text}`
949
- })
950
- .filter(Boolean)
951
- return lines.join('\n')
952
- }
953
-
954
- async function volcanoTranscribeAudio(volc: VolcanoConfig, audioPath: string, rec: Recording): Promise<string> {
955
- const ext = extname(audioPath).toLowerCase() || '.mp3'
956
- const format = ext.replace(/^\./, '')
957
- const contentType = volcanoContentTypeFromExt(ext)
958
- const { month } = dateParts(rec.recordedAt)
959
- const key = `voicenote/${month}/${rec.sourceId}-${Date.now()}${ext}`
960
- const object = tosObject(volc.tos, key)
961
- console.log(`Volcano: upload audio to TOS as ${key}`)
962
- await withHeartbeat('upload audio to TOS', () => object.write(Bun.file(audioPath), { type: contentType }), 30)
963
- let cleanedUp = false
964
- const cleanup = async () => {
965
- if (cleanedUp || volc.tos.keep) return
966
- cleanedUp = true
967
- await object.delete().catch(e => warnSideEffect(`delete TOS object ${key}`, e))
968
- }
969
- try {
970
- const audioUrl = object.presign({ method: 'GET', expiresIn: 6 * 3600 })
971
- const taskId = randomUUID()
972
- console.log(`Volcano: submit ASR task ${taskId} (resource=${volc.resourceId}, format=${format})`)
973
- await volcanoSubmitTask(volc, taskId, audioUrl, format)
974
- const started = Date.now()
975
- const expectedSeconds = rec.durationSeconds || 0
976
- const maxWaitMs = Math.max(20 * 60 * 1000, Math.ceil(expectedSeconds * 1000 * 1.5))
977
- let lastStatusLog = 0
978
- let lastStatus = ''
979
- // Tolerate transient failures while polling: by this point the audio is
980
- // uploaded and the ASR task is submitted (money spent) — one dropped
981
- // socket or an HTTP-level error (gateway 5xx returns no X-Api-Status-Code
982
- // header, so q.status comes back empty) must not fail the whole job and
983
- // trigger a full re-upload + re-submit on the next tick. Only give up
984
- // after many failures in a row; throws when the budget or deadline is hit.
985
- let queryFailures = 0
986
- const transientQueryFailure = (desc: string): void => {
987
- queryFailures++
988
- if (queryFailures >= 10) throw new Error(`Volcano query failed ${queryFailures}x in a row: ${desc}`)
989
- if (Date.now() - started > maxWaitMs) throw new Error(`Volcano: timeout after ${formatElapsed(Date.now() - started)} (last error: ${desc})`)
990
- // console.error (not log) so wireDailyLog tags it [ERROR] and `vn errors`
991
- // surfaces it — matching chatCompleteViaPi's transient-retry logging.
992
- // A repeatedly-near-threshold ASR wobble is exactly what ops wants to see.
993
- console.error(`… Volcano: transient query failure (attempt ${queryFailures}/10, will retry): ${desc}`)
994
- }
995
- for (;;) {
996
- await new Promise(res => setTimeout(res, 8000))
997
- let q: VolcanoQueryResult
998
- try {
999
- q = await volcanoQueryResult(volc, taskId)
1000
- } catch (e: any) {
1001
- transientQueryFailure(String(e?.message || e))
1002
- continue
1003
- }
1004
- if (!q.status) {
1005
- transientQueryFailure(`empty status header (HTTP-level error, body: ${q.message || 'none'})`)
1006
- continue
1007
- }
1008
- queryFailures = 0
1009
- if (q.status === '20000000' && q.result) {
1010
- console.log(`✓ Volcano: ASR done in ${formatElapsed(Date.now() - started)}; audio_duration=${q.audio_info?.duration ?? 'unknown'}ms`)
1011
- return volcanoFormatTranscript(q.result)
1012
- }
1013
- if (q.status === '20000001' || q.status === '20000002') {
1014
- if (q.status !== lastStatus || Date.now() - lastStatusLog > 60_000) {
1015
- const label = q.status === '20000002' ? 'queued' : 'processing'
1016
- console.log(`… Volcano: ${label} (status=${q.status}, ${formatElapsed(Date.now() - started)} elapsed)`)
1017
- lastStatusLog = Date.now()
1018
- lastStatus = q.status
1019
- }
1020
- if (Date.now() - started > maxWaitMs) throw new Error(`Volcano: timeout after ${formatElapsed(Date.now() - started)} (last status=${q.status})`)
1021
- continue
1022
- }
1023
- if (q.status === '20000003') throw new Error('Volcano: 20000003 silent audio (no speech detected)')
1024
- throw new Error(`Volcano query failed: status=${q.status} message=${q.message}`)
1025
- }
1026
- } finally {
1027
- await cleanup()
1028
- }
1029
- }
1030
-
1031
- async function transcribeAudio(config: Config, audioPath: string, rec: Recording): Promise<string> {
1032
- if (!config.volcano) throw new Error('Volcano ASR not configured. Set VOLCANO_ASR_KEY / VOLCANO_TOS_* in config.json.')
1033
- return volcanoTranscribeAudio(config.volcano, audioPath, rec)
1034
- }
1035
-
1036
-
1037
- function speakerContextBlock(speakers: SpeakersConfig): string {
1038
- const selfPart = speakers.self.name
1039
- ? `The user: ${speakers.self.name}${speakers.self.aliases.length ? ` (aliases: ${speakers.self.aliases.join(', ')})` : ''}`
1040
- : "The user's name is not configured."
1041
- const knownPart = speakers.known.length
1042
- ? speakers.known.map(k => `- ${k.name}${k.aliases?.length ? ` (aliases: ${k.aliases.join(', ')})` : ''}${k.relationship ? `, ${k.relationship}` : ''}`).join('\n')
1043
- : '(no other known speakers)'
1044
- return `Speaker context (use it to map Speaker A/B/C back to real names, but only when the evidence is solid):\n- ${selfPart}\n- Other known speakers:\n${knownPart}\n\nRules:\n- If the recording has a single speaker and the user's name is configured, treat Speaker A as the user.\n- In multi-speaker conversations, if a speaker is addressed by the user's name/alias, that speaker is the user.\n- In multi-speaker conversations, if a speaker is addressed by a known speaker's name/alias, that speaker is that known person.\n- Otherwise keep Speaker A/B/C as-is; never guess.`
1045
- }
1046
-
1047
-
1048
- function summaryMessages(config: Config, transcript: string, rec: Recording, localAudioPath: string): { role: 'system' | 'user'; content: string }[] {
1049
- const readerName = config.speakers.self.name?.trim() || 'the user'
1050
- const system = `You are ${readerName}'s personal semantic note-taking assistant, not a generic meeting-minutes template generator.
1051
-
1052
- Your goal is not to reproduce a "meeting minutes" format, but to turn a recording into the most efficient understanding material: let ${readerName} quickly grasp what the discussion was really about, why it matters, what ideas/judgments/items it contains, what deserves attention, and what to do next.
1053
-
1054
- Important: do not output only compressed "conclusions". Much of a recording's value lies in how views were raised, challenged, argued, and revised, and how consensus or disagreement formed. Without mechanically copying the transcript, reconstruct the key speakers' views, reasoning, debates, decision evolution, and how consensus emerged.
1055
-
1056
- Core principles:
1057
- 1. Structure is entirely determined by content. Do not apply any fixed template or emit fixed sections for form's sake.
1058
- 2. Prioritize semantic value over paragraph-by-paragraph retelling; but do not flatten the process into conclusions. Important thinking, debate, validation, concession, rebuttal, and consensus-building are themselves semantic value.
1059
- 3. Multi-person conversations must be reconstructed as much as possible: each side's initial concerns/positions, their reasons and examples, who raised challenges or rebuttals, how the discussion pivoted, which views were revised, what consensus formed, and which disagreements remain open.
1060
- 4. Solo thinking must also have its reasoning path reconstructed: how the question was raised, how hypotheses were tested, why some options were ruled out, which experience/analogies supported the judgment, and why the current conclusion formed.
1061
- 5. Freely choose the form: short memo, strategy memo, question tree, decision record, action list, mind-map-style hierarchy, phase review, debate review, study notes, product/technical analysis, etc.; pick whichever fits the content best.
1062
- 6. If the discussion is conceptual/exploratory, focus on helping the reader understand the train of thought, key concepts, reasoning chains, shifts in views, and passages worth revisiting; do not force-extract to-dos.
1063
- 7. If the discussion is execution/project-oriented, then besides conclusions, items, owners, risks, and next steps, also explain how those conclusions were reached: what constraints applied, which options were compared, and why the current path was chosen.
1064
- 8. If the discussion is short, output only the minimal useful content; if long, you may start with a reading guide and then expand. For long content, err on the side of length rather than dropping key reasoning and debates.
1065
- 9. Avoid filler, boilerplate, and formalistic headings. Every heading should carry information.
1066
- 10. If real names appear in the transcript (see Speaker context below), use them directly; keep Speaker A/B/C only when unsure.
1067
- 11. Explicitly flag uncertain or likely mis-transcribed words; do not treat them as facts.
1068
- 12. Default is Integrated notes mode: the input transcript may not have been separately cleaned. Before generating content, internally perform necessary cleanup: fix obvious typos, unify terminology, restore speakers, merge verbal repetition, fix punctuation and sentence breaks; but never invent information not in the source, and never scrub away the genuine thinking process.
1069
-
1070
- Write all output content (title, markdown, structured fields) in the dominant language of the transcript.
1071
-
1072
- Output must be valid JSON, no markdown fences.
1073
-
1074
- ${speakerContextBlock(config.speakers)}`
1075
-
1076
- const user = `Generate a "semantic notes" document from the transcript below.
1077
-
1078
- Processing mode: Integrated notes mode (no separate transcript cleanup pass; perform necessary cleanup, error correction, organization, and speaker restoration while generating the notes)
1079
-
1080
- The reading scenario you serve:
1081
- - When ${readerName} opens these notes later, they should immediately know: what is worth reading in this recording, what the core ideas/items are, how those views were discussed/argued, what needs understanding, which questions remain open, and what to do next.
1082
- - Do not assume this is a "meeting"; it may be thinking aloud, product ideation, a technical discussion, a business judgment, study notes, an idea capture, a phone call, or task execution.
1083
- - Do not follow Feishu/generic meeting-minutes structures. The markdown structure is determined by the content's semantics.
1084
- - For multi-person discussions, the notes should help ${readerName} review the process: who raised what question, who held what view, who challenged what, how it was answered, where the turning points were, and how consensus formed or disagreements remained.
1085
- - If the transcript clearly contains discussion, debate, joint reasoning, option comparison, or evolving views, the markdown body must include a section that carries this "process reconstruction" (title up to you, e.g. "How the discussion unfolded", "How the views evolved", "Debate and consensus"); a bare conclusion list is not acceptable.
1086
-
1087
- Recording info:
1088
- - Source file: ${rec.sourcePath}
1089
- - Local audio: ${localAudioPath}
1090
- - Time inferred from filename: ${rec.recordedAt.toISOString()}
1091
- - File size: ${rec.sizeBytes} bytes
1092
- - Duration: ${rec.durationSeconds} seconds
1093
-
1094
- Output JSON with these fields:
1095
- {
1096
- "title": "A title in the transcript's language that captures the real topic and value; avoid generic 'meeting minutes' phrasing",
1097
- "date": "YYYY-MM-DD",
1098
- "start_time": "HH:mm|null",
1099
- "end_time": "HH:mm|null",
1100
- "participants": ["Only actually identified real names (including the user's); never Speaker A/B"],
1101
- "organizations": ["string"],
1102
- "projects": ["string"],
1103
- "markdown": "Full markdown body. Must start with an # H1 title. Structure is entirely yours based on the semantics; do not include the trailing source details block, the system appends it.",
1104
- "discussion_flow": [{"stage": "discussion stage/topic", "what_happened": "what happened in this stage", "speaker_positions": [{"speaker": "real name or Speaker label", "position": "view/concern/reasoning"}], "turning_point": "key pivot or change of view|null", "outcome": "stage consensus/disagreement/open|null"}],
1105
- "consensus_points": [{"point": "consensus reached", "how_reached": "how this consensus formed through discussion/argument|null"}],
1106
- "disagreements": [{"issue": "point of disagreement", "positions": [{"speaker": "real name or Speaker label", "position": "stance and reasoning"}], "status": "resolved|unresolved|partially_resolved|null"}],
1107
- "action_items": [{"task": "string", "owner": "string|null", "due_date": "YYYY-MM-DD|null", "priority": "high|medium|low|null", "note": "string|null"}],
1108
- "decisions": [{"decision": "string", "reason": "string|null", "owner": "string|null", "date": "YYYY-MM-DD|null", "how_reached": "how this decision was reached|null"}],
1109
- "open_questions": [{"question": "string", "next_step": "string|null"}],
1110
- "key_quotes_or_details": ["string"],
1111
- "transcription_uncertainties": ["string"]
1112
- }
1113
-
1114
- Markdown quality requirements:
1115
- - The first screen must have a high signal-to-noise ratio: the reader should know why this content is worth keeping without reading the full transcript.
1116
- - No empty sections; no placeholder content like "no clear record / unknown / unidentified".
1117
- - Do not force headings like "Summary, To-dos, Smart sections, Key decisions, Quotes"; use them only when semantically warranted.
1118
- - If there are action items, use concrete actionable language; if there are none, do not fabricate any.
1119
- - If there are ideas/judgments, write out the reasoning chain, not just conclusions.
1120
- - If there was discussion, debate, or joint reasoning, preserve the key process: view raised → challenge/addition → response/rebuttal → revision/pivot → consensus/disagreement. Do not compress it into a single "in the end they concluded…".
1121
- - The markdown body should primarily reconstruct the process in natural language; do not just fill discussion_flow/consensus_points/disagreements as metadata and stop — those structured fields only aid your thinking and indexing.
1122
- - For important consensus, explain how it was reached; for important disagreements, state who held what view, why, and whether it was resolved.
1123
- - If a conclusion went through option comparison or trade-offs, write out the compared options, the criteria, and why one was dropped or chosen.
1124
- - For long meetings, review by topic/stage rather than as a running log, but keep each stage's key turning points and representative speakers' views.
1125
- - Clearly flag controversies, risks, and unverified assumptions.
1126
- - Timestamps may be used sparingly when they help revisit key passages; do not build a full timeline for form's sake.
1127
- - If the transcript has uncertain words, surface them in context as reminders; do not treat them as facts.
1128
- - In Integrated notes mode, especially avoid carrying stutters, repetitions, and typos from the raw transcript into the notes; the body should present cleaned, organized content while preserving the genuine reasoning, debates, and evolution of views.
1129
-
1130
- Transcript:
1131
- ${transcript}`
1132
- return [{ role: 'system', content: system }, { role: 'user', content: user }]
1133
- }
1134
-
1135
- // ───────────────────────────────────────────────────────────────────────
1136
- // Summary via pi. Provider and credentials are pi's own configuration. The
1137
- // optional VOICENOTE_PI_MODEL pins a model; otherwise pi's selected model writes
1138
- // the notes. VoiceNote does not implement a provider fallback chain.
1139
- // ───────────────────────────────────────────────────────────────────────
1140
-
1141
- // pi can't be `bun build --compile`'d (it reads data files from disk), so the
1142
- // bundled GUI ships pi as plain JS and runs it under a bundled bun. When
1143
- // `pi.cli` is set, `pi.bin` is the runtime (bun) and the cli.js is prepended to
1144
- // pi's args — `<bun> <cli.js> <args>`, no wrapper script and no shell (critical
1145
- // on Windows, where pi args include a huge --system-prompt that a .cmd/%*
1146
- // wrapper would mangle). CLI users with a real `pi` on PATH leave it unset.
1147
- function piInvocation(pi: PiConfig, args: string[]): { bin: string; args: string[] } {
1148
- return pi.cli ? { bin: pi.bin, args: [pi.cli, ...args] } : { bin: pi.bin, args }
1149
- }
1150
-
1151
- // ───────────────────────────────────────────────────────────────────────
1152
- // ChatGPT (OpenAI Codex) OAuth login. The browser callback is the default;
1153
- // --device-code is available for accounts that opted into that flow. This
1154
- // exposes pi's login as a plain command for non-TUI and GUI users.
1155
- // We reuse pi's own OAuth implementation (@earendil-works/pi-ai) and persist
1156
- // to pi's auth.json in the exact shape it reads: { type: 'oauth', ...creds }.
1157
- // ───────────────────────────────────────────────────────────────────────
1158
-
1159
- async function persistPiOAuth(authPath: string, providerId: string, creds: Record<string, unknown>): Promise<void> {
1160
- await mkdir(dirname(authPath), { recursive: true })
1161
- let existing: Json = {}
1162
- if (existsSync(authPath)) {
1163
- try { existing = JSON.parse(await readFile(authPath, 'utf8')) as Json } catch (e) { warnSideEffect(`parse ${authPath}`, e) }
1164
- }
1165
- existing[providerId] = { type: 'oauth', ...creds }
1166
- const tmp = `${authPath}.tmp-${process.pid}`
1167
- await writeFile(tmp, JSON.stringify(existing, null, 2) + '\n', { mode: 0o600 })
1168
- await rename(tmp, authPath)
1169
- }
1170
-
1171
- async function loginChatGPT(opts: { json?: boolean; deviceCode?: boolean; emit?: (o: Record<string, unknown>) => void }): Promise<void> {
1172
- // OpenAI's OAuth endpoint is geo-blocked in some regions; getConfig() resolves
1173
- // the proxy into this process's env before any request goes out.
1174
- const authPath = getConfig().pi.authPath
1175
- const json = !!opts.json
1176
- const emit = opts.emit ?? ((o: Record<string, unknown>) => { if (json) console.log(JSON.stringify(o)) })
1177
- try {
1178
- const oauth = await import('@earendil-works/pi-ai/oauth')
1179
- let creds: Record<string, unknown>
1180
- if (opts.deviceCode) {
1181
- // Device-code flow: no localhost server, but the account must first enable
1182
- // "device code authorization for Codex" in ChatGPT > Settings > Security.
1183
- creds = await oauth.loginOpenAICodexDeviceCode({
1184
- onDeviceCode: (info) => {
1185
- if (json) emit({ event: 'device_code', userCode: info.userCode, verificationUri: info.verificationUri, intervalSeconds: info.intervalSeconds, expiresInSeconds: info.expiresInSeconds })
1186
- else {
1187
- console.log('\nTo sign in to ChatGPT (device code):')
1188
- console.log(` 1. Open ${info.verificationUri}`)
1189
- console.log(` 2. Enter code: ${info.userCode}`)
1190
- console.log('\nIf you see "Enable device code authorization", turn it on in')
1191
- console.log('ChatGPT > Settings > Security — or just rerun `vn login` (browser flow).')
1192
- console.log('\nWaiting for authorization…')
1193
- }
1194
- },
1195
- }) as Record<string, unknown>
1196
- } else {
1197
- // Default: browser-callback flow (same as pi `/login` and the official Codex
1198
- // CLI). Spins up localhost:1455/auth/callback; no account setting required.
1199
- creds = await oauth.loginOpenAICodex({
1200
- onAuth: ({ url }) => {
1201
- if (json) emit({ event: 'auth_url', url })
1202
- else {
1203
- console.log('\nOpening your browser to sign in to ChatGPT…')
1204
- console.log(`If it doesn't open, paste this into a browser on THIS machine:\n ${url}`)
1205
- }
1206
- // Best-effort auto-open; the URL is printed/emitted above as fallback.
1207
- void openPath(url)
1208
- },
1209
- onPrompt: async ({ message }) => {
1210
- // Only reached if the localhost:1455 callback can't complete (port busy,
1211
- // or browser on another machine). Fail loudly rather than hang.
1212
- throw new Error(`${message} — automatic callback failed (is localhost:1455 free, and is your browser on this machine?). Retry, or use --device-code.`)
1213
- },
1214
- }) as Record<string, unknown>
1215
- }
1216
- await persistPiOAuth(authPath, oauth.openaiCodexOAuthProvider.id, creds)
1217
- if (json) emit({ event: 'success', provider: oauth.openaiCodexOAuthProvider.id })
1218
- else console.log(`\n✓ Signed in. Credentials saved to ${authPath}. Verify with: vn doctor`)
1219
- } catch (e: any) {
1220
- let message = String(e?.message || e)
1221
- if (/unsupported_country_region_territory|\b403\b/.test(message)) {
1222
- message += ' — OpenAI blocks this region without a proxy. Set LOCAL_PROXY_HOST/LOCAL_PROXY_PORT (or http_proxy) and retry; Volcano stays direct.'
1223
- }
1224
- if (json) emit({ event: 'error', message })
1225
- else console.error(`\nLogin failed: ${message}`)
1226
- process.exitCode = 1
1227
- }
1228
- }
1229
-
1230
- // ───────────────────────────────────────────────────────────────────────
1231
- // File-based config (~/.config/voicenote/config.json) — written by the GUI
1232
- // via `vn config set`, read by loadEnvConfig(). ENV config uses ENV_KEYS;
1233
- // identity lives under the same file's `speakers` object.
1234
- // ───────────────────────────────────────────────────────────────────────
1235
-
1236
- function readStdin(): Promise<string> {
1237
- return new Promise((resolve) => {
1238
- let data = ''
1239
- process.stdin.setEncoding('utf8')
1240
- process.stdin.on('data', d => { data += d })
1241
- process.stdin.on('end', () => resolve(data))
1242
- process.stdin.on('error', () => resolve(data))
1243
- })
1244
- }
1245
-
1246
- function configFileEnv(raw = loadConfigJson()): Record<string, string> {
1247
- const env: Record<string, string> = {}
1248
- for (const k of ENV_KEYS) if (typeof raw[k] === 'string') env[k] = raw[k] as string
1249
- return env
1250
- }
1251
-
1252
- function configGetData(): { path: string; env: Record<string, string>; self: { name: string | null; aliases: string[] } } {
1253
- const current = loadConfigJson()
1254
- const speakers = normalizeSpeakers(current.speakers ?? DEFAULT_SPEAKERS)
1255
- return {
1256
- path: CONFIG_ENV_PATH,
1257
- env: configFileEnv(current),
1258
- self: { name: speakers.self.name, aliases: speakers.self.aliases },
1259
- }
1260
- }
1261
-
1262
- function configGet(): void { console.log(JSON.stringify(configGetData(), null, 2)) }
1263
-
1264
- type ConfigSetPayload = { env?: Record<string, unknown>; self?: { name?: string | null; aliases?: string[] } }
1265
-
1266
- async function writeConfigJson(value: Record<string, unknown>): Promise<void> {
1267
- await mkdir(CONFIG_DIR, { recursive: true })
1268
- const tmp = `${CONFIG_ENV_PATH}.tmp-${process.pid}`
1269
- await writeFile(tmp, JSON.stringify(value, null, 2) + '\n', { mode: 0o600 })
1270
- await rename(tmp, CONFIG_ENV_PATH)
1271
- }
1272
-
1273
- async function configSetData(payload: ConfigSetPayload): Promise<{ ok: true; path: string; ignoredKeys?: string[] }> {
1274
- if (!payload || typeof payload !== 'object' || Array.isArray(payload)) throw new Error('Config payload must be a JSON object')
1275
- const current = loadConfigJson()
1276
- const known = ENV_KEYS as readonly string[]
1277
- const ignored: string[] = []
1278
- if (payload.env) {
1279
- for (const [key, value] of Object.entries(payload.env)) {
1280
- if (!known.includes(key)) { ignored.push(key); continue }
1281
- if (value === null) delete current[key]
1282
- else if (typeof value === 'string') current[key] = value
1283
- else throw new Error(`Config value ${key} must be a string or null`)
1284
- }
1285
- }
1286
- if (payload.self) {
1287
- const speakers = normalizeSpeakers(current.speakers ?? DEFAULT_SPEAKERS)
1288
- if (payload.self.name !== undefined) {
1289
- if (payload.self.name !== null && typeof payload.self.name !== 'string') throw new Error('self.name must be a string or null')
1290
- speakers.self.name = payload.self.name
1291
- }
1292
- if (payload.self.aliases !== undefined) {
1293
- if (!Array.isArray(payload.self.aliases) || payload.self.aliases.some(alias => typeof alias !== 'string')) throw new Error('self.aliases must contain only strings')
1294
- speakers.self.aliases = payload.self.aliases
1295
- }
1296
- current.speakers = speakers
1297
- }
1298
- await writeConfigJson(current)
1299
- return { ok: true, path: CONFIG_ENV_PATH, ...(ignored.length ? { ignoredKeys: ignored } : {}) }
1300
- }
1301
-
1302
- async function configSet(): Promise<void> {
1303
- let payload: ConfigSetPayload
1304
- try { payload = JSON.parse(await readStdin()) }
1305
- catch (e: any) { console.error(`Invalid JSON on stdin: ${e?.message || e}`); process.exitCode = 1; return }
1306
- // Every other key is re-read by the agent on each run, but VOICENOTE_PI_BIN
1307
- // is snapshotted into the scheduler as a resolved absolute path at install
1308
- // time (launchd's fixed PATH can't find it otherwise). The GUI reinstalls on
1309
- // save; the CLI path must be told — but only when the value actually CHANGES.
1310
- // A GUI-style client resubmits every field on every save, so `in payload`
1311
- // alone would nag on every unrelated edit.
1312
- const PI_BIN = 'VOICENOTE_PI_BIN'
1313
- const before = String(loadConfigJson()[PI_BIN] ?? '')
1314
- console.log(JSON.stringify(await configSetData(payload)))
1315
- const piBinChanged = payload.env && PI_BIN in payload.env && String(payload.env[PI_BIN] ?? '') !== before
1316
- if (piBinChanged) {
1317
- console.error(`Note: ${PI_BIN} changed — re-run \`vn install-launch-agent\` to apply it to the background scheduler.`)
1318
- }
1319
- }
1320
-
1321
- function extractFirstJsonObject(text: string): string {
1322
- const raw = text.trim()
1323
- // Models often wrap JSON in a ```json fence; strip it before looking inside.
1324
- const trimmed = raw.match(/^```(?:json)?\s*([\s\S]*?)\s*```\s*$/i)?.[1]?.trim() ?? raw
1325
- if (trimmed.startsWith('{') && trimmed.endsWith('}')) return trimmed
1326
- // Find the first balanced {...}
1327
- let depth = 0, start = -1, inString = false, escape = false
1328
- for (let i = 0; i < trimmed.length; i++) {
1329
- const ch = trimmed[i]!
1330
- if (escape) { escape = false; continue }
1331
- if (inString) {
1332
- if (ch === '\\') { escape = true; continue }
1333
- if (ch === '"') inString = false
1334
- continue
1335
- }
1336
- if (ch === '"') { inString = true; continue }
1337
- if (ch === '{') { if (depth === 0) start = i; depth++ }
1338
- else if (ch === '}') { depth--; if (depth === 0 && start !== -1) return trimmed.slice(start, i + 1) }
1339
- }
1340
- return trimmed
1341
- }
1342
-
1343
- type PiRunOptions = {
1344
- systemPrompt: string
1345
- userPrompt: string
1346
- timeoutMs?: number
1347
- thinking?: string
1348
- tools?: string // e.g. 'read,grep'; empty/undefined = --no-tools
1349
- appendSystemPrompt?: string
1350
- cwd?: string // agent working dir: the knowledge base, so read/grep/find default there
1351
- }
1352
-
1353
- async function runPi(config: Config, opts: PiRunOptions): Promise<string> {
1354
- const args = [
1355
- '-p',
1356
- '--mode', 'text',
1357
- '--no-extensions', '--no-skills', '--no-context-files', '--no-session', '--no-prompt-templates', '--no-themes',
1358
- '--system-prompt', opts.systemPrompt,
1359
- ]
1360
- // Unset means pi's own default model and provider. There is no second
1361
- // provider to fall back to either way.
1362
- if (config.pi.model) args.push('--model', config.pi.model)
1363
- if (opts.thinking) args.push('--thinking', opts.thinking)
1364
- if (opts.tools && opts.tools.trim()) args.push('--tools', opts.tools.trim())
1365
- else args.push('--no-tools')
1366
- if (opts.appendSystemPrompt) args.push('--append-system-prompt', opts.appendSystemPrompt)
1367
- return new Promise<string>((resolve, reject) => {
1368
- const inv = piInvocation(config.pi, args)
1369
- const child = spawn(inv.bin, inv.args, { stdio: ['pipe', 'pipe', 'pipe'], cwd: opts.cwd, windowsHide: true, env: { ...process.env, ...config.childEnv } })
1370
- let stdout = '', stderr = ''
1371
- const timer = opts.timeoutMs ? setTimeout(() => child.kill('SIGKILL'), opts.timeoutMs) : null
1372
- child.stdout.on('data', d => stdout += String(d))
1373
- child.stderr.on('data', d => stderr += String(d))
1374
- child.on('error', err => { if (timer) clearTimeout(timer); reject(err) })
1375
- child.on('close', code => {
1376
- if (timer) clearTimeout(timer)
1377
- if (code !== 0) return reject(new Error(`pi exited ${code}: ${(stderr || stdout).slice(0, 800)}`))
1378
- const text = stdout.trim()
1379
- if (!text) return reject(new Error('pi returned empty output'))
1380
- resolve(text)
1381
- })
1382
- // A pi that dies before draining stdin (bad flags, crash on startup) closes the
1383
- // pipe mid-write. Without this handler the EPIPE is an unhandled 'error' event
1384
- // that kills the whole run, hiding pi's actual error; 'close' below reports it.
1385
- child.stdin.on('error', (e: NodeJS.ErrnoException) => {
1386
- if (e.code !== 'EPIPE') warnSideEffect('write prompt to pi stdin', e)
1387
- })
1388
- child.stdin.end(opts.userPrompt)
1389
- })
1390
- }
1391
-
1392
- // A transient pi failure (proxy reset, dropped socket, upstream 5xx/429) is
1393
- // retried: a momentary blip must not cost a run its notes. Quota/auth/4xx are NOT
1394
- // transient — retrying them only wastes time, so they fail the summary at once.
1395
- function isTransientPiError(e: any): boolean {
1396
- const msg = String(e?.message || e).toLowerCase()
1397
- if (/quota|unauthorized|invalid.*(key|token|credential)|forbidden|\b40[0-4]\b/.test(msg)) return false
1398
- return /socket connection was closed|socket hang up|econnreset|etimedout|esockettimedout|enetunreach|econnrefused|eai_again|fetch failed|network error|timed ?out|temporarily|overloaded|\b(429|500|502|503|504)\b/.test(msg)
1399
- }
1400
-
1401
- async function chatCompleteViaPi(config: Config, opts: PiRunOptions): Promise<string> {
1402
- const maxAttempts = config.pi.retries
1403
- for (let attempt = 1; ; attempt++) {
1404
- try {
1405
- return await runPi(config, opts)
1406
- } catch (e: any) {
1407
- if (attempt >= maxAttempts || !isTransientPiError(e)) throw e
1408
- const backoffMs = Math.min(30000, 2000 * 2 ** (attempt - 1))
1409
- console.error(`pi transient error (attempt ${attempt}/${maxAttempts}); retrying in ${backoffMs}ms: ${e?.message || e}`)
1410
- await new Promise(res => setTimeout(res, backoffMs))
1411
- }
1412
- }
1413
- }
1414
-
1415
- function piSummaryToolsHint(contextDir: string): string {
1416
- return `Before writing the notes you have two read-only tools: read and grep. Your current working directory (cwd) is \`${contextDir}\` (the configured notes/reference directory); use relative paths for grep/read.\n\nGoal: use existing context to align names, speakers, client/project names, product names, and domain terms in this note; do not maintain or assume a separate glossary.\n\nSuggested flow:\n- First extract the most likely client/project/product keywords from the title, filename, and transcript.\n- If a clear topic matches, prefer grep/read on related index pages, project docs, status records, or the 3-5 most recent related notes in the same directory; use them to identify Speaker B/C/F etc., common aliases, product names, and term spellings.\n- If no clear topic matches, grep the current directory with keywords and read only the few most relevant files.\n- Before output, do one names/terms lint pass: eliminate leftover Speaker A/B/C, obviously misheard names, product-name variants, and outdated names; when context is insufficient, keep the uncertainty — never guess.\n\nConstraints:\n- At most 10 tool calls total; if the transcript alone is sufficient, make none.\n- Read only within \`${contextDir}\`; skip directories that clearly involve personal privacy/credentials/finance (e.g. identity / credentials / finance).\n- Found information is only for consistency and background calibration; never write content absent from this transcript into the notes as new meeting facts.\n- Do not attempt to write files or call bash (those tools are not enabled).`
1417
- }
1418
-
1419
- // Summary runs through pi. The agent's working dir IS the knowledge
1420
- // base, so read/grep/find operate there directly. If a configured context dir is
1421
- // missing, say so loudly and run without tools rather than searching the wrong
1422
- // tree (tools, the cwd hint, and the spawn cwd move together).
1423
- async function chatComplete(opts: { systemPrompt: string; userPrompt: string; config: Config }): Promise<string> {
1424
- const { pi } = opts.config
1425
- const ctx = pi.tools ? pi.contextDir : undefined
1426
- const ctxExists = ctx ? existsSync(ctx) : false
1427
- if (ctx && !ctxExists) console.error(`Warning: context dir ${ctx} does not exist; summary agent runs WITHOUT read/grep cross-reference.`)
1428
- const toolsActive = !!ctx && ctxExists
1429
- return chatCompleteViaPi(opts.config, {
1430
- systemPrompt: opts.systemPrompt,
1431
- userPrompt: opts.userPrompt,
1432
- timeoutMs: 60 * 60 * 1000,
1433
- thinking: pi.thinking,
1434
- tools: toolsActive ? pi.tools : undefined,
1435
- appendSystemPrompt: toolsActive ? piSummaryToolsHint(ctx!) : undefined,
1436
- cwd: toolsActive ? ctx : undefined,
1437
- })
1438
- }
1439
-
1440
- async function summarizeTranscript(config: Config, transcript: string, rec: Recording, localAudioPath: string): Promise<Json> {
1441
- const messages = summaryMessages(config, transcript, rec, localAudioPath)
1442
- const systemPrompt = String(messages[0]!.content)
1443
- const userPrompt = String(messages[1]!.content)
1444
- const text = await chatComplete({ systemPrompt, userPrompt, config })
1445
- const jsonText = extractFirstJsonObject(text)
1446
- try {
1447
- return JSON.parse(jsonText || '{}') as Json
1448
- } catch (e: any) {
1449
- throw new Error(`summary returned non-JSON output (${e?.message || e}). First 400 chars: ${text.slice(0, 400)}`)
1450
- }
1451
- }
1452
-
1453
- // ────────────────────────────────────────────────────────────────────────────
1454
- // Metadata + markdown
1455
- // ────────────────────────────────────────────────────────────────────────────
1456
-
1457
- function isSpeakerLabel(text: string): boolean {
1458
- return /^\s*speaker\s+[a-z]\s*$/i.test(text) || /^\s*说话人\s*[A-ZA-Za-za-z一二三四五六七八九十0-9]+\s*$/.test(text)
1459
- }
1460
-
1461
- function normalizeMetadata(meta: Json, rec: Recording): Json {
1462
- const d = rec.recordedAt
1463
- meta.date ||= `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`
1464
- meta.start_time ||= `${pad(d.getHours())}:${pad(d.getMinutes())}`
1465
- meta.end_time ??= null
1466
- for (const key of ['participants', 'organizations', 'projects', 'discussion_flow', 'consensus_points', 'disagreements', 'action_items', 'decisions', 'open_questions', 'key_quotes_or_details', 'transcription_uncertainties']) {
1467
- if (!Array.isArray(meta[key])) meta[key] = []
1468
- }
1469
- meta.participants = meta.participants.filter((p: any) => typeof p === 'string' && p.trim() && !isSpeakerLabel(p))
1470
- return meta
1471
- }
1472
-
1473
- const SOURCE_MARKER = '<!-- voicenote:source -->'
1474
- function sourceDetails(audioPath: string, transcriptPath: string): string {
1475
- return `${SOURCE_MARKER}\n<details>\n<summary>Source</summary>\n\n- Generated by: voicenote automatic transcription\n- Original audio: \`${audioPath}\`\n- Full transcript: \`${transcriptPath}\`\n\n</details>`
1476
- }
1477
-
1478
- function markdownNotes(meta: Json, audioPath: string, transcriptPath: string): string {
1479
- let body = typeof meta.markdown === 'string' && meta.markdown.trim() ? meta.markdown.trim() : `# ${meta.title || 'Untitled recording notes'}\n`
1480
- if (!body.startsWith('#')) body = `# ${meta.title || 'Untitled recording notes'}\n\n${body}`
1481
- if (!body.includes(SOURCE_MARKER)) body = `${body.trim()}\n\n${sourceDetails(audioPath, transcriptPath)}`
1482
- return `${body.trim()}\n`
1483
- }
1484
-
1485
- async function markdownToPdf(markdownPath: string): Promise<string> {
1486
- const pdfPath = markdownPath.replace(/\.md$/i, '.pdf')
1487
- const tempBase = join(os.tmpdir(), `voicenote-pdf-${Date.now()}-${Math.random().toString(36).slice(2)}`)
1488
- const htmlPath = `${tempBase}.html`
1489
- const cssPath = `${tempBase}.css`
1490
- const css = `
1491
- :root { color-scheme: light; }
1492
- body { font-family: -apple-system, BlinkMacSystemFont, "PingFang SC", "Hiragino Sans GB", "Microsoft YaHei", "Noto Sans CJK SC", sans-serif; line-height: 1.68; color: #1f2328; max-width: 860px; margin: 40px auto; padding: 0 32px; font-size: 15px; }
1493
- h1, h2, h3 { line-height: 1.32; margin-top: 1.8em; color: #111827; }
1494
- h1 { font-size: 28px; border-bottom: 1px solid #e5e7eb; padding-bottom: 12px; }
1495
- h2 { font-size: 22px; border-bottom: 1px solid #eef2f7; padding-bottom: 6px; }
1496
- h3 { font-size: 18px; }
1497
- p, ul, ol, blockquote, table { margin: 0.9em 0; }
1498
- blockquote { border-left: 4px solid #d0d7de; padding-left: 16px; color: #57606a; }
1499
- code { font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, monospace; background: #f6f8fa; padding: 0.15em 0.35em; border-radius: 4px; }
1500
- table { border-collapse: collapse; width: 100%; }
1501
- th, td { border: 1px solid #d0d7de; padding: 8px 10px; vertical-align: top; }
1502
- th { background: #f6f8fa; }
1503
- details { margin-top: 2em; color: #57606a; font-size: 13px; }
1504
- @page { size: A4; margin: 18mm 16mm; }
1505
- @media print { body { margin: 0; padding: 0; max-width: none; } h1, h2, h3 { break-after: avoid; } table, blockquote { break-inside: avoid; } }
1506
- `
1507
- await writeFile(cssPath, css, 'utf8')
1508
- try {
1509
- const title = basename(markdownPath, extname(markdownPath))
1510
- const pandoc = await runCommand('pandoc', [markdownPath, '--from', 'markdown+smart', '--to', 'html5', '--standalone', '--metadata', `title=${title}`, '--css', cssPath, '-o', htmlPath], 120000)
1511
- if (pandoc.code !== 0) throw new Error(`pandoc failed: ${pandoc.stderr || pandoc.stdout}`)
1512
- const chromePath = existsSync('/Applications/Google Chrome.app/Contents/MacOS/Google Chrome') ? '/Applications/Google Chrome.app/Contents/MacOS/Google Chrome' : 'google-chrome'
1513
- const chrome = await runCommand(chromePath, ['--headless', '--disable-gpu', '--no-pdf-header-footer', `--print-to-pdf=${pdfPath}`, pathToFileURL(htmlPath).href], 120000)
1514
- if (chrome.code !== 0 || !existsSync(pdfPath)) throw new Error(`chrome pdf failed: ${chrome.stderr || chrome.stdout}`)
1515
- return pdfPath
1516
- } finally {
1517
- await unlink(htmlPath).catch(() => {})
1518
- await unlink(cssPath).catch(() => {})
1519
- }
1520
- }
1521
-
1522
- function transcriptMarkdown(config: Config, rec: Recording, transcript: string, opts: { mode?: RunMode } = {}): string {
1523
- const transcribeBackend = `Volcano Doubao (resource ${config.volcano?.resourceId || 'volc.seedasr.auc'})`
1524
- return `# Transcript: ${basename(rec.sourcePath)}\n\n- Source file: \`${rec.sourcePath}\`\n- Transcription backend: ${transcribeBackend}\n- Mode: ${opts.mode || 'notes'}\n- Recorded at: ${rec.recordedAt.toISOString()}\n- File size: ${rec.sizeBytes} bytes\n- Duration: ${rec.durationSeconds ?? 'unknown'} seconds\n- Transcribed at: ${nowIso()}\n\n---\n\n${RAW_TRANSCRIPT_MARKER}${transcript.trim()}`
1525
- }
1526
-
1527
- // ────────────────────────────────────────────────────────────────────────────
1528
- // Pipeline
1529
- // ────────────────────────────────────────────────────────────────────────────
1530
-
1531
- async function processRecording(config: Config, rec: Recording, opts: any): Promise<Json> {
1532
- const jobStarted = Date.now()
1533
- let files = (opts.resumeFromTranscriptFiles as LocalFiles | null) || layout(config, rec)
1534
- const mode = normalizeRunMode(opts)
1535
- const needsNotes = mode === 'notes'
1536
- const resumeSummary = needsNotes && Boolean(opts.resumeFromTranscriptFiles)
1537
- const transcribeBackendLabel = `volcano:${config.volcano?.resourceId || 'volc.seedasr.auc'}`
1538
- const llmBackendLabel = needsNotes && !opts.dryRun ? 'pi' : null
1539
- const plan = resumeSummary
1540
- ? 'reuse saved transcript → integrated semantic notes → write metadata/index (no auto move)'
1541
- : `copy audio → transcribe → write transcript${needsNotes ? ' → integrated semantic notes' : ''} → write metadata/index (no auto move)`
1542
-
1543
- console.log(`\n=== voicenote job: ${basename(rec.sourcePath)} ===`)
1544
- console.log(`Source: ${rec.sourcePath}`)
1545
- console.log(`Audio: duration=${rec.durationSeconds == null ? 'unknown' : formatSeconds(rec.durationSeconds)}, size=${formatBytes(rec.sizeBytes)}, mode=${mode}, asr=${transcribeBackendLabel}${llmBackendLabel ? `, llm=${llmBackendLabel}` : ''}`)
1546
- console.log(`Plan: ${plan}`)
1547
- if (opts.dryRun) return { source_path: rec.sourcePath, source_id: rec.sourceId, would_copy_to: files.audio, resume_from_transcript: resumeSummary ? files.transcript : null, size_bytes: rec.sizeBytes, duration_seconds: rec.durationSeconds, mode }
1548
-
1549
- const totalSteps = resumeSummary ? 3 : needsNotes ? 4 : 3
1550
- let stepNo = 0
1551
- const nextStep = () => ++stepNo
1552
-
1553
- let transcript = ''
1554
- let meta: Json = {
1555
- title: basename(rec.sourcePath, extname(rec.sourcePath)),
1556
- markdown: '',
1557
- }
1558
-
1559
- if (resumeSummary) {
1560
- progressStep(nextStep(), totalSteps, 'Reuse saved transcript', files.transcript)
1561
- transcript = await readSavedTranscript(files.transcript)
1562
- console.log(`✓ Reusing transcript: ${files.transcript}`)
1563
- if (!existsSync(files.audio)) {
1564
- await mkdir(dirname(files.audio), { recursive: true })
1565
- await copyFile(rec.sourcePath, files.audio)
1566
- console.log(`✓ Local audio restored: ${files.audio}`)
1567
- }
1568
- } else {
1569
- progressStep(nextStep(), totalSteps, 'Copy audio to workspace', files.audio)
1570
- await mkdir(dirname(files.audio), { recursive: true })
1571
- await copyFile(rec.sourcePath, files.audio)
1572
- console.log(`✓ Local audio ready: ${files.audio}`)
1573
-
1574
- progressStep(nextStep(), totalSteps, 'Transcribe audio', transcribeBackendLabel)
1575
- transcript = await withHeartbeat('transcribe audio', () => transcribeAudio(config, files.audio, rec), 90)
1576
-
1577
- // Persist transcript IMMEDIATELY so an expensive ASR result is never lost
1578
- // if a later step (summary) blows up. We use the initial (untitled) path;
1579
- // if summary succeeds we'll move it to the titled path below.
1580
- await mkdir(dirname(files.transcript), { recursive: true })
1581
- // Atomic: "transcript exists on disk" is what makes a later run skip ASR, so
1582
- // a run killed mid-write must not leave a truncated file behind. The raw
1583
- // marker sits near the top, so a partial write would still pass
1584
- // readSavedTranscript()'s checks and get summarised as if complete.
1585
- await writeFileAtomic(files.transcript, transcriptMarkdown(config, rec, transcript, { mode }))
1586
- console.log(`✓ Transcript saved: ${files.transcript}`)
1587
- }
1588
-
1589
- let summaryError: any = null
1590
- if (needsNotes) {
1591
- progressStep(nextStep(), totalSteps, 'Generate integrated semantic notes', `via pi, model=${config.pi.model || "pi's own default"}`)
1592
- try {
1593
- meta = await withHeartbeat('generate integrated semantic notes', () => summarizeTranscript(config, transcript, rec, files.audio), 60)
1594
- } catch (e: any) {
1595
- summaryError = e
1596
- console.error(`Summary step failed; transcript is preserved. Error: ${e?.message || e}`)
1597
- console.error(`Hint: fix LLM auth/credits, then re-run with: vn run --latest`)
1598
- }
1599
- }
1600
-
1601
- meta = normalizeMetadata(meta, rec)
1602
- meta.processing_mode = mode
1603
- meta.source_audio_path = rec.sourcePath
1604
- meta.source_id = rec.sourceId
1605
- meta.source_size_bytes = rec.sizeBytes
1606
- meta.source_modified_at = rec.modifiedAt
1607
- meta.duration_seconds = rec.durationSeconds
1608
- meta.asr_provider = 'volcano'
1609
- meta.transcribe_model = config.volcano?.resourceId || 'volc.seedasr.auc'
1610
- // pi picks the model, so we cannot name it here. Null when no summary ran —
1611
- // summary_error says why.
1612
- meta.llm_backend = needsNotes && !summaryError ? 'pi' : null
1613
- meta.processed_at = nowIso()
1614
- if (summaryError) meta.summary_error = String(summaryError?.message || summaryError)
1615
-
1616
- progressStep(nextStep(), totalSteps, 'Write outputs and index')
1617
- let failedStubPathToRemove: string | null = null
1618
- if (needsNotes && !summaryError) {
1619
- const titled = layout(config, rec, meta.title)
1620
- await promoteOutputs(files, titled)
1621
- // The stub note of a failed attempt is removed only after the real note is
1622
- // written, so a failure in between still leaves the user a pointer to the
1623
- // saved transcript.
1624
- if (files.notes !== titled.notes) failedStubPathToRemove = files.notes
1625
- files = titled
1626
- }
1627
- await mkdir(dirname(files.notes), { recursive: true })
1628
- await mkdir(dirname(files.metadata), { recursive: true })
1629
-
1630
- if (needsNotes && !summaryError) {
1631
- await writeFile(files.notes, markdownNotes(meta, files.audio, files.transcript), 'utf8')
1632
- console.log(`✓ Notes: ${files.notes}`)
1633
- if (failedStubPathToRemove) await removeFailedSummaryStub(failedStubPathToRemove)
1634
- if (opts.pdf) {
1635
- const pdf = await withHeartbeat('render notes PDF', () => markdownToPdf(files.notes), 30)
1636
- meta.local_paths = { ...files, pdf }
1637
- console.log(`✓ PDF: ${pdf}`)
1638
- }
1639
- } else if (needsNotes && summaryError) {
1640
- // No unconditional "just re-run" promise: after MAX_ATTEMPTS the job is
1641
- // `gave_up` and further runs skip it, so the note has to name both ways out.
1642
- const stubBody = `# Pending summary: ${basename(rec.sourcePath)}\n\n> ⚠ Transcription completed and saved, but the summary stage failed; retry needed.\n\n- Transcript file: \`${files.transcript}\`\n- Original audio: \`${rec.sourcePath}\`\n- Failure reason: ${meta.summary_error}\n- Retry: the next \`vn run\` reuses the saved transcript automatically (no new transcription cost). After ${MAX_ATTEMPTS} failed attempts it stops retrying — run \`vn forget ${basename(rec.sourcePath)}\` to queue it again.\n`
1643
- await writeFile(files.notes, stubBody, 'utf8')
1644
- console.log(`⚠ Stub notes (summary failed): ${files.notes}`)
1645
- } else if (opts.pdf) {
1646
- console.log('PDF skipped: --pdf only applies to --mode notes.')
1647
- }
1648
-
1649
- meta.local_paths = { ...files, ...(meta.local_paths?.pdf ? { pdf: meta.local_paths.pdf } : {}) }
1650
- meta.final_paths = {
1651
- audio: files.audio,
1652
- transcript: files.transcript,
1653
- notes: needsNotes ? files.notes : null,
1654
- metadata: files.metadata,
1655
- ...(meta.local_paths?.pdf ? { pdf: meta.local_paths.pdf } : {}),
1656
- }
1657
-
1658
- if (summaryError) {
1659
- meta.status = SUMMARY_FAILED_STATUS
1660
- } else if (needsNotes) {
1661
- meta.status = 'completed'
1662
- } else {
1663
- meta.status = 'transcript_only'
1664
- }
1665
-
1666
- await writeJson(files.metadata, meta)
1667
- await appendJsonl(await notesIndexPath(config), meta)
1668
- console.log(`✓ Completed: ${meta.title || basename(rec.sourcePath)} (${formatElapsed(Date.now() - jobStarted)} total)`)
1669
- if (needsNotes) console.log(`Final notes: ${files.notes}`)
1670
- else console.log(`Final transcript: ${files.transcript}`)
1671
- return meta
1672
- }
1673
-
1674
- // ────────────────────────────────────────────────────────────────────────────
1675
- // Job state — `vn run` is the only writer; every view is a pure read of this.
1676
- // ────────────────────────────────────────────────────────────────────────────
1677
-
1678
- // Named for what it holds: every recording's job state, not just the processed
1679
- // ones. (Pre-0.18 this was `processed.json` with two reason-keyed buckets.)
1680
- const statePathFor = (config: Config) => join(config.workspace, '_state', 'jobs.json')
1681
-
1682
- const legacyStatePathFor = (config: Config) => join(config.workspace, '_state', 'processed.json')
1683
-
1684
- /**
1685
- * Read-only load. On an un-migrated workspace this converts in memory and does
1686
- * NOT write: `vn jobs` and the GUI's poll both come through here without the run
1687
- * lock, and a write from a view could race a live `vn run`. Persisting the
1688
- * conversion is migrateStateOnDisk()'s job, under the lock.
1689
- */
1690
- // The legacy read is the one irreversible read in the codebase, so it gets the
1691
- // same strictness as the new format — `readJson` swallows a parse failure and
1692
- // returns `{}`, which here would mean "nothing was ever processed" and re-pay
1693
- // for every recording's ASR.
1694
- async function readLegacyState(config: Config): Promise<Json> {
1695
- const path = legacyStatePathFor(config)
1696
- return parseStrictJson(await readFile(path, 'utf8'), path) as Json
1697
- }
1698
-
1699
- async function loadState(config: Config): Promise<StateFile> {
1700
- const path = statePathFor(config)
1701
- if (!existsSync(path) && existsSync(legacyStatePathFor(config))) {
1702
- return migrateLegacyState(await readLegacyState(config), nowIso())
1703
- }
1704
- const store = existsSync(path) ? parseStateFile(await readFile(path, 'utf8'), path) : emptyState()
1705
- lastSavedState = JSON.stringify(store)
1706
- return store
1707
- }
1708
-
1709
- // Inline rather than a repo script: most installs are the GUI's compiled
1710
- // sidecar, which has no checkout to run a script from — and starting from empty
1711
- // is not an option, it would re-transcribe everything and pay for ASR twice.
1712
- // Call only with the run lock held.
1713
- async function migrateStateOnDisk(config: Config): Promise<void> {
1714
- const path = statePathFor(config)
1715
- const legacy = legacyStatePathFor(config)
1716
- if (existsSync(path) || !existsSync(legacy)) return
1717
- const store = migrateLegacyState(await readLegacyState(config), nowIso())
1718
- await writeJson(path, store)
1719
- lastSavedState = JSON.stringify(store)
1720
- await rename(legacy, `${legacy}.v1.bak`).catch(e => warnSideEffect('archive pre-0.18 state', e))
1721
- console.log(`Converted ${basename(legacy)} → ${basename(path)} (${Object.keys(store.jobs).length} records; old file kept as .v1.bak)`)
1722
- }
1723
-
1724
- // Workspaces are often synced folders; skip writes when a run did not change
1725
- // the state.
1726
- let lastSavedState = ''
1727
- async function saveState(config: Config, store: StateFile): Promise<void> {
1728
- const serialized = JSON.stringify(store)
1729
- if (serialized === lastSavedState) return
1730
- await writeJson(statePathFor(config), store)
1731
- lastSavedState = serialized
1732
- }
1733
-
1734
- /** Upsert the scan-time facts; never touches lifecycle fields. */
1735
- function recordFor(store: StateFile, rec: Recording): JobRecord {
1736
- const existing = store.jobs[rec.sourceId]
1737
- const next: JobRecord = existing ?? {
1738
- name: basename(rec.sourcePath), source_path: rec.sourcePath, recorded_at: localIso(rec.recordedAt),
1739
- size_bytes: rec.sizeBytes, duration_seconds: rec.durationSeconds,
1740
- state: 'queued', code: null, detail: null, attempts: 0, updated_at: nowIso(), title: null, paths: null,
1741
- }
1742
- next.source_path = rec.sourcePath
1743
- next.size_bytes = rec.sizeBytes
1744
- next.duration_seconds = rec.durationSeconds
1745
- store.jobs[rec.sourceId] = next
1746
- return next
1747
- }
1748
-
1749
- // The live job, declared by the run itself. Lives next to run.lock (machine
1750
- // state, not workspace data) and carries the pid so a reader can tell a live
1751
- // job from one whose process was killed.
1752
- const CURRENT_PATH = join(STATE_DIR, 'current.json')
1753
-
1754
- function writeCurrent(sourceId: string, step: string, startedAt: string): void {
1755
- try {
1756
- mkdirSync(STATE_DIR, { recursive: true })
1757
- // tmp+rename, same rule as writeFileAtomic: this file's existence and
1758
- // contents are the live-job signal, and progressStep rewrites it at every
1759
- // step. A truncated write would read back as null and show a running job
1760
- // as queued.
1761
- const tmp = `${CURRENT_PATH}.tmp`
1762
- writeFileSync(tmp, JSON.stringify({ pid: process.pid, source_id: sourceId, step, started_at: startedAt } satisfies CurrentJob))
1763
- renameSync(tmp, CURRENT_PATH)
1764
- } catch (e) { warnSideEffect('write current job', e) }
1765
- }
1766
-
1767
- function clearCurrent(): void {
1768
- try { unlinkSync(CURRENT_PATH) } catch (e: any) { if (e?.code !== 'ENOENT') warnSideEffect('clear current job', e) }
1769
- }
1770
-
1771
- // Step reporting from inside the pipeline: a job is only "the current job" for
1772
- // as long as this run says so, so the step is written, never guessed from logs.
1773
- let currentJobId: string | null = null
1774
- let currentJobStartedAt = ''
1775
- function reportStep(step: string): void {
1776
- if (currentJobId) writeCurrent(currentJobId, step, currentJobStartedAt)
1777
- }
1778
-
1779
- function readCurrent(): CurrentJob | null {
1780
- let raw: string
1781
- try { raw = readFileSync(CURRENT_PATH, 'utf8') } catch (e: any) {
1782
- if (e?.code !== 'ENOENT') warnSideEffect('read current job', e)
1783
- return null
1784
- }
1785
- // A damaged file means a live job shows up as queued; treating it as "no job"
1786
- // is the safe read, but it must not be silent.
1787
- try {
1788
- const c = JSON.parse(raw)
1789
- if (Number.isFinite(c?.pid) && typeof c?.source_id === 'string') return c
1790
- warnSideEffect('read current job', new Error(`${CURRENT_PATH} has no pid/source_id`))
1791
- } catch (e) { warnSideEffect('read current job', e) }
1792
- return null
1793
- }
1794
-
1795
- function pidAlive(pid: number): boolean {
1796
- if (!(pid > 0)) return false
1797
- try { process.kill(pid, 0); return true } catch (e: any) { return e?.code === 'EPERM' }
1798
- }
1799
-
1800
- async function runPipeline(file: string | undefined, opts: any): Promise<void> {
1801
- wireDailyLog()
1802
- const config = getConfig()
1803
- opts = { ...opts, file }
1804
- const lock = await acquireRunLock()
1805
- if (!lock) {
1806
- console.log('voicenote pipeline already running; skip')
1807
- return
1808
- }
1809
- try {
1810
- await runPipelineLocked(config, opts)
1811
- } finally {
1812
- await lock.release()
1813
- }
1814
- }
1815
-
1816
- async function runPipelineLocked(config: Config, opts: any): Promise<void> {
1817
- await ensureDirs(config)
1818
- // --dry-run is a zero-side-effect diagnostic; the migration renames the legacy
1819
- // file and permanently drops its `error:*` entries. loadState converts in
1820
- // memory, so a dry run still sees the right picture.
1821
- if (!opts.dryRun) await migrateStateOnDisk(config)
1822
- const store = await loadState(config)
1823
- // We hold the run lock, so nothing else can own a `running` record: any that
1824
- // survive are debris from a killed run. Their attempt was already counted, so
1825
- // this is what makes the retry cap cover crashes as well as thrown errors.
1826
- const interrupted = reconcileInterrupted(store.jobs, nowIso())
1827
- if (interrupted.length) console.log(`Reclaimed ${interrupted.length} job(s) left running by an interrupted run: ${interrupted.slice(0, 3).map(j => j.name).join(', ')}`)
1828
-
1829
- // Explicit file: process exactly that path, wherever it lives. Nothing is
1830
- // scanned, so the listing is never "complete" (no pruning), and the recorder
1831
- // filters (age/size/duration) don't apply — the user named the file.
1832
- const single = opts.file ? resolve(String(opts.file)) : null
1833
- if (single && !statSync(single, { throwIfNoEntry: false })?.isFile()) throw new Error(`Not a file: ${single}`)
1834
- if (!single && !existsSync(config.recordDir)) {
1835
- if (shouldLogIdleStatus(`missing:${config.recordDir}`)) {
1836
- console.log(`Idle: recorder not mounted or record dir missing: ${config.recordDir} (repeated idle logs suppressed for 30m)`)
1837
- }
1838
- return
1839
- }
1840
- const { recordings, complete: scanComplete } = single
1841
- ? { recordings: [await toRecording(config, single)], complete: false }
1842
- : await scanRecordings(config)
1843
- const mode = normalizeRunMode(opts)
1844
- const force = Boolean(opts.force)
1845
- const eligible: Recording[] = []
1846
- const skipCounts: Record<string, number> = {}
1847
- const skipSamples: Record<string, string[]> = {}
1848
- // An explicitly named file that gets skipped must say why, not fall into the
1849
- // idle-suppressed silence meant for the 60s scheduler tick.
1850
- const verboseSkips = Boolean(opts.verbose || opts.dryRun || single)
1851
- const seen = new Set<string>()
1852
- const limits = single
1853
- ? { maxAgeHours: 0, minBytes: 0, minDurationSeconds: 0 }
1854
- : { maxAgeHours: config.maxAgeHours, minBytes: config.minBytes, minDurationSeconds: config.minDurationSeconds }
1855
- for (const rec of recordings) {
1856
- seen.add(rec.sourceId)
1857
- const entry = recordFor(store, rec)
1858
- const verdict = classify(rec, store.jobs[rec.sourceId], limits, { force, notesMode: mode === 'notes', now: Date.now() })
1859
- if (verdict.run) { eligible.push(rec); continue }
1860
- skipCounts[verdict.code] = (skipCounts[verdict.code] || 0) + 1
1861
- ;(skipSamples[verdict.code] ||= []).push(entry.name)
1862
- if (verdict.persist) patchJob(entry, { state: 'filtered', code: verdict.code, detail: verdict.detail }, nowIso())
1863
- if (verboseSkips) console.log(` Skip: ${entry.name} (${verdict.code}${verdict.detail ? `: ${verdict.detail}` : ''})`)
1864
- }
1865
- // Only prune against a listing we believe to be complete: if the recorder went
1866
- // away mid-glob the scan is partial, and pruning would wipe live queue entries
1867
- // (they'd return on the next scan, but their retry counters would not).
1868
- const dropped = pruneUnseen(store.jobs, seen, !single && scanComplete && existsSync(config.recordDir))
1869
- // The only routine path that deletes state — never do it silently.
1870
- if (dropped.length) console.log(`Forgot ${dropped.length} record(s) whose source is no longer on the recorder: ${dropped.slice(0, 3).map(j => j.name).join(', ')}${dropped.length > 3 ? `…(+${dropped.length - 3})` : ''}`)
1871
- const skipSummary = Object.entries(skipCounts).map(([reason, count]) => `${reason}=${count}`).join(', ') || 'none'
1872
- const scanLine = `Scan summary: found=${recordings.length}; eligible=${eligible.length}; skipped=${recordings.length - eligible.length} (${skipSummary})`
1873
- const samplesLine = !verboseSkips && Object.keys(skipSamples).length
1874
- ? `Skipped samples: ${Object.entries(skipSamples).map(([reason, names]) => `${reason}: ${names.slice(0, 3).join(', ')}${names.length > 3 ? `…(+${names.length - 3})` : ''}`).join(' | ')}`
1875
- : ''
1876
- const latestOnly = Boolean(opts.latest)
1877
- const targets = latestOnly ? eligible.slice(-1) : eligible
1878
- // Preflight: if there is work but the run cannot complete, skip BEFORE spending
1879
- // ASR money, rather than failing per-recording on every 60s StartInterval tick.
1880
- // Idle-suppressed so a misconfigured daemon doesn't spam logs. Skipped for
1881
- // --dry-run, which is a zero-side-effect diagnostic and should still print the
1882
- // plan even on an unconfigured machine.
1883
- if (targets.length && !opts.dryRun) {
1884
- const needsAsr = targets.some(rec => !resumableTranscriptFiles(config, rec, store, mode, force))
1885
- if (needsAsr && !config.volcano) {
1886
- if (shouldLogIdleStatus(`asr-misconfig:${config.recordDir}`)) console.error('ASR not configured: Volcano needs VOLCANO_ASR_KEY / VOLCANO_TOS_*. Skipping; run `vn doctor`, fix config, then re-run.')
1887
- return
1888
- }
1889
- }
1890
- if (!targets.length) {
1891
- if (verboseSkips || shouldLogIdleStatus(`idle:${config.recordDir}:${recordings.length}:${skipSummary}:${samplesLine}`)) {
1892
- console.log(scanLine)
1893
- if (samplesLine) console.log(samplesLine)
1894
- console.log('Idle: no new recordings to process. (repeated idle logs suppressed for 30m)')
1895
- }
1896
- } else {
1897
- console.log(scanLine)
1898
- if (samplesLine) console.log(samplesLine)
1899
- console.log(`Queue: processing ${targets.length} recording(s)${latestOnly ? ' (--latest)' : ''}. Remaining after this run: ${Math.max(0, eligible.length - targets.length)}`)
1900
- }
1901
- if (opts.dryRun) {
1902
- // Print the plan and touch nothing: no attempt counted, no state written.
1903
- for (const rec of targets) {
1904
- const plan = await processRecording(config, rec, { ...opts, resumeFromTranscriptFiles: resumableTranscriptFiles(config, rec, store, mode, force) })
1905
- console.log(JSON.stringify(plan, null, 2))
1906
- }
1907
- return
1908
- }
1909
- await saveState(config, store)
1910
-
1911
- for (const rec of targets) {
1912
- const entry = store.jobs[rec.sourceId]!
1913
- // --force means "start over", so it refunds the retry budget too. Without
1914
- // this it only skips one refusal: a spent record would be back at `gave_up`
1915
- // the moment this attempt failed.
1916
- if (force) patchJob(entry, { attempts: 0 }, nowIso())
1917
- currentJobId = rec.sourceId
1918
- currentJobStartedAt = nowIso()
1919
- startAttempt(entry, nowIso())
1920
- await saveState(config, store)
1921
- writeCurrent(rec.sourceId, 'starting', currentJobStartedAt)
1922
- try {
1923
- const resumeFromTranscriptFiles = resumableTranscriptFiles(config, rec, store, mode, force)
1924
- const result = await processRecording(config, rec, { ...opts, resumeFromTranscriptFiles })
1925
- applyOutcome(entry, result.status === SUMMARY_FAILED_STATUS
1926
- ? { kind: 'summary_failed', title: result.title ?? null, paths: result.final_paths ?? null, message: String(result.summary_error ?? 'summary failed; transcript saved') }
1927
- : { kind: 'done', title: result.title ?? null, paths: result.final_paths ?? null }, nowIso())
1928
- } catch (e: any) {
1929
- const message = String(e?.message || e)
1930
- console.error(`ERROR processing ${rec.sourcePath}: ${message}`)
1931
- // Source vanished mid-run (recorder unplugged, file deleted) AND nothing
1932
- // was produced: that's not a failed job, it's a job that no longer exists.
1933
- // Drop it so it can't linger as a permanent "failed" row. A record that
1934
- // already owns output is history — same rule pruneUnseen follows — and
1935
- // deleting it would re-pay for ASR when the recorder comes back.
1936
- if (!existsSync(rec.sourcePath) && !ownsOutput(entry)) {
1937
- console.log(`Forgot ${entry.name}: source left the recorder before it produced anything`)
1938
- delete store.jobs[rec.sourceId]
1939
- } else {
1940
- applyOutcome(entry, { kind: 'failed', message }, nowIso())
1941
- }
1942
- } finally {
1943
- currentJobId = null
1944
- clearCurrent()
1945
- await saveState(config, store) // per job, not per batch: a kill -9 costs one job, not the batch
1946
- }
1947
- // Whole recorder went away — every remaining target would fail the same way
1948
- // and churn ASR-free but noisy retries. Stop and let the next run rescan.
1949
- if (!single && !existsSync(config.recordDir)) {
1950
- console.error(`Recorder disappeared mid-run (${config.recordDir}); stopping. Remaining recordings stay queued.`)
1951
- break
1952
- }
1953
- }
1954
- }
1955
-
1956
- // ────────────────────────────────────────────────────────────────────────────
1957
- // LaunchAgent
1958
- // ────────────────────────────────────────────────────────────────────────────
1959
-
1960
- // Bun standalone executables embed source in a virtual FS, so import.meta.url is
1961
- // NOT a real on-disk path: "/$bunfs/..." on mac/Linux, "B:\~BUN\root\..." on
1962
- // Windows. Either marker means we're the compiled exe (run it directly via
1963
- // process.execPath); otherwise we're bun + cli.ts on disk. NOTE: matching only
1964
- // $bunfs (the old check) misfired on Windows and leaked the virtual path into the
1965
- // scheduled task's arguments.
1966
- function resolveCli(): { cliPath: string; compiled: boolean } {
1967
- const cliPath = fileURLToPath(import.meta.url)
1968
- return { cliPath, compiled: /\$bunfs|~BUN/i.test(cliPath) }
1969
- }
1970
-
1971
- function plistPath(): string {
1972
- return join(os.homedir(), 'Library', 'LaunchAgents', `${LAUNCH_AGENT_LABEL}.plist`)
1973
- }
1974
-
1975
- function xmlEscape(s: string): string {
1976
- return s.replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;').replace(/"/g, '&quot;').replace(/'/g, '&apos;')
1977
- }
1978
-
1979
- // Scheduled runs read all business settings from config.json. The plist only
1980
- // carries a fixed PATH and desktop-bundled runtime paths that do not exist in
1981
- // that file.
1982
- async function launchAgentEnv(config: Config): Promise<Record<string, string>> {
1983
- const env: Record<string, string> = {
1984
- PATH: `${os.homedir()}/.local/bin:/opt/homebrew/bin:/opt/homebrew/sbin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin`,
1985
- }
1986
- // Provenance matters here, so this reads the raw sources rather than Config:
1987
- // only paths the GUI injected into our environment (and that config.json does
1988
- // not already carry) have to be written into the plist.
1989
- const fileEnv = configFileEnv()
1990
- for (const key of ['VOICENOTE_PI_CLI', 'VOICENOTE_FFPROBE_BIN'] as const) {
1991
- if (process.env[key] && process.env[key] !== fileEnv[key]) env[key] = process.env[key]!
1992
- }
1993
- const configuredPi = config.pi.bin
1994
- if (configuredPi.startsWith('/')) env.VOICENOTE_PI_BIN = configuredPi
1995
- else {
1996
- const found = await runCommand(IS_WINDOWS ? 'where' : 'which', [configuredPi], 5000)
1997
- const path = found.code === 0 ? (found.stdout.trim().split(/\r?\n/)[0] || '') : ''
1998
- if (path && existsSync(path)) env.VOICENOTE_PI_BIN = path
1999
- }
2000
- return env
2001
- }
2002
-
2003
- async function installLaunchAgent(opts: { load?: boolean } = {}): Promise<void> {
2004
- const { cliPath, compiled } = resolveCli()
2005
- const programArgs = compiled
2006
- ? [process.execPath, 'run']
2007
- : [existsSync('/opt/homebrew/bin/bun') ? '/opt/homebrew/bin/bun' : process.execPath, cliPath, 'run']
2008
- const programArgsXml = programArgs.map(a => ` <string>${xmlEscape(a)}</string>`).join('\n')
2009
- const plist = plistPath()
2010
- await mkdir(dirname(plist), { recursive: true })
2011
- await mkdir(LOG_DIR, { recursive: true })
2012
- const env = await launchAgentEnv(getConfig())
2013
- const envEntries = Object.entries(env)
2014
- .map(([k, v]) => ` <key>${xmlEscape(k)}</key>\n <string>${xmlEscape(v)}</string>`).join('\n')
2015
- const content = `<?xml version="1.0" encoding="UTF-8"?>
2016
- <!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
2017
- <plist version="1.0">
2018
- <dict>
2019
- <key>Label</key>
2020
- <string>${LAUNCH_AGENT_LABEL}</string>
2021
- <key>ProgramArguments</key>
2022
- <array>
2023
- ${programArgsXml}
2024
- </array>
2025
- <key>RunAtLoad</key>
2026
- <true/>
2027
- <key>StartInterval</key>
2028
- <integer>60</integer>
2029
- <key>StandardOutPath</key>
2030
- <string>${LOG_DIR}/launchd.out.log</string>
2031
- <key>StandardErrorPath</key>
2032
- <string>${LOG_DIR}/launchd.err.log</string>
2033
- <key>WorkingDirectory</key>
2034
- <string>${os.homedir()}</string>
2035
- <key>EnvironmentVariables</key>
2036
- <dict>
2037
- ${envEntries}
2038
- </dict>
2039
- </dict>
2040
- </plist>
2041
- `
2042
- await writeFile(plist, content, 'utf8')
2043
- // Keep scheduler details private and tighten permissions on older plists.
2044
- await chmod(plist, 0o600)
2045
- const summary = Object.keys(env).join(', ')
2046
- console.log(`LaunchAgent written: ${plist}`)
2047
- console.log(`Embedded env keys: ${summary}`)
2048
- if (opts.load) {
2049
- const uid = process.getuid?.()
2050
- // Remove the legacy-label agent so old installs don't double-run vn.
2051
- const legacyPlist = join(os.homedir(), 'Library', 'LaunchAgents', `${LAUNCH_AGENT_LABEL_LEGACY}.plist`)
2052
- if (existsSync(legacyPlist)) {
2053
- await runCommand('launchctl', ['bootout', `gui/${uid}/${LAUNCH_AGENT_LABEL_LEGACY}`], 10000)
2054
- await unlink(legacyPlist).catch(e => warnSideEffect(`remove legacy LaunchAgent ${legacyPlist}`, e))
2055
- }
2056
- await runCommand('launchctl', ['bootout', `gui/${uid}`, plist], 10000) // ignore if not loaded
2057
- const r = await runCommand('launchctl', ['bootstrap', `gui/${uid}`, plist], 10000)
2058
- await runCommand('launchctl', ['enable', `gui/${uid}/${LAUNCH_AGENT_LABEL}`], 10000)
2059
- if (r.code === 0) console.log('LaunchAgent loaded (launchctl bootstrap).')
2060
- else console.error(`bootstrap exit ${r.code}: ${(r.stderr || r.stdout).trim().slice(0, 200)}`)
2061
- } else {
2062
- console.log(`Enable with: launchctl bootstrap gui/$(id -u) ${plist}`)
2063
- }
2064
- }
2065
-
2066
- async function uninstallLaunchAgent(): Promise<void> {
2067
- await runCommand('launchctl', ['bootout', `gui/${process.getuid?.()}`, plistPath()], 10000)
2068
- console.log(`Bootout attempted: ${plistPath()}`)
2069
- }
2070
-
2071
- // ────────────────────────────────────────────────────────────────────────────
2072
- // Windows Task Scheduler (parallel to the mac LaunchAgent above)
2073
- // ────────────────────────────────────────────────────────────────────────────
2074
-
2075
- function taskXmlPath(): string { return join(STATE_DIR, 'task.xml') }
2076
- function taskVbsPath(): string { return join(STATE_DIR, 'run-hidden.vbs') }
2077
-
2078
- // Run via the interpreter currently executing us: process.execPath is the
2079
- // absolute bun.exe (or the compiled vn.exe). Mirrors installLaunchAgent's
2080
- // compiled-vs-script detection.
2081
- function schedulerProgramArgs(): { command: string; argLine: string } {
2082
- const { cliPath, compiled } = resolveCli()
2083
- const args = compiled ? ['run'] : [cliPath, 'run']
2084
- const argLine = args.map(a => (/\s/.test(a) ? `"${a}"` : a)).join(' ')
2085
- return { command: process.execPath, argLine }
2086
- }
2087
-
2088
- async function installScheduledTask(opts: { load?: boolean } = {}): Promise<void> {
2089
- await mkdir(STATE_DIR, { recursive: true })
2090
- await mkdir(LOG_DIR, { recursive: true })
2091
- // The task carries no env (Task Scheduler has no per-task env block), so the
2092
- // bundled CLI paths the GUI injected via process env (pi runtime + cli.js +
2093
- // ffprobe) must be persisted to config.json, which `vn run` reads on startup.
2094
- // (On mac these ride in the LaunchAgent plist instead.)
2095
- const persist: Record<string, string> = {}
2096
- for (const k of ['VOICENOTE_PI_BIN', 'VOICENOTE_PI_CLI', 'VOICENOTE_FFPROBE_BIN'] as const) {
2097
- if (process.env[k]) persist[k] = process.env[k]!
2098
- }
2099
- if (Object.keys(persist).length) {
2100
- await mkdir(CONFIG_DIR, { recursive: true })
2101
- const current = loadConfigJson()
2102
- Object.assign(current, persist)
2103
- await writeConfigJson(current)
2104
- }
2105
- const { command, argLine } = schedulerProgramArgs()
2106
- // bun.exe / vn.exe are console-subsystem: an InteractiveToken task flashes a
2107
- // console window on every tick. Launch through wscript with window style 0
2108
- // (hidden). wait=True keeps wscript alive for the duration of `vn run` so
2109
- // IgnoreNew still prevents overlap, and WScript.Quit propagates vn's exit
2110
- // code so the task's Last Run Result stays meaningful. UTF-16 BOM so
2111
- // non-ASCII paths survive (wscript reads BOM-less files as ANSI).
2112
- const fullCmd = `"${command}" ${argLine}`
2113
- const vbs = `WScript.Quit CreateObject("WScript.Shell").Run("${fullCmd.replace(/"/g, '""')}", 0, True)\r\n`
2114
- await writeFile(taskVbsPath(), '\ufeff' + vbs, 'utf16le')
2115
- const wscript = join(process.env.SystemRoot || 'C:\\Windows', 'System32', 'wscript.exe')
2116
- // Register the task as the current user (DOMAIN\user; DOMAIN == machine name for
2117
- // local accounts). Without an explicit <UserId>, `schtasks /create /xml` can't tell
2118
- // who to register as and a standard (non-admin) user gets "Access is denied".
2119
- const taskUser = process.env.USERDOMAIN && process.env.USERNAME
2120
- ? `${process.env.USERDOMAIN}\\${process.env.USERNAME}`
2121
- : (process.env.USERNAME || os.userInfo().username)
2122
- // Local-time StartBoundary for the TimeTrigger (Task Scheduler wants no zone).
2123
- const n = new Date()
2124
- const startBoundary = `${n.getFullYear()}-${pad(n.getMonth() + 1)}-${pad(n.getDate())}T${pad(n.getHours())}:${pad(n.getMinutes())}:${pad(n.getSeconds())}`
2125
- // The task just runs `vn run`; config comes from config.json (vn config set /
2126
- // the GUI), so unlike the mac plist there's no env to embed. A TimeTrigger that
2127
- // repeats every PT1M (mirrors the working `schtasks /sc minute /mo 1` form; a
2128
- // LogonTrigger gave "Access is denied" for standard users) + IgnoreNew is the
2129
- // StartInterval(60)+flock equivalent.
2130
- const xml = `<?xml version="1.0" encoding="UTF-16"?>
2131
- <Task version="1.2" xmlns="http://schemas.microsoft.com/windows/2004/02/mit/task">
2132
- <RegistrationInfo>
2133
- <Description>VoiceNote: watch the recorder and process new recordings.</Description>
2134
- </RegistrationInfo>
2135
- <Triggers>
2136
- <TimeTrigger>
2137
- <StartBoundary>${startBoundary}</StartBoundary>
2138
- <Enabled>true</Enabled>
2139
- <Repetition>
2140
- <Interval>PT1M</Interval>
2141
- <StopAtDurationEnd>false</StopAtDurationEnd>
2142
- </Repetition>
2143
- </TimeTrigger>
2144
- </Triggers>
2145
- <Principals>
2146
- <Principal id="Author">
2147
- <UserId>${xmlEscape(taskUser)}</UserId>
2148
- <LogonType>InteractiveToken</LogonType>
2149
- <RunLevel>LeastPrivilege</RunLevel>
2150
- </Principal>
2151
- </Principals>
2152
- <Settings>
2153
- <MultipleInstancesPolicy>IgnoreNew</MultipleInstancesPolicy>
2154
- <DisallowStartIfOnBatteries>false</DisallowStartIfOnBatteries>
2155
- <StopIfGoingOnBatteries>false</StopIfGoingOnBatteries>
2156
- <StartWhenAvailable>true</StartWhenAvailable>
2157
- <ExecutionTimeLimit>PT2H</ExecutionTimeLimit>
2158
- <AllowHardTerminate>true</AllowHardTerminate>
2159
- <Enabled>true</Enabled>
2160
- <Hidden>false</Hidden>
2161
- </Settings>
2162
- <Actions Context="Author">
2163
- <Exec>
2164
- <Command>${xmlEscape(wscript)}</Command>
2165
- <Arguments>${xmlEscape(`//B //Nologo "${taskVbsPath()}"`)}</Arguments>
2166
- </Exec>
2167
- </Actions>
2168
- </Task>
2169
- `
2170
- const xmlPath = taskXmlPath()
2171
- // schtasks /xml wants UTF-16; prepend a BOM so non-ASCII paths survive.
2172
- await writeFile(xmlPath, '\ufeff' + xml, 'utf16le')
2173
- const r = await runCommand('schtasks', ['/create', '/tn', TASK_NAME, '/xml', xmlPath, '/f'], 15000)
2174
- if (r.code !== 0) {
2175
- console.error(`schtasks /create failed (exit ${r.code}): ${(r.stderr || r.stdout).trim()}`)
2176
- process.exitCode = 1
2177
- return
2178
- }
2179
- console.log(`Scheduled task '${TASK_NAME}' installed — runs \`vn run\` every 60s at/after logon.`)
2180
- console.log(`Command: ${command} ${argLine} (launched hidden via wscript)`)
2181
- console.log('Note: the task reads config from config.json — set it with `vn config set` (or the GUI) so the background run is configured.')
2182
- if (opts.load) await runCommand('schtasks', ['/run', '/tn', TASK_NAME], 10000)
2183
- }
2184
-
2185
- async function uninstallScheduledTask(): Promise<void> {
2186
- const r = await runCommand('schtasks', ['/delete', '/tn', TASK_NAME, '/f'], 10000)
2187
- // Remove our artifacts too: the VBS is the task's actual entry point, and a
2188
- // leftover copy could make schedulerIsCurrent misjudge a future install.
2189
- // Only when the task is actually gone — deleting the VBS while the task is
2190
- // still registered would turn every tick into a silent wscript failure.
2191
- if (r.code === 0) {
2192
- for (const p of [taskVbsPath(), taskXmlPath()]) {
2193
- try { unlinkSync(p) } catch (e: any) { if (e?.code !== 'ENOENT') warnSideEffect(`remove scheduler artifact ${p}`, e) }
2194
- }
2195
- }
2196
- console.log(r.code === 0 ? `Scheduled task '${TASK_NAME}' removed.` : `schtasks /delete: ${(r.stderr || r.stdout).trim()}`)
2197
- }
2198
-
2199
- // ── Cross-platform scheduler dispatch ──
2200
- function installScheduler(opts: { load?: boolean } = {}): Promise<void> {
2201
- return IS_WINDOWS ? installScheduledTask(opts) : installLaunchAgent(opts)
2202
- }
2203
- function uninstallScheduler(): Promise<void> {
2204
- return IS_WINDOWS ? uninstallScheduledTask() : uninstallLaunchAgent()
2205
- }
2206
- async function printSchedulerStatus(): Promise<void> {
2207
- if (IS_WINDOWS) {
2208
- const r = await runCommand('schtasks', ['/query', '/tn', TASK_NAME, '/v', '/fo', 'LIST'], 10000)
2209
- process.stdout.write(r.stdout || r.stderr || `Task '${TASK_NAME}' not found.\n`)
2210
- return
2211
- }
2212
- const r = await runCommand('launchctl', ['print', `gui/${process.getuid?.()}/${LAUNCH_AGENT_LABEL}`], 10000)
2213
- process.stdout.write(r.stdout || r.stderr)
2214
- }
2215
-
2216
- // ────────────────────────────────────────────────────────────────────────────
2217
- // Browse / debug commands
2218
- // ────────────────────────────────────────────────────────────────────────────
2219
-
2220
- async function listMeetings(opts: { month?: string }): Promise<void> {
2221
- const config = getConfig()
2222
- const month = opts.month || `${new Date().getFullYear()}-${pad(new Date().getMonth() + 1)}`
2223
- const dir = join(config.workspace, month)
2224
- if (!existsSync(dir)) {
2225
- console.log(`No notes in ${dir}`)
2226
- return
2227
- }
2228
- const entries = (await readdir(dir)).filter(f => f.endsWith('.md')).sort()
2229
- if (!entries.length) {
2230
- console.log(`No notes in ${dir}`)
2231
- return
2232
- }
2233
- for (const name of entries) {
2234
- console.log(join(dir, name))
2235
- }
2236
- }
2237
-
2238
- // One-time migration: pre-0.15.4 wrote _index/meetings.jsonl. Rename it to the new
2239
- // canonical notes.jsonl on first access so all history stays in a single file.
2240
- async function notesIndexPath(config: Config): Promise<string> {
2241
- const p = join(config.workspace, '_index', 'notes.jsonl')
2242
- const legacy = join(config.workspace, '_index', 'meetings.jsonl')
2243
- if (!existsSync(p) && existsSync(legacy)) await rename(legacy, p).catch(e => warnSideEffect(`rename ${legacy}`, e))
2244
- return p
2245
- }
2246
-
2247
- async function lastMeeting(): Promise<void> {
2248
- const config = getConfig()
2249
- const indexPath = await notesIndexPath(config)
2250
- if (!existsSync(indexPath)) {
2251
- console.log('No notes indexed yet.')
2252
- return
2253
- }
2254
- const lines = (await readFile(indexPath, 'utf8')).trim().split('\n').filter(Boolean)
2255
- const last = lines[lines.length - 1]
2256
- if (!last) {
2257
- console.log('No notes indexed yet.')
2258
- return
2259
- }
2260
- let obj: Json
2261
- try { obj = JSON.parse(last) } catch { console.log(last); return }
2262
- console.log(`Title: ${obj.title}`)
2263
- console.log(`Date: ${obj.date} ${obj.start_time || ''}-${obj.end_time || ''}`)
2264
- console.log(`Status: ${obj.status || 'unknown'}`)
2265
- console.log(`Notes: ${obj.final_paths?.notes || obj.local_paths?.notes}`)
2266
- console.log(`Transcript: ${obj.final_paths?.transcript || obj.local_paths?.transcript}`)
2267
- console.log(`Audio: ${obj.final_paths?.audio || obj.local_paths?.audio}`)
2268
- }
2269
-
2270
-
2271
- async function openTarget(arg?: string): Promise<void> {
2272
- const config = getConfig()
2273
- let target = config.workspace
2274
- if (arg === 'config') {
2275
- target = CONFIG_DIR
2276
- } else if (arg === 'logs') {
2277
- target = LOG_DIR
2278
- } else if (arg) {
2279
- // Try matching most recent file in current month containing arg.
2280
- const month = `${new Date().getFullYear()}-${pad(new Date().getMonth() + 1)}`
2281
- const dir = join(config.workspace, month)
2282
- if (existsSync(dir)) {
2283
- const matches = (await readdir(dir)).filter(f => f.includes(arg) && f.endsWith('.md'))
2284
- if (matches.length) target = join(dir, matches[matches.length - 1]!)
2285
- }
2286
- }
2287
- await openPath(target)
2288
- console.log(`open ${target}`)
2289
- }
2290
-
2291
- async function forgetRecording(needle: string): Promise<void> {
2292
- const config = getConfig()
2293
- // Under the run lock: `vn run` holds the state file in memory for the length
2294
- // of a batch and re-saves after every job, so an unlocked delete here would be
2295
- // silently resurrected by the next save.
2296
- const lock = await acquireRunLock()
2297
- if (!lock) { console.error('A voicenote run is in progress, so the state file is busy. Re-run this once it finishes (`vn jobs` shows what it is working on).'); process.exitCode = 1; return }
2298
- try {
2299
- await migrateStateOnDisk(config)
2300
- const store = await loadState(config)
2301
- let removed = 0
2302
- for (const [id, entry] of Object.entries(store.jobs)) {
2303
- if (id === needle || entry.source_path.includes(needle) || entry.name.includes(needle)) {
2304
- delete store.jobs[id]
2305
- removed++
2306
- }
2307
- }
2308
- await saveState(config, store)
2309
- console.log(`forgot ${removed} record(s)`)
2310
- } finally { await lock.release() }
2311
- }
2312
-
2313
- async function retryRecording(id: string): Promise<void> {
2314
- const config = getConfig()
2315
- const lock = await acquireRunLock()
2316
- if (!lock) throw new Error('A voicenote run is in progress. Retry once it finishes.')
2317
- try {
2318
- await migrateStateOnDisk(config)
2319
- const store = await loadState(config)
2320
- const entry = store.jobs[id]
2321
- if (!entry) throw new Error('Recording no longer exists in the processing list.')
2322
- if (!requeueFailed(entry, nowIso())) throw new Error(`Cannot retry a recording in state '${entry.state}'.`)
2323
- await saveState(config, store)
2324
- console.log(`queued ${entry.name} for retry`)
2325
- } finally { await lock.release() }
2326
- }
2327
-
2328
- async function showLog(opts: { lines?: number; follow?: boolean; err?: boolean; date?: string }): Promise<void> {
2329
- const lines = Number(opts.lines || 30)
2330
- const wanted = [opts.date ? join(LOG_DIR, `${opts.date}.log`) : dailyLogPath()]
2331
- if (opts.err) wanted.push(join(LOG_DIR, 'launchd.err.log'))
2332
- const files = wanted.filter(f => existsSync(f))
2333
- if (!files.length) {
2334
- console.log(`No log file: ${wanted.join(', ')}`)
2335
- return
2336
- }
2337
- await tailFiles(files, lines, !!opts.follow)
2338
- }
2339
-
2340
- async function showErrors(opts: { lines?: number }): Promise<void> {
2341
- if (!existsSync(LOG_DIR)) {
2342
- console.log('No logs.')
2343
- return
2344
- }
2345
- // Only the daily rolling logs (YYYY-MM-DD.log) carry timestamped [ERROR] lines;
2346
- // launchd.out.log/launchd.err.log are raw, never-truncated stdout/stderr mirrors
2347
- // that sort after dated files alphabetically ('l' > digit) and would otherwise
2348
- // crowd out the real recent logs in the slice(-3) below.
2349
- const files = (await readdir(LOG_DIR)).filter(f => /^\d{4}-\d{2}-\d{2}\.log$/.test(f)).sort().slice(-3)
2350
- if (!files.length) {
2351
- // Distinguish "no dated logs yet" (fresh install) from "scanned, no errors".
2352
- console.log('No logs.')
2353
- return
2354
- }
2355
- const lineCount = Number(opts.lines || 20)
2356
- const errors: string[] = []
2357
- for (const f of files) {
2358
- const content = await readFile(join(LOG_DIR, f), 'utf8').catch(() => '')
2359
- for (const line of content.split('\n')) {
2360
- if (line.includes('[ERROR]') || line.includes('ERROR processing')) errors.push(line)
2361
- }
2362
- }
2363
- for (const line of errors.slice(-lineCount)) console.log(line)
2364
- }
2365
-
2366
- async function upgradeSelf(): Promise<void> {
2367
- // The registry fetch needs the configured proxy: `bun add -g` only sees it if
2368
- // we pass it, because the proxy lives in config.json, not in the shell.
2369
- const env = { ...process.env, ...getConfig().childEnv }
2370
- // Plain `bun` from PATH: vn is started by bun (`#!/usr/bin/env bun`), so an
2371
- // interactive upgrade always has it. If it is somehow missing, the spawn error
2372
- // below says so instead of the command silently "failing".
2373
- // `bun add -g` upgrades in place: verified no dependency loop on npm→npm re-add
2374
- // (the steady-state upgrade path) nor on replacing an old git-ref install. No
2375
- // remove-first, so a failed add leaves the running vn intact.
2376
- console.log('$ bun add -g @fastagent-sh/voicenote')
2377
- const addCode = await new Promise<number>(res =>
2378
- spawn('bun', ['add', '-g', '@fastagent-sh/voicenote'], { stdio: 'inherit', shell: IS_WINDOWS, env })
2379
- .on('close', c => res(c ?? 1))
2380
- .on('error', (e: Error) => { console.error(`Cannot run bun: ${e.message}`); res(1) }))
2381
- if (addCode !== 0) {
2382
- console.error(`Upgrade failed: \`bun add -g @fastagent-sh/voicenote\` exited ${addCode}. Your current install is unchanged; retry later.`)
2383
- process.exitCode = 1
2384
- return
2385
- }
2386
- // Refresh the background scheduler so it points at the upgraded version. This
2387
- // process is still the OLD code in memory, so invoke the freshly installed binary
2388
- // to regenerate.
2389
- if (IS_WINDOWS) {
2390
- const installed = (await runCommand('schtasks', ['/query', '/tn', TASK_NAME], 10000)).code === 0
2391
- if (installed) {
2392
- const code = await new Promise<number>(res =>
2393
- spawn('vn', ['install-launch-agent'], { stdio: 'inherit', shell: true })
2394
- .on('close', c => res(c ?? 1)).on('error', () => res(1)))
2395
- console.log(code === 0 ? 'Scheduled task refreshed.' : 'Warning: `vn install-launch-agent` failed; re-register manually.')
2396
- }
2397
- return
2398
- }
2399
- if (existsSync(plistPath())) {
2400
- console.log('Refreshing LaunchAgent plist for the upgraded version…')
2401
- const code = await new Promise<number>(res =>
2402
- spawn('vn', ['install-launch-agent'], { stdio: 'inherit' })
2403
- .on('close', c => res(c ?? 1)).on('error', () => res(1)))
2404
- if (code !== 0) {
2405
- console.error(`Warning: \`vn install-launch-agent\` failed (exit ${code}); the LaunchAgent still points at the previous version. Ensure vn is on PATH and re-run \`vn install-launch-agent\`.`)
2406
- return
2407
- }
2408
- const uid = process.getuid?.()
2409
- await runCommand('launchctl', ['bootout', `gui/${uid}`, plistPath()], 10000) // ok if not currently loaded
2410
- const bs = await runCommand('launchctl', ['bootstrap', `gui/${uid}`, plistPath()], 10000)
2411
- if (bs.code !== 0) {
2412
- console.error(`Warning: launchctl bootstrap failed: ${(bs.stderr || bs.stdout).trim()}. Reload manually: launchctl bootstrap gui/$(id -u) ${plistPath()}`)
2413
- return
2414
- }
2415
- console.log('LaunchAgent reloaded.')
2416
- }
2417
- }
2418
-
2419
- // ────────────────────────────────────────────────────────────────────────────
2420
- // Doctor
2421
- // ────────────────────────────────────────────────────────────────────────────
2422
-
2423
- // Read up to the last `maxBytes` of a (possibly large, ever-appending) log file
2424
- // without slurping the whole thing — used to surface the agent's latest activity.
2425
- function readLogTail(path: string, maxBytes: number): string {
2426
- try {
2427
- const size = statSync(path).size
2428
- const start = Math.max(0, size - maxBytes)
2429
- const len = size - start
2430
- const fd = openSync(path, 'r')
2431
- try {
2432
- const buf = Buffer.alloc(len)
2433
- readSync(fd, buf, 0, len, start)
2434
- return buf.toString('utf8')
2435
- } finally { closeSync(fd) }
2436
- } catch { return '' }
2437
- }
2438
-
2439
- // Where the background agent's latest activity lands. mac: launchd redirects
2440
- // the agent's stdout to launchd.out.log. Windows: Task Scheduler redirects
2441
- // nothing — the agent's own daily rolling log is the only mirror of its
2442
- // output. wireDailyLog captures the log path once at process start, so a run
2443
- // spanning midnight keeps writing to its START day's file; pick the
2444
- // most-recently-modified dated log rather than today's by name, or a
2445
- // still-running cross-midnight job would look idle on the dashboard.
2446
- function agentLogPath(): string {
2447
- if (!IS_WINDOWS) return join(LOG_DIR, 'launchd.out.log')
2448
- try {
2449
- const dated = readdirSync(LOG_DIR)
2450
- .filter(f => /^\d{4}-\d{2}-\d{2}\.log$/.test(f))
2451
- .map(f => join(LOG_DIR, f))
2452
- let newest: string | null = null
2453
- let newestMs = -Infinity
2454
- for (const p of dated) {
2455
- const ms = statSync(p).mtimeMs
2456
- if (ms > newestMs) { newestMs = ms; newest = p }
2457
- }
2458
- return newest ?? dailyLogPath()
2459
- } catch { return dailyLogPath() }
2460
- }
2461
-
2462
- // Is the background scheduler installed at all (any version)? Cheaper cousin
2463
- // of schedulerIsCurrent(), used for the dashboard's installed/not-installed
2464
- // pill — mac checks the plist file, Windows must ask schtasks (there is no
2465
- // file whose existence tracks task registration).
2466
- async function schedulerInstalledAtAll(): Promise<boolean> {
2467
- if (IS_WINDOWS) return (await runCommand('schtasks', ['/query', '/tn', TASK_NAME], 10000)).code === 0
2468
- return existsSync(plistPath())
2469
- }
2470
-
2471
- // Background agent snapshot for the dashboard (LaunchAgent / Scheduled Task).
2472
- async function agentStatus() {
2473
- const logFile = agentLogPath()
2474
- let logTail: string[] = []
2475
- let logAt: string | null = null
2476
- if (existsSync(logFile)) {
2477
- try { logAt = statSync(logFile).mtime.toISOString() } catch {}
2478
- logTail = readLogTail(logFile, 16384).split('\n').map(s => s.trim()).filter(Boolean).slice(-8)
2479
- }
2480
- // `scheduler` points at the on-disk scheduler entry for `vn doctor` to show.
2481
- // mac: the plist IS the registration (its existence == installed). Windows:
2482
- // the task XML is only the staging file we wrote; registration lives in Task
2483
- // Scheduler (queried by `installed`), so the XML may lag reality — it's an
2484
- // inspection aid, not proof of registration.
2485
- return { installed: await schedulerInstalledAtAll(), scheduler: IS_WINDOWS ? taskXmlPath() : plistPath(), logAt, logTail }
2486
- }
2487
-
2488
- // Structured health/config snapshot. Single source for both `vn doctor` (text)
2489
- // and `vn doctor --json` (consumed by the GUI status dashboard).
2490
- async function collectDoctor() {
2491
- const config = getConfig()
2492
- // pi is a bun-based CLI; cold start (esp. behind a proxy) can take >5s, so
2493
- // give --version a generous timeout to avoid a false 'missing' on a healthy pi.
2494
- const piInv = piInvocation(config.pi, ['--version'])
2495
- const piCheck = await runCommand(piInv.bin, piInv.args, 15000)
2496
- const ff = await runCommand(config.ffprobeBin, ['-version'], 5000)
2497
- const v = config.volcano
2498
- const { pi } = config
2499
- return {
2500
- version: VERSION,
2501
- bun: process.versions.bun || null,
2502
- node: process.version,
2503
- recorder: { dir: config.recordDir, exists: existsSync(config.recordDir) },
2504
- workspace: config.workspace,
2505
- volcano: v
2506
- ? {
2507
- configured: true as const,
2508
- auth: 'new-console',
2509
- resourceId: v.resourceId,
2510
- tos: { bucket: v.tos.bucket, region: v.tos.region, endpoint: v.tos.endpoint, keep: v.tos.keep, accessKey: !!v.tos.accessKey, secretKey: !!v.tos.secretKey },
2511
- language: v.language ?? null,
2512
- }
2513
- : { configured: false as const },
2514
- // Provider/model/credentials are pi's own configuration; `pi.available` is
2515
- // all we can honestly report about whether a summary can run.
2516
- summary: { backend: 'pi', model: pi.model, thinking: pi.thinking, tools: pi.tools || null, contextDir: pi.tools ? pi.contextDir : null },
2517
- pi: { bin: pi.bin, version: piCheck.code === 0 ? (piCheck.stdout.trim() || piCheck.stderr.trim() || null) : null, available: piCheck.code === 0, auth: existsSync(pi.authPath), authPath: pi.authPath },
2518
- // Outbound proxy for HTTPS endpoints (updater/GitHub). The GUI reads this to
2519
- // route its own update check, so it reports the resolved value.
2520
- proxy: { url: config.childEnv.https_proxy ?? null },
2521
- identity: { self: config.speakers.self.name || null, aliases: config.speakers.self.aliases, knownCount: config.speakers.known.length },
2522
- // The thresholds that silently decide what never gets processed. Without
2523
- // them here, confirming a change to VOICENOTE_MAX_AGE_HOURS meant planting
2524
- // a test recording and watching the scan — not a reasonable way to check
2525
- // a setting.
2526
- filters: { maxAgeHours: config.maxAgeHours, minBytes: config.minBytes, minDurationSeconds: config.minDurationSeconds },
2527
- deps: { ffprobe: ff.code === 0 },
2528
- agent: await agentStatus(),
2529
- }
2530
- }
2531
-
2532
- // The dashboard/CLI view of every recording's processing status. A pure read of
2533
- // the state file `vn run` writes, grouped by jobs.ts. Nothing here rescans
2534
- // the recorder or parses logs: the queue shown IS the queue that runs, and it
2535
- // stays visible when the recorder is unplugged.
2536
- async function jobsListData(limit: number): Promise<{ items: Json[]; total: number; queued_total: number; recorder_present: boolean }> {
2537
- const config = getConfig()
2538
- const store = await loadState(config)
2539
- // One existsSync on the mount point — not the recursive glob the old pending
2540
- // section ran on every poll, and always current.
2541
- return buildJobsView(store, readCurrent(), { limit, alive: pidAlive, recorderPresent: existsSync(config.recordDir) })
2542
- }
2543
-
2544
- async function jobsList(opts: { limit?: number; json?: boolean }): Promise<void> {
2545
- let limit: number
2546
- try { limit = parseJobsLimit(opts.limit, 30) } catch (e: any) { console.error(e.message); process.exitCode = 1; return }
2547
- const data = await jobsListData(limit)
2548
- if (opts.json) { console.log(JSON.stringify(data, null, 2)); return }
2549
- if (!data.items.length) {
2550
- console.log(data.recorder_present ? 'No jobs yet.' : 'No jobs yet. (recorder not connected)')
2551
- return
2552
- }
2553
- for (const j of data.items) {
2554
- const suffix = [j.step, j.detail].filter(Boolean).join(' \u00b7 ')
2555
- console.log(`[${j.status}] ${j.title || j.name}${suffix ? ' \u00b7 ' + suffix.slice(0, 140) : ''}`)
2556
- }
2557
- // Truncation used to be silent, which is how a 126-entry backlog read as 27.
2558
- if (data.total > data.items.length) console.log(`\u2026 ${data.total - data.items.length} more (vn jobs --limit 0 to show all)`)
2559
- if (!data.recorder_present) {
2560
- console.log(data.queued_total ? `Recorder not connected \u2014 ${data.queued_total} recording(s) waiting for it.` : 'Recorder not connected.')
2561
- }
2562
- }
2563
-
2564
- async function doctor(opts: { json?: boolean } = {}): Promise<void> {
2565
- const s = await collectDoctor()
2566
- if (opts.json) { console.log(JSON.stringify(s, null, 2)); return }
2567
- console.log(`version=${s.version}`)
2568
- console.log(`bun=${s.bun || 'not-bun'}`)
2569
- console.log(`node=${s.node}`)
2570
- console.log(`recordDir=${s.recorder.dir} exists=${s.recorder.exists}`)
2571
- console.log(`workspace=${s.workspace}`)
2572
- console.log(`filters=maxAge:${s.filters.maxAgeHours > 0 ? `${s.filters.maxAgeHours}h` : 'none'} minSize:${(s.filters.minBytes / 1000).toFixed(0)}KB minDuration:${s.filters.minDurationSeconds}s`)
2573
- if (s.volcano.configured) {
2574
- console.log(`volcano.auth=${s.volcano.auth}`)
2575
- console.log(`volcano.resourceId=${s.volcano.resourceId}`)
2576
- console.log(`volcano.tos=bucket:${s.volcano.tos.bucket} region:${s.volcano.tos.region} endpoint:${s.volcano.tos.endpoint} keep:${s.volcano.tos.keep}`)
2577
- console.log(`volcano.tos.accessKey=${s.volcano.tos.accessKey ? 'loaded' : 'missing'} secretKey=${s.volcano.tos.secretKey ? 'loaded' : 'missing'}`)
2578
- if (s.volcano.language) console.log(`volcano.language=${s.volcano.language}`)
2579
- } else {
2580
- console.log(`volcano=not configured`)
2581
- }
2582
- console.log(`summaryBackend=${s.summary.backend}`)
2583
- console.log(`pi.bin=${s.pi.bin} model=${s.summary.model || "<pi's own default>"}`)
2584
- console.log(`pi.thinking=${s.summary.thinking}`)
2585
- console.log(`pi.summaryTools=${s.summary.tools || '<disabled>'}`)
2586
- if (s.summary.contextDir) console.log(`pi.contextDir=${s.summary.contextDir} (summary agent cwd + read/grep cross-reference root)`)
2587
- console.log(`pi.version=${s.pi.version || 'missing'}`)
2588
- // Neutral fact, not an instruction: an API-key user has no auth.json and needs
2589
- // nothing fixed.
2590
- console.log(`pi.auth=${s.pi.authPath} ${s.pi.auth ? '(present)' : '(missing — fine if a provider API key is set)'}`)
2591
- console.log(`defaultMode=notes`)
2592
- console.log(`proxy=${s.proxy.url || '<unset>'}`)
2593
- console.log(`speakers.self=${s.identity.self || '<unset>'}`)
2594
- console.log(`speakers.known=${s.identity.knownCount}`)
2595
- console.log(`scheduler=${s.agent.scheduler}`)
2596
- console.log(`ffprobe=${s.deps.ffprobe ? 'ok' : 'missing'}`)
2597
- }
2598
-
2599
- // Is the background scheduler installed and pointing at this binary?
2600
- async function schedulerIsCurrent(): Promise<boolean> {
2601
- const exe = process.execPath
2602
- if (IS_WINDOWS) {
2603
- if ((await runCommand('schtasks', ['/query', '/tn', TASK_NAME], 10000)).code !== 0) return false
2604
- // The task XML points at wscript; the actual CLI path lives in the VBS.
2605
- try { return readFileSync(taskVbsPath(), 'utf16le').includes(exe) } catch { return false }
2606
- }
2607
- try { return readFileSync(plistPath(), 'utf8').includes(exe) } catch { return false }
2608
- }
2609
-
2610
- async function ensureScheduler(force: boolean): Promise<{ ok: true; skipped?: boolean }> {
2611
- if (!force && await schedulerIsCurrent()) return { ok: true, skipped: true }
2612
- await installScheduler({ load: true })
2613
- return { ok: true }
2614
- }
2615
-
2616
- // ────────────────────────────────────────────────────────────────────────────
2617
- // CLI commands
2618
- // ────────────────────────────────────────────────────────────────────────────
2619
-
2620
- const cli = cac('vn')
2621
-
2622
- cli.command('run [file]', 'Scan recorder and process recordings, or process one audio file by path (Volcano ASR + pi notes)')
2623
- .option('--mode <mode>', 'Output mode: notes (default) | transcript', { default: 'notes' })
2624
- .option('--latest', 'Only process newest eligible recording')
2625
- .option('--force', 'Reprocess already processed recordings')
2626
- .option('--dry-run', 'Do not copy / transcribe / write files')
2627
- .option('--pdf', 'Also render notes to PDF (only meaningful for --mode notes)')
2628
- .option('--verbose', 'Print per-file skip details during scan')
2629
- .action(runPipeline)
2630
-
2631
- cli.command('list', 'List notes in a month')
2632
- .option('--month <YYYY-MM>', 'Month to list (default: current month)')
2633
- .action(listMeetings)
2634
-
2635
- cli.command('last', 'Print summary of most recent processed recording').action(lastMeeting)
2636
- cli.command('jobs', 'Show every recording\'s processing status (running, queued, done, failed, gave up, filtered)')
2637
- .option('--limit <n>', 'How many to list', { default: 30 })
2638
- .option('--json', 'Output as JSON (for the GUI)')
2639
- .action((opts: { limit?: number; json?: boolean }) => jobsList(opts))
2640
-
2641
-
2642
- cli.command('open [target]', 'Open notes dir, config dir (`config`), logs dir (`logs`), or a note matching the slug').action((target?: string) => openTarget(target))
2643
-
2644
- cli.command('forget <key>', 'Drop a recording\'s job record so it is queued again (a saved transcript on disk is still reused)').action((key: string) => forgetRecording(key))
2645
- cli.command('retry <id>', 'Requeue one failed recording while retaining saved outputs').action((id: string) => retryRecording(id))
2646
-
2647
- cli.command('log', 'Print the daily log (today by default)')
2648
- .option('--lines <n>', 'How many trailing lines to print', { default: 30 })
2649
- .option('-f, --follow', 'Follow the log live (tail -F)')
2650
- .option('--err', 'Also include launchd.err.log')
2651
- .option('--date <YYYY-MM-DD>', 'Show a specific day instead of today')
2652
- .action(showLog)
2653
-
2654
- cli.command('errors', 'Show recent ERROR lines from daily logs').option('--lines <n>', 'How many lines to print', { default: 20 }).action(showErrors)
2655
-
2656
- cli.command('upgrade', 'Upgrade to the latest published version via bun add -g').action(upgradeSelf)
2657
-
2658
- cli.command('doctor', 'Check environment')
2659
- .option('--json', 'Output structured status as JSON (for the GUI)')
2660
- .action((opts: { json?: boolean }) => doctor(opts))
2661
- cli.command('login', 'Sign in to ChatGPT (Codex OAuth) for the pi summary backend')
2662
- .option('--json', 'Emit machine-readable JSON events (for the GUI client)')
2663
- .option('--device-code', 'Use the device-code flow instead of the browser callback (needs the ChatGPT security-settings opt-in)')
2664
- .action((opts: { json?: boolean; deviceCode?: boolean }) => loginChatGPT(opts))
2665
- cli.command('config <action>', 'Read/write file-based config. action: get (print JSON) | set (write from stdin JSON)')
2666
- .action((action: string) => {
2667
- if (action === 'set') return configSet()
2668
- if (action === 'get') return configGet()
2669
- console.error(`Unknown config action '${action}'. Use: vn config get | vn config set`)
2670
- process.exitCode = 1
2671
- })
2672
- cli.command('install-launch-agent', 'Install background scheduler (mac LaunchAgent / Windows Task Scheduler)')
2673
- .option('--load', 'Also (re)load/start it immediately')
2674
- .action((opts: { load?: boolean }) => installScheduler(opts))
2675
- cli.command('ensure-launch-agent', 'Install the background scheduler when missing or stale')
2676
- .option('--force', 'Reinstall even when the scheduler is current')
2677
- .action((opts: { force?: boolean }) => ensureScheduler(!!opts.force))
2678
- cli.command('uninstall-launch-agent', 'Remove the background scheduler').action(uninstallScheduler)
2679
- cli.command('status', 'Print background scheduler status').action(printSchedulerStatus)
2680
-
2681
- cli.help()
2682
- cli.version(VERSION)
2683
- // Run the command ourselves so a thrown error (bad config, unreadable state
2684
- // file) reaches the user as the one line it is, not as a bun stack trace.
2685
- const parsed = cli.parse(process.argv, { run: false })
2686
- // cac prints --help/--version itself and then reports no matched command; any
2687
- // OTHER unmatched invocation is a typo, which it would ignore in silence.
2688
- if (!cli.matchedCommand && !parsed.options.help && !parsed.options.version) {
2689
- if (parsed.args.length) console.error(`vn: unknown command '${parsed.args[0]}'`)
2690
- cli.outputHelp()
2691
- process.exit(parsed.args.length ? 1 : 0)
2692
- }
2693
- try {
2694
- await cli.runMatchedCommand()
2695
- } catch (e: any) {
2696
- console.error(`vn: ${e?.message || e}`)
2697
- process.exit(1)
2698
- }