@fastagent-sh/voicenote 0.22.2 → 1.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/cli.ts DELETED
@@ -1,2800 +0,0 @@
1
- #!/usr/bin/env bun
2
- import { cac } from 'cac'
3
- import packageJson from '../package.json' with { type: 'json' }
4
- import { parseLockOwner } from './runLock'
5
- import { tosObject, type TosConfig as VolcanoTosConfig } from './tos'
6
- import { applyOutcome, buildJobsView, classify, emptyState, localIso, MAX_ATTEMPTS, migrateLegacyState, ownsOutput, parseJobsLimit, parseStateFile, parseStrictJson, patchJob, pruneUnseen, reconcileInterrupted, requeueFailed, startAttempt, SUMMARY_FAILED_STATUS, type CurrentJob, type JobRecord, type StateFile } from './jobs'
7
- import { createHash, randomUUID } from 'node:crypto'
8
- import { appendFile, chmod, mkdir, readFile, writeFile, copyFile, rename, unlink, stat, readdir, rmdir, utimes } from 'node:fs/promises'
9
- import { existsSync, readFileSync, readdirSync, mkdirSync, writeFileSync, appendFileSync, openSync, closeSync, statSync, readSync, unlinkSync, renameSync } from 'node:fs'
10
- import { dlopen, FFIType, suffix } from 'bun:ffi'
11
- import { basename, dirname, extname, join, resolve } from 'node:path'
12
- import { fileURLToPath, pathToFileURL } from 'node:url'
13
- import { spawn, spawnSync } from 'node:child_process'
14
- import os from 'node:os'
15
-
16
- const VERSION = packageJson.version
17
- const LAUNCH_AGENT_LABEL = 'sh.fastagent.voicenote'
18
- const LAUNCH_AGENT_LABEL_LEGACY = 'com.kid7st.voicenote' // pre-fastagent installs; cleaned up on install
19
- const TASK_NAME = 'VoiceNote' // Windows Task Scheduler name (mac uses LAUNCH_AGENT_LABEL)
20
-
21
- // Single switch every platform branch routes through. Declared before the path
22
- // consts so they can read it.
23
- const IS_WINDOWS = process.platform === 'win32'
24
- const IS_MAC = process.platform === 'darwin'
25
-
26
- // Per-OS base dirs. Windows -> native AppData (Roaming for config, Local for
27
- // logs/lock/state); mac/Linux -> ~/.config and ~/.local/state. appConfigDir /
28
- // appStateDir are hoisted function decls (defined just below).
29
- const CONFIG_DIR = appConfigDir()
30
- const STATE_DIR = appStateDir()
31
- const LOG_DIR = join(STATE_DIR, 'logs')
32
- const LOCK_PATH = join(STATE_DIR, 'run.lock')
33
- const CONFIG_ENV_PATH = join(CONFIG_DIR, 'config.json')
34
-
35
- const AUDIO_EXTENSIONS = new Set(['.mp3', '.wav', '.m4a', '.wma', '.aac', '.flac'])
36
-
37
- // Per-OS base directory resolution (see CONFIG_DIR / STATE_DIR above).
38
- function appConfigDir(): string {
39
- if (IS_WINDOWS) return join(process.env.APPDATA || join(os.homedir(), 'AppData', 'Roaming'), 'voicenote')
40
- return join(os.homedir(), '.config', 'voicenote')
41
- }
42
- function appStateDir(): string {
43
- if (IS_WINDOWS) return join(process.env.LOCALAPPDATA || join(os.homedir(), 'AppData', 'Local'), 'voicenote')
44
- return join(os.homedir(), '.local', 'state', 'voicenote')
45
- }
46
-
47
- type Json = Record<string, any>
48
-
49
- type Recording = {
50
- sourcePath: string
51
- sizeBytes: number
52
- modifiedAt: string
53
- durationSeconds: number | null
54
- sourceId: string
55
- contentHash: string
56
- recordedAt: Date
57
- imported: boolean
58
- }
59
-
60
- type LocalFiles = {
61
- audio: string
62
- transcript: string
63
- notes: string
64
- metadata: string
65
- }
66
-
67
- type SpeakerSelf = { name: string | null; aliases: string[] }
68
- type SpeakerKnown = { name: string; aliases: string[]; relationship?: string | null }
69
- type SpeakersConfig = { self: SpeakerSelf; known: SpeakerKnown[] }
70
-
71
-
72
- type VolcanoConfig = {
73
- apiKey: string // X-Api-Key (new Volcano console)
74
- resourceId: string
75
- language?: string
76
- tos: VolcanoTosConfig
77
- }
78
-
79
- /** How this install runs pi: which binary, which model, what it may read. */
80
- type PiConfig = {
81
- bin: string
82
- /** Set when pi ships as plain JS next to a bundled bun: `<bin> <cli> <args>`. */
83
- cli: string | null
84
- model: string | null
85
- thinking: string
86
- /** Comma-separated tool list; empty = run the summary without tools. */
87
- tools: string
88
- contextDir: string
89
- retries: number
90
- authPath: string
91
- }
92
-
93
- type Config = {
94
- recordDir: string
95
- workspace: string
96
- minBytes: number
97
- minDurationSeconds: number
98
- maxAgeHours: number
99
- speakers: SpeakersConfig
100
- volcano: VolcanoConfig | null
101
- ffprobeBin: string
102
- pi: PiConfig
103
- /** Added to the environment of every process vn spawns. */
104
- childEnv: Record<string, string>
105
- }
106
-
107
- // ────────────────────────────────────────────────────────────────────────────
108
- // Settings → Config
109
- //
110
- // config.json is the only persisted source; the inherited environment overrides
111
- // it for this process only. Everything the program needs is resolved once, in
112
- // getConfig(), and passed down as a frozen Config — no code below reads a
113
- // business setting out of process.env, so behaviour can never depend on whether
114
- // some earlier call happened to hydrate it.
115
- // ────────────────────────────────────────────────────────────────────────────
116
-
117
- // Config keys accepted by `vn config set` and loaded from config.json when the
118
- // inherited environment does not already define them.
119
- const ENV_KEYS = [
120
- 'VOICENOTE_DEVICE_VOLUME',
121
- 'VOICENOTE_RECORD_DIR',
122
- 'VOICENOTE_WORKSPACE',
123
- 'VOICENOTE_MIN_BYTES',
124
- 'VOICENOTE_MIN_DURATION_SECONDS',
125
- 'VOICENOTE_MAX_AGE_HOURS',
126
- 'VOLCANO_ASR_KEY',
127
- 'VOLCANO_ASR_RESOURCE_ID',
128
- 'VOLCANO_ASR_LANGUAGE',
129
- 'VOLCANO_TOS_REGION',
130
- 'VOLCANO_TOS_ENDPOINT',
131
- 'VOLCANO_TOS_BUCKET',
132
- 'VOLCANO_TOS_ACCESS_KEY',
133
- 'VOLCANO_TOS_SECRET_KEY',
134
- 'VOLCANO_TOS_KEEP',
135
- 'VOICENOTE_PI_BIN',
136
- 'VOICENOTE_PI_CLI',
137
- 'PI_CODING_AGENT_DIR',
138
- 'VOICENOTE_FFPROBE_BIN',
139
- 'VOICENOTE_PI_MODEL',
140
- 'VOICENOTE_PI_RETRIES',
141
- 'VOICENOTE_PI_THINKING',
142
- 'VOICENOTE_PI_SUMMARY_TOOLS',
143
- 'VOICENOTE_CONTEXT_DIR',
144
- 'http_proxy', 'https_proxy', 'all_proxy', 'no_proxy',
145
- 'HTTP_PROXY', 'HTTPS_PROXY', 'ALL_PROXY', 'NO_PROXY',
146
- 'LOCAL_PROXY_HOST', 'LOCAL_PROXY_PORT', 'LOCAL_NO_PROXY',
147
- 'OPENAI_API_KEY',
148
- 'DEEPSEEK_API_KEY',
149
- ]
150
-
151
- // Volcano endpoints (TOS object storage + openspeech ASR) should NEVER go through
152
- // the SOCKS/HTTP proxy that pi (ChatGPT Codex OAuth) may need:
153
- // 1) the proxy bandwidth often chokes on multi-megabyte PUTs to TOS
154
- // 2) routing China-mainland Volcano APIs through an overseas proxy is slower / unreliable
155
- const VOLCANO_NO_PROXY_HOSTS = ['.volces.com', '.volcengineapi.com', 'openspeech.bytedance.com']
156
-
157
- function systemProxyUrl(): string | null {
158
- if (process.platform !== 'darwin') return null
159
- try {
160
- const out = spawnSync('scutil', ['--proxy'], { encoding: 'utf8', timeout: 3000 })
161
- if (out.status !== 0 || !out.stdout) return null
162
- const get = (k: string) => out.stdout.match(new RegExp(`\\b${k}\\s*:\\s*(\\S+)`))?.[1]
163
- if (get('HTTPSEnable') === '1' && get('HTTPSProxy') && get('HTTPSPort')) return `http://${get('HTTPSProxy')}:${get('HTTPSPort')}`
164
- if (get('HTTPEnable') === '1' && get('HTTPProxy') && get('HTTPPort')) return `http://${get('HTTPProxy')}:${get('HTTPPort')}`
165
- return null
166
- } catch { return null }
167
- }
168
-
169
- type Settings = Record<string, string>
170
-
171
- /** This run's settings: config.json, overridden by the inherited environment. */
172
- function readSettings(file: Record<string, unknown>): Settings {
173
- const settings: Settings = {}
174
- for (const key of ENV_KEYS) {
175
- const inherited = process.env[key]
176
- if (inherited !== undefined) settings[key] = inherited
177
- else if (typeof file[key] === 'string') settings[key] = file[key] as string
178
- }
179
- return settings
180
- }
181
-
182
- /**
183
- * Proxy variables, resolved from settings or the macOS system proxy. Returned as
184
- * a map instead of being pushed onto process.env alone because Bun does not hand
185
- * a child the variables this process added after startup — every spawn site
186
- * passes them explicitly (covered by summary.test.ts).
187
- */
188
- function proxyEnv(s: Settings): Record<string, string> {
189
- const url = s.https_proxy || s.HTTPS_PROXY || s.http_proxy || s.HTTP_PROXY || s.all_proxy || s.ALL_PROXY
190
- || (s.LOCAL_PROXY_HOST && s.LOCAL_PROXY_PORT ? `http://${s.LOCAL_PROXY_HOST}:${s.LOCAL_PROXY_PORT}` : '')
191
- || systemProxyUrl()
192
- if (!url) return {}
193
- const env: Record<string, string> = {}
194
- for (const key of ['http_proxy', 'https_proxy', 'all_proxy', 'HTTP_PROXY', 'HTTPS_PROXY', 'ALL_PROXY']) env[key] = s[key] || url
195
- const base = s.LOCAL_NO_PROXY || s.no_proxy || s.NO_PROXY || 'localhost,127.0.0.1,::1'
196
- const bypass = [...new Set([...base.split(',').map(v => v.trim()).filter(Boolean), ...VOLCANO_NO_PROXY_HOSTS])].join(',')
197
- env.no_proxy = bypass
198
- env.NO_PROXY = bypass
199
- return env
200
- }
201
-
202
- function volcanoFrom(s: Settings): VolcanoConfig | null {
203
- const apiKey = s.VOLCANO_ASR_KEY || ''
204
- const tosAccess = s.VOLCANO_TOS_ACCESS_KEY
205
- const tosSecret = s.VOLCANO_TOS_SECRET_KEY
206
- const bucket = s.VOLCANO_TOS_BUCKET
207
- if (!apiKey || !tosAccess || !tosSecret || !bucket) return null
208
- const region = s.VOLCANO_TOS_REGION || 'cn-guangzhou'
209
- const endpoint = s.VOLCANO_TOS_ENDPOINT || `tos-s3-${region}.volces.com`
210
- const keep = ['1', 'true', 'yes'].includes((s.VOLCANO_TOS_KEEP || '0').toLowerCase())
211
- return {
212
- apiKey,
213
- resourceId: s.VOLCANO_ASR_RESOURCE_ID || 'volc.seedasr.auc',
214
- language: s.VOLCANO_ASR_LANGUAGE || undefined,
215
- tos: { endpoint, region, bucket, accessKey: tosAccess, secretKey: tosSecret, keep },
216
- }
217
- }
218
-
219
- function volcanoAuthHeaders(volc: VolcanoConfig, taskId: string, includeSequence: boolean): Record<string, string> {
220
- const base: Record<string, string> = {
221
- 'X-Api-Resource-Id': volc.resourceId,
222
- 'X-Api-Request-Id': taskId,
223
- 'Content-Type': 'application/json',
224
- }
225
- if (includeSequence) base['X-Api-Sequence'] = '-1'
226
- base['X-Api-Key'] = volc.apiKey
227
- return base
228
- }
229
-
230
- function settingNumber(s: Settings, key: string, fallback: number): number {
231
- const raw = s[key]
232
- const value = raw === undefined || raw === '' ? fallback : Number(raw)
233
- if (!Number.isFinite(value) || value < 0) throw new Error(`Invalid ${key}: expected a non-negative number, got '${raw}'`)
234
- return value
235
- }
236
-
237
- let configCache: Config | null = null
238
-
239
- function getConfig(): Config {
240
- if (configCache) return configCache
241
- const file = loadConfigJson()
242
- const s = readSettings(file)
243
- const proxy = proxyEnv(s)
244
- // vn's own fetch (the ChatGPT OAuth flow) reads the proxy from the process
245
- // environment, so the derived values have to land there as well.
246
- for (const [key, value] of Object.entries(proxy)) process.env[key] = value
247
- // Passed to every child: the proxy, plus the credentials and config dir that
248
- // pi — not vn — resolves for itself.
249
- const childEnv = { ...proxy }
250
- for (const key of ['PI_CODING_AGENT_DIR', 'OPENAI_API_KEY', 'DEEPSEEK_API_KEY']) if (s[key]) childEnv[key] = s[key]!
251
-
252
- const deviceVolume = s.VOICENOTE_DEVICE_VOLUME || 'VTR6500'
253
- const workspace = expandHome(s.VOICENOTE_WORKSPACE || '~/Documents/meetings')
254
- // pi keeps credentials in its config dir, which PI_CODING_AGENT_DIR relocates.
255
- // Point it at a voicenote-owned directory to get an auth.json that only the
256
- // pipeline reads and refreshes: an interactive pi session rewrites its own
257
- // auth.json wholesale on exit and has already dropped entries that way.
258
- const piAgentDir = expandHome(s.PI_CODING_AGENT_DIR || join(os.homedir(), '.pi', 'agent'))
259
- configCache = Object.freeze({
260
- recordDir: expandHome(s.VOICENOTE_RECORD_DIR || `/Volumes/${deviceVolume}/RECORD`),
261
- workspace,
262
- minBytes: settingNumber(s, 'VOICENOTE_MIN_BYTES', 100000),
263
- minDurationSeconds: settingNumber(s, 'VOICENOTE_MIN_DURATION_SECONDS', 60),
264
- // Only recordings from the last N hours are picked up (0 = no limit), so a
265
- // fresh install doesn't drain the recorder's entire history.
266
- maxAgeHours: settingNumber(s, 'VOICENOTE_MAX_AGE_HOURS', 48),
267
- volcano: volcanoFrom(s),
268
- speakers: normalizeSpeakers(file.speakers ?? DEFAULT_SPEAKERS),
269
- // ffprobe is the only ffmpeg-suite binary the pipeline uses (duration
270
- // detection); a configurable path lets the GUI point at its bundled copy.
271
- ffprobeBin: expandHome(s.VOICENOTE_FFPROBE_BIN || 'ffprobe'),
272
- pi: {
273
- bin: expandHome(s.VOICENOTE_PI_BIN || 'pi'),
274
- cli: s.VOICENOTE_PI_CLI ? expandHome(s.VOICENOTE_PI_CLI) : null,
275
- // pi's --model accepts "provider/id" (e.g. openai-codex/gpt-5.6-sol), so
276
- // this one setting pins both. Null = whatever pi is configured to use.
277
- model: (s.VOICENOTE_PI_MODEL || '').trim() || null,
278
- thinking: s.VOICENOTE_PI_THINKING || 'high',
279
- // Default ON: let the summary model read/grep prior notes for cross-reference
280
- // consistency. Set VOICENOTE_PI_SUMMARY_TOOLS='' to disable.
281
- tools: s.VOICENOTE_PI_SUMMARY_TOOLS === undefined ? 'read,grep' : s.VOICENOTE_PI_SUMMARY_TOOLS.trim(),
282
- // Directory the summary model may read/grep. The published default must not
283
- // reach outside the configured workspace.
284
- contextDir: expandHome(s.VOICENOTE_CONTEXT_DIR || workspace),
285
- retries: Math.max(1, Math.floor(settingNumber(s, 'VOICENOTE_PI_RETRIES', 3))),
286
- authPath: join(piAgentDir, 'auth.json'),
287
- },
288
- childEnv,
289
- })
290
- return configCache
291
- }
292
-
293
- // ────────────────────────────────────────────────────────────────────────────
294
- // Config files (~/.config/voicenote)
295
- // ────────────────────────────────────────────────────────────────────────────
296
-
297
- const DEFAULT_SPEAKERS: SpeakersConfig = { self: { name: null, aliases: [] }, known: [] }
298
-
299
- function normalizeSpeakers(data: unknown): SpeakersConfig {
300
- const raw = (data && typeof data === 'object') ? data as Partial<SpeakersConfig> : {}
301
- return {
302
- self: {
303
- name: typeof raw.self?.name === 'string' ? raw.self.name : null,
304
- aliases: Array.isArray(raw.self?.aliases) ? raw.self!.aliases.filter((a): a is string => typeof a === 'string') : [],
305
- },
306
- known: Array.isArray(raw.known)
307
- ? raw.known
308
- .filter((k): k is SpeakerKnown => !!k && typeof k === 'object' && typeof (k as SpeakerKnown).name === 'string')
309
- .map(k => ({ name: k.name, aliases: Array.isArray(k.aliases) ? k.aliases.filter((a): a is string => typeof a === 'string') : [], relationship: k.relationship ?? null }))
310
- : [],
311
- }
312
- }
313
-
314
- function loadConfigJson(): Record<string, unknown> {
315
- if (!existsSync(CONFIG_ENV_PATH)) return {}
316
- let value: unknown
317
- try { value = JSON.parse(readFileSync(CONFIG_ENV_PATH, 'utf8')) } catch (e: any) {
318
- throw new Error(`${CONFIG_ENV_PATH} is invalid JSON: ${e?.message || e}`)
319
- }
320
- if (!value || typeof value !== 'object' || Array.isArray(value)) throw new Error(`${CONFIG_ENV_PATH} must contain a JSON object`)
321
- return value as Record<string, unknown>
322
- }
323
-
324
- // ────────────────────────────────────────────────────────────────────────────
325
- // Misc helpers
326
- // ────────────────────────────────────────────────────────────────────────────
327
-
328
- function expandHome(path: string): string {
329
- return path.replace(/^(?:~|\$\{?HOME\}?)(?=\/|$)/, os.homedir())
330
- }
331
-
332
- function nowIso(): string { return new Date().toISOString() }
333
- function pad(n: number): string { return String(n).padStart(2, '0') }
334
-
335
- function dateParts(d: Date): { month: string; prefix: string } {
336
- const month = `${d.getFullYear()}-${pad(d.getMonth() + 1)}`
337
- // Local time in filenames uses HH-MM only (the recorder cannot produce two recordings within the same minute)
338
- const prefix = `${month}-${pad(d.getDate())}-${pad(d.getHours())}-${pad(d.getMinutes())}`
339
- return { month, prefix }
340
- }
341
-
342
- function safeSlug(text: string, maxLen = 48): string {
343
- const cleaned = (text || '').trim().replace(/[\\/:*?"<>|\n\r\t]+/g, '-').replace(/\s+/g, '-').replace(/^-+|-+$/g, '')
344
- return cleaned.slice(0, maxLen).replace(/-+$/g, '') || 'note'
345
- }
346
-
347
- function formatSeconds(seconds: number | null | undefined): string {
348
- const total = Math.max(0, Math.round(seconds || 0))
349
- const h = Math.floor(total / 3600)
350
- const m = Math.floor((total % 3600) / 60)
351
- const s = total % 60
352
- return h ? `${pad(h)}:${pad(m)}:${pad(s)}` : `${pad(m)}:${pad(s)}`
353
- }
354
-
355
- // ────────────────────────────────────────────────────────────────────────────
356
- // File state IO
357
- // ────────────────────────────────────────────────────────────────────────────
358
-
359
- const inboxPathFor = (config: Config) => join(config.workspace, '_inbox')
360
-
361
- async function ensureDirs(config: Config): Promise<void> {
362
- for (const dir of ['_state', '_index', '_audio', '_transcripts', '_metadata', '_inbox']) {
363
- await mkdir(join(config.workspace, dir), { recursive: true })
364
- }
365
- }
366
-
367
- // Write via tmp+rename so readers only ever see a complete file. Anything whose
368
- // mere existence is later treated as a signal MUST go through this: a half
369
- // written file that still parses is worse than no file at all.
370
- async function writeFileAtomic(path: string, body: string): Promise<void> {
371
- await mkdir(dirname(path), { recursive: true })
372
- const tmp = `${path}.tmp`
373
- await writeFile(tmp, body, 'utf8')
374
- await rename(tmp, path)
375
- }
376
-
377
- const writeJson = (path: string, data: any) => writeFileAtomic(path, JSON.stringify(data, null, 2))
378
-
379
- async function appendJsonl(path: string, data: any): Promise<void> {
380
- await mkdir(dirname(path), { recursive: true })
381
- await appendFile(path, JSON.stringify(data) + '\n', 'utf8')
382
- }
383
-
384
- const RAW_TRANSCRIPT_MARKER = '## Raw transcript (no lossy cleanup)\n\n'
385
- const RAW_TRANSCRIPT_MARKER_LEGACY = '## 原始 transcript(不做 lossy 清洗)\n\n' // pre-0.18 files on disk
386
-
387
-
388
-
389
- // ────────────────────────────────────────────────────────────────────────────
390
- // Logging (rolling daily log)
391
- // ────────────────────────────────────────────────────────────────────────────
392
-
393
- function dailyLogPath(): string {
394
- const d = new Date()
395
- return join(LOG_DIR, `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}.log`)
396
- }
397
-
398
- // Best-effort side effects (logging, idle-state) must not crash the pipeline, but
399
- // failures should still be observable. Write directly to stderr (not console.error,
400
- // which wireDailyLog wraps and would recurse into the same failing file) once per tag.
401
- const sideEffectWarned = new Set<string>()
402
- function warnSideEffect(where: string, e: unknown): void {
403
- if (sideEffectWarned.has(where)) return
404
- sideEffectWarned.add(where)
405
- process.stderr.write(`[voicenote] non-fatal: ${where} failed: ${e instanceof Error ? e.message : String(e)}\n`)
406
- }
407
-
408
- let logWired = false
409
- function wireDailyLog(): void {
410
- if (logWired) return
411
- logWired = true
412
- try { mkdirSync(LOG_DIR, { recursive: true }) } catch (e) { warnSideEffect('log dir mkdir', e) }
413
- const path = dailyLogPath()
414
- const append = (level: 'INFO' | 'ERROR', args: any[]) => {
415
- const line = args.map(a => (typeof a === 'string' ? a : JSON.stringify(a))).join(' ')
416
- const stamped = `${nowIso()} [${level}] ${line}\n`
417
- try { appendFileSync(path, stamped, 'utf8') } catch (e) { warnSideEffect('daily log append', e) }
418
- }
419
- const origLog = console.log.bind(console)
420
- const origErr = console.error.bind(console)
421
- console.log = (...a: any[]) => { append('INFO', a); origLog(...a) }
422
- console.error = (...a: any[]) => { append('ERROR', a); origErr(...a) }
423
- }
424
-
425
- function formatBytes(bytes: number): string {
426
- if (!Number.isFinite(bytes)) return 'unknown size'
427
- const units = ['B', 'KB', 'MB', 'GB']
428
- let value = bytes
429
- let unit = 0
430
- while (value >= 1024 && unit < units.length - 1) { value /= 1024; unit++ }
431
- return `${value.toFixed(unit === 0 ? 0 : 1)} ${units[unit]}`
432
- }
433
-
434
- function formatElapsed(ms: number): string {
435
- const total = Math.max(0, Math.round(ms / 1000))
436
- const h = Math.floor(total / 3600)
437
- const m = Math.floor((total % 3600) / 60)
438
- const s = total % 60
439
- if (h) return `${h}h ${m}m ${s}s`
440
- if (m) return `${m}m ${s}s`
441
- return `${s}s`
442
- }
443
-
444
- function progressStep(step: number, total: number, title: string, detail?: string): void {
445
- console.log(`▶ Step ${step}/${total}: ${title}${detail ? ` — ${detail}` : ''}`)
446
- // Single hook for live progress: the dashboard shows the same string the log
447
- // does, instead of regex-guessing the step from log text.
448
- reportStep(title)
449
- }
450
-
451
- async function withHeartbeat<T>(label: string, work: () => Promise<T>, heartbeatSeconds = 60): Promise<T> {
452
- const started = Date.now()
453
- const timer = setInterval(() => {
454
- console.log(`… Still working: ${label} (${formatElapsed(Date.now() - started)} elapsed)`)
455
- }, Math.max(10, heartbeatSeconds) * 1000)
456
- ;(timer as any).unref?.()
457
- try {
458
- const result = await work()
459
- console.log(`✓ Done: ${label} (${formatElapsed(Date.now() - started)})`)
460
- return result
461
- } catch (e) {
462
- console.error(`✗ Failed: ${label} after ${formatElapsed(Date.now() - started)}`)
463
- throw e
464
- } finally {
465
- clearInterval(timer)
466
- }
467
- }
468
-
469
- function shouldLogIdleStatus(key: string, intervalMs = 30 * 60 * 1000): boolean {
470
- const path = join(LOG_DIR, 'idle-status.json')
471
- const now = Date.now()
472
- let prev: any = null
473
- try { prev = JSON.parse(readFileSync(path, 'utf8')) } catch {}
474
- const should = prev?.key !== key || now - Number(prev?.at || 0) >= intervalMs
475
- if (should) {
476
- try {
477
- mkdirSync(LOG_DIR, { recursive: true })
478
- writeFileSync(path, JSON.stringify({ key, at: now, iso: nowIso() }, null, 2) + '\n', 'utf8')
479
- } catch (e) { warnSideEffect('idle-status write', e) }
480
- }
481
- return should
482
- }
483
-
484
- type RunMode = 'notes' | 'transcript'
485
- function normalizeRunMode(opts: any): RunMode {
486
- const raw = String(opts.mode || 'notes').toLowerCase()
487
- if (raw === 'note') return 'notes'
488
- if (raw === 'notes' || raw === 'transcript') return raw
489
- throw new Error(`Invalid --mode "${raw}". Use: notes|transcript`)
490
- }
491
-
492
- // ────────────────────────────────────────────────────────────────────────────
493
- // Cross-process lock
494
- // ────────────────────────────────────────────────────────────────────────────
495
-
496
- // Single-instance mutual exclusion via an OS advisory lock (flock) held on an open
497
- // fd. The kernel releases it automatically when the process exits — including
498
- // SIGKILL/crash — so there is NO pid / mtime / heartbeat / stale-steal logic to
499
- // race on. flock is loaded from libSystem, so it is macOS-only; every other
500
- // platform uses the pid+timestamp lockfile below.
501
- const flockFn = (() => {
502
- try {
503
- const lib = dlopen(`libSystem.${suffix}`, { flock: { args: [FFIType.i32, FFIType.i32], returns: FFIType.i32 } })
504
- return lib.symbols.flock as (fd: number, op: number) => number
505
- } catch { return null }
506
- })()
507
- const FLOCK_EX_NB = 2 | 4 // LOCK_EX | LOCK_NB
508
- const FLOCK_UN = 8
509
-
510
- // Lockfile used wherever flock is not available (Windows, Linux). A pid+timestamp
511
- // file, created atomically with 'wx'. We only reclaim an existing lock when its
512
- // owner pid is dead OR the lock is stale (older than STALE_MS). The holder refreshes its timestamp every 5 minutes
513
- // (heartbeat below), so a legitimately long RUNNING job — ASR on a multi-hour
514
- // recording — never looks stale. The staleness escape exists for the pid-reuse
515
- // false positive (owner died, an unrelated process now has its pid, the aliveness
516
- // probe lies); its known cost: a machine asleep >30min can lose the lock on wake
517
- // (timers don't fire while asleep), so the heartbeat verifies ownership before
518
- // each refresh and, if the lock was reclaimed, stops touching it and warns — the
519
- // old run finishes unprotected rather than corrupting the new holder's record.
520
- // Task Scheduler's IgnoreNew already blocks the common 60s overlap; this only has
521
- // to cover a manual `vn run` racing the scheduled one. The tiny create/reclaim
522
- // window is acceptable: its failure mode is conservatively skipping one run (same
523
- // as mac when flock is already held).
524
- async function acquireRunLockFile(): Promise<{ release: () => Promise<void> } | null> {
525
- await mkdir(dirname(LOCK_PATH), { recursive: true })
526
- const STALE_MS = 30 * 60 * 1000
527
- const tryCreate = (): number | null => {
528
- try { return openSync(LOCK_PATH, 'wx') }
529
- catch (e: any) { if (e?.code === 'EEXIST') return null; throw e }
530
- }
531
- let fd = tryCreate()
532
- if (fd === null) {
533
- let reclaim = false
534
- try {
535
- const data = JSON.parse(readFileSync(LOCK_PATH, 'utf8'))
536
- const pid = Number(data.pid), ts = Number(data.ts)
537
- const alive = pidAlive(pid)
538
- const fresh = Number.isFinite(ts) && (Date.now() - ts) < STALE_MS
539
- reclaim = !alive || !fresh
540
- } catch { reclaim = true } // unreadable/corrupt lock -> reclaim
541
- if (!reclaim) return null // another live run holds it
542
- try { unlinkSync(LOCK_PATH) } catch {}
543
- fd = tryCreate()
544
- if (fd === null) return null // someone grabbed it in the gap
545
- }
546
- writeFileSync(fd, JSON.stringify({ pid: process.pid, ts: Date.now() }))
547
- closeSync(fd)
548
- // Tri-state ownership (parse logic + its 'unknown'-on-read-failure invariant
549
- // are pure + tested in runLock.ts). A transient read failure must NOT be
550
- // treated as loss of ownership: that would kill the heartbeat and hand the
551
- // lock away over a momentary glitch — the overlap the heartbeat prevents.
552
- const lockOwnership = () => {
553
- let raw: string | null
554
- try { raw = readFileSync(LOCK_PATH, 'utf8') } catch { raw = null }
555
- return parseLockOwner(raw, process.pid)
556
- }
557
- // Heartbeat: keep ts fresh while we hold the lock; verify ownership first
558
- // (see header comment — the lock can be reclaimed after a long sleep).
559
- // The refresh writes a temp file and renames it into place: a plain
560
- // truncate+write would open a window where a concurrent acquire reads
561
- // empty/partial JSON, treats the lock as corrupt, and reclaims a LIVE lock.
562
- const heartbeat = setInterval(() => {
563
- const owner = lockOwnership()
564
- if (owner === 'reclaimed') {
565
- clearInterval(heartbeat)
566
- console.error('Run lock was reclaimed by another process (machine slept >30min?); this run continues but is no longer protected against overlap.')
567
- return
568
- }
569
- if (owner === 'unknown') { warnSideEffect('run lock heartbeat read', new Error('lock unreadable this tick; will retry')); return }
570
- try {
571
- const tmp = `${LOCK_PATH}.hb-${process.pid}`
572
- writeFileSync(tmp, JSON.stringify({ pid: process.pid, ts: Date.now() }))
573
- renameSync(tmp, LOCK_PATH) // atomic replace, also on Windows
574
- } catch (e) { warnSideEffect('run lock heartbeat', e) }
575
- }, 5 * 60 * 1000)
576
- ;(heartbeat as any).unref?.()
577
- let released = false
578
- const release = async () => {
579
- if (released) return
580
- released = true
581
- clearInterval(heartbeat)
582
- // Only remove the lock when it is provably still OURS. Not 'reclaimed'
583
- // (that's the new holder's lock) and not 'unknown' either: a transient
584
- // read failure could be a reclaimer mid-swap, and the heartbeat does NOT
585
- // rewrite on 'unknown', so unlinking here would leave a real vacuum until
586
- // STALE_MS. Leaking our own lock on a rare transient failure is the lesser
587
- // evil — it self-heals after STALE_MS via the staleness check.
588
- try { if (lockOwnership() === 'mine') unlinkSync(LOCK_PATH) } catch {}
589
- }
590
- process.once('exit', () => { void release() })
591
- process.once('SIGINT', () => { void release(); process.exit(130) })
592
- process.once('SIGTERM', () => { void release(); process.exit(143) })
593
- return { release }
594
- }
595
-
596
- async function acquireRunLock(): Promise<{ release: () => Promise<void> } | null> {
597
- if (!flockFn) return acquireRunLockFile()
598
- await mkdir(dirname(LOCK_PATH), { recursive: true })
599
- // The lock is a regular file we keep open. Builds ≤ 0.15.2 used a *directory*
600
- // here, held purely by its existence, with no pid or refreshed mtime inside — so
601
- // a leftover legacy dir carries NO reliable signal about whether an old `vn run`
602
- // still holds it. Rather than guess (and risk deleting a live lock → concurrent
603
- // double-processing), refuse to auto-reclaim it: warn and skip. Normal upgrades
604
- // don't hit this (≤ 0.15.2 removes its own dir lock on SIGTERM/exit); it only
605
- // appears after a hard crash of an old build, where a one-time manual cleanup is
606
- // the safe move.
607
- let fd: number
608
- try { fd = openSync(LOCK_PATH, 'w') }
609
- catch (e: any) {
610
- if (e?.code !== 'EISDIR') throw e
611
- console.error(`Found a legacy (≤ 0.15.2) lock directory at ${LOCK_PATH}; it carries no liveness info and can't be auto-reclaimed safely. If no 'vn run' is active, remove it once: rm -rf "${LOCK_PATH}" — skipping this run.`)
612
- return null
613
- }
614
- if (flockFn(fd, FLOCK_EX_NB) !== 0) { closeSync(fd); return null } // another run holds it
615
- let released = false
616
- const release = async () => {
617
- if (released) return
618
- released = true
619
- try { flockFn(fd, FLOCK_UN) } catch {}
620
- try { closeSync(fd) } catch {}
621
- }
622
- process.once('exit', () => { void release() })
623
- process.once('SIGINT', () => { void release(); process.exit(130) })
624
- process.once('SIGTERM', () => { void release(); process.exit(143) })
625
- return { release }
626
- }
627
-
628
- // ────────────────────────────────────────────────────────────────────────────
629
- // Recording scan
630
- // ────────────────────────────────────────────────────────────────────────────
631
-
632
- function parseRecordedAt(path: string): Date {
633
- const stem = basename(path, extname(path))
634
- const m = stem.match(/(20\d{12})/)
635
- if (m?.[1]) {
636
- const s = m[1]
637
- return new Date(Number(s.slice(0, 4)), Number(s.slice(4, 6)) - 1, Number(s.slice(6, 8)), Number(s.slice(8, 10)), Number(s.slice(10, 12)), Number(s.slice(12, 14)))
638
- }
639
- return new Date()
640
- }
641
-
642
- async function sha256File(path: string): Promise<string> {
643
- const h = createHash('sha256')
644
- const reader = Bun.file(path).stream().getReader()
645
- while (true) {
646
- const { done, value } = await reader.read()
647
- if (done) break
648
- h.update(value)
649
- }
650
- return h.digest('hex')
651
- }
652
-
653
- function sourceIdFor(path: string, size: number, mtimeMs: number, digest: string, imported = false): string {
654
- // Manual imports are content-addressed: dropping the same audio again must
655
- // find its existing job even after the temporary inbox copy was removed.
656
- return imported ? `import:${digest}` : createHash('sha256').update(`${path}|${size}|${Math.floor(mtimeMs / 1000)}|${digest}`).digest('hex')
657
- }
658
-
659
- function runCommand(command: string, args: string[], timeoutMs = 20000): Promise<{ stdout: string; stderr: string; code: number }> {
660
- return new Promise((res) => {
661
- const child = spawn(command, args, { stdio: ['ignore', 'pipe', 'pipe'], windowsHide: true })
662
- let stdout = '', stderr = ''
663
- const timer = setTimeout(() => child.kill('SIGKILL'), timeoutMs)
664
- child.stdout.on('data', d => stdout += String(d))
665
- child.stderr.on('data', d => stderr += String(d))
666
- child.on('close', code => { clearTimeout(timer); res({ stdout, stderr, code: code ?? 1 }) })
667
- child.on('error', err => { clearTimeout(timer); res({ stdout, stderr: String(err), code: 1 }) })
668
- })
669
- }
670
-
671
- // Cross-platform "reveal in file manager / open URL in default app".
672
- // macOS `open`, Linux `xdg-open`, Windows `start` (a cmd builtin, so via `cmd /c`;
673
- // the empty "" is start's title arg so a quoted path/URL isn't swallowed as title).
674
- function openPath(target: string, timeoutMs = 5000): Promise<{ stdout: string; stderr: string; code: number }> {
675
- if (IS_WINDOWS) return runCommand('cmd', ['/c', 'start', '', target], timeoutMs)
676
- if (IS_MAC) return runCommand('open', [target], timeoutMs)
677
- return runCommand('xdg-open', [target], timeoutMs)
678
- }
679
-
680
- // Cross-platform replacement for `tail -n N [-F] files`. Windows ships no `tail`,
681
- // so even a non-follow `vn log` would break; a pure-JS implementation also drops a
682
- // process dependency on mac/Linux. Follow mode polls appended bytes every second.
683
- async function tailFiles(files: string[], lines: number, follow: boolean): Promise<void> {
684
- const header = files.length > 1
685
- const lastLines = (text: string, n: number) => {
686
- const arr = text.split('\n')
687
- if (arr.length && arr[arr.length - 1] === '') arr.pop()
688
- return arr.slice(-n).join('\n')
689
- }
690
- const sizes = new Map<string, number>()
691
- for (const f of files) {
692
- const text = await readFile(f, 'utf8').catch(() => '')
693
- if (header) process.stdout.write(`==> ${f} <==\n`)
694
- const tail = lastLines(text, lines)
695
- if (tail) process.stdout.write(tail + '\n')
696
- sizes.set(f, Buffer.byteLength(text))
697
- }
698
- if (!follow) return
699
- await new Promise<void>((resolve) => {
700
- let stop = false
701
- process.once('SIGINT', () => { stop = true; resolve() })
702
- const poll = () => {
703
- if (stop) return
704
- for (const f of files) {
705
- try {
706
- const size = statSync(f).size
707
- const prev = sizes.get(f) ?? 0
708
- if (size > prev) {
709
- const fd = openSync(f, 'r')
710
- try {
711
- const buf = Buffer.alloc(size - prev)
712
- readSync(fd, buf, 0, buf.length, prev)
713
- if (header) process.stdout.write(`==> ${f} <==\n`)
714
- process.stdout.write(buf.toString('utf8'))
715
- } finally { closeSync(fd) }
716
- sizes.set(f, size)
717
- } else if (size < prev) {
718
- sizes.set(f, size) // rotated/truncated
719
- }
720
- } catch (e) { warnSideEffect(`follow ${f}`, e) }
721
- }
722
- if (!stop) setTimeout(poll, 1000)
723
- }
724
- setTimeout(poll, 1000)
725
- })
726
- }
727
-
728
- async function ffprobeDuration(config: Config, path: string): Promise<number | null> {
729
- const result = await runCommand(config.ffprobeBin, ['-v', 'error', '-show_entries', 'format=duration', '-of', 'default=noprint_wrappers=1:nokey=1', path])
730
- if (result.code !== 0) return null
731
- const v = Number(result.stdout.trim())
732
- return Number.isFinite(v) ? v : null
733
- }
734
-
735
- function isCandidateFile(path: string): boolean {
736
- const name = basename(path)
737
- if (name.startsWith('._') || name.startsWith('.')) return false
738
- if (!AUDIO_EXTENSIONS.has(extname(path).toLowerCase())) return false
739
- const parts = path.split(/[/\\]/)
740
- if (parts.includes('.Spotlight-V100') || parts.includes('.fseventsd') || parts.includes('System Volume Information')) return false
741
- return true
742
- }
743
-
744
- /**
745
- * `complete` is false when any part of the listing was lost — the glob threw, or
746
- * a file we had just seen could not be read. It gates pruning: "not in the scan"
747
- * only means "gone from the recorder" if the scan actually saw everything, and
748
- * treating a half-read device as authoritative would delete live queue entries
749
- * along with their retry counters.
750
- */
751
- async function toRecording(config: Config, file: string, imported = false): Promise<Recording> {
752
- const st = await stat(file)
753
- const contentHash = await sha256File(file)
754
- return {
755
- sourcePath: file,
756
- sizeBytes: st.size,
757
- modifiedAt: st.mtime.toISOString(),
758
- durationSeconds: await ffprobeDuration(config, file),
759
- sourceId: sourceIdFor(file, st.size, st.mtimeMs, contentHash, imported),
760
- contentHash,
761
- recordedAt: parseRecordedAt(file),
762
- imported,
763
- }
764
- }
765
-
766
- async function scanRecordings(config: Config): Promise<{ recordings: Recording[]; complete: boolean }> {
767
- const recordings: Recording[] = []
768
- const recorderPresent = existsSync(config.recordDir)
769
- let complete = recorderPresent
770
- const roots = [
771
- ...(recorderPresent ? [{ dir: config.recordDir, imported: false }] : []),
772
- ...(existsSync(inboxPathFor(config)) ? [{ dir: inboxPathFor(config), imported: true }] : []),
773
- ]
774
- for (const root of roots) {
775
- try {
776
- for await (const file of new Bun.Glob('**/*').scan({ cwd: root.dir, absolute: true, dot: true })) {
777
- if (!isCandidateFile(file)) continue
778
- const st = await stat(file).catch(() => null)
779
- // Listed a moment ago but unreadable now: the device is going away, or
780
- // this file is. Either way the listing is no longer trustworthy.
781
- if (!st) { complete = false; continue }
782
- if (!st.isFile()) continue
783
- try {
784
- recordings.push(await toRecording(config, file, root.imported))
785
- } catch (e) { complete = false; warnSideEffect(`read ${basename(file)} during scan`, e) }
786
- }
787
- } catch (e) {
788
- complete = false
789
- warnSideEffect(`scan ${root.imported ? 'import inbox' : 'recorder'}`, e)
790
- }
791
- }
792
- // Explicit imports go first; each group remains oldest-first.
793
- recordings.sort((a, b) => Number(b.imported) - Number(a.imported) || a.recordedAt.getTime() - b.recordedAt.getTime())
794
- return { recordings, complete }
795
- }
796
-
797
- // ────────────────────────────────────────────────────────────────────────────
798
- // File path planning
799
- // ────────────────────────────────────────────────────────────────────────────
800
-
801
- /**
802
- * The one place the output layout is written down. A job starts out untitled
803
- * (timestamp only) and moves to its titled names once the summary produces a
804
- * title; pass `title` — including a null/empty one — for the titled form.
805
- */
806
- function layout(config: Config, rec: Recording, title?: string | null): LocalFiles {
807
- const { month, prefix } = dateParts(rec.recordedAt)
808
- const untitled = title === undefined
809
- const base = untitled ? prefix : `${prefix}-${safeSlug(title || 'note')}`
810
- return {
811
- audio: join(config.workspace, '_audio', month, `${base}-original${extname(rec.sourcePath).toLowerCase()}`),
812
- transcript: join(config.workspace, '_transcripts', month, `${base}-transcript.md`),
813
- notes: join(config.workspace, month, untitled ? `${base}-note.md` : `${base}.md`),
814
- metadata: join(config.workspace, '_metadata', month, `${base}-metadata.json`),
815
- }
816
- }
817
-
818
- // Resume on the evidence, not on a state label: if the transcript is on disk,
819
- // re-running ASR is money spent for nothing. Keying this off `notes_failed`
820
- // instead meant `vn forget` (which drops the record) silently re-paid for ASR,
821
- // even though the transcript was still sitting there.
822
- function resumableTranscriptFiles(config: Config, rec: Recording, store: StateFile, mode: RunMode, force: boolean): LocalFiles | null {
823
- if (force || mode !== 'notes') return null
824
- // Paths recorded by an earlier attempt win: that attempt may already have
825
- // moved its outputs to titled names.
826
- const fallback = layout(config, rec)
827
- const recorded = store.jobs[rec.sourceId]?.paths || {}
828
- const files = Object.fromEntries(
829
- Object.entries(fallback).map(([key, path]) => [key, typeof recorded[key] === 'string' ? recorded[key] : path]),
830
- ) as LocalFiles
831
- return existsSync(files.transcript) ? files : null
832
- }
833
-
834
- async function readSavedTranscript(path: string): Promise<string> {
835
- const markdown = await readFile(path, 'utf8')
836
- const marker = [RAW_TRANSCRIPT_MARKER, RAW_TRANSCRIPT_MARKER_LEGACY].find(m => markdown.includes(m))
837
- if (!marker) throw new Error(`Cannot resume summary: saved transcript is missing raw transcript marker: ${path}`)
838
- const transcript = markdown.slice(markdown.indexOf(marker) + marker.length).trim()
839
- if (!transcript) throw new Error(`Cannot resume summary: saved transcript is empty: ${path}`)
840
- return transcript
841
- }
842
-
843
- async function removeFailedSummaryStub(path: string): Promise<void> {
844
- if (!existsSync(path)) return
845
- try {
846
- const body = await readFile(path, 'utf8')
847
- if (body.startsWith('# Pending summary: ') || body.startsWith('# 待补纪要:')) await unlink(path)
848
- } catch (e) { warnSideEffect(`remove failed-summary stub ${path}`, e) }
849
- }
850
-
851
- /**
852
- * Move a job's existing outputs onto their titled paths. Audio and the
853
- * transcript written before the summary ran move together — they used to be
854
- * renamed in two different places, and the one left behind became an orphan.
855
- * Notes and metadata are rewritten by the caller, so their stale copies from a
856
- * failed attempt are dropped instead of moved.
857
- */
858
- async function promoteOutputs(from: LocalFiles, to: LocalFiles): Promise<void> {
859
- for (const key of ['audio', 'transcript'] as const) {
860
- if (from[key] === to[key] || !existsSync(from[key])) continue
861
- await mkdir(dirname(to[key]), { recursive: true })
862
- if (existsSync(to[key])) await unlink(to[key])
863
- await rename(from[key], to[key])
864
- }
865
- if (from.metadata !== to.metadata && existsSync(from.metadata)) {
866
- await unlink(from.metadata).catch(e => warnSideEffect(`remove orphaned metadata ${from.metadata}`, e))
867
- }
868
- }
869
-
870
- // ───────────────────────────────────────────────────────────────────────
871
- // Volcano (Doubao ASR + TOS upload)
872
- // ───────────────────────────────────────────────────────────────────────
873
-
874
- function volcanoContentTypeFromExt(ext: string): string {
875
- const e = ext.replace(/^\./, '').toLowerCase()
876
- switch (e) {
877
- case 'mp3': return 'audio/mpeg'
878
- case 'wav': return 'audio/wav'
879
- case 'm4a': return 'audio/mp4'
880
- case 'aac': return 'audio/aac'
881
- case 'ogg': return 'audio/ogg'
882
- case 'flac': return 'audio/flac'
883
- default: return 'application/octet-stream'
884
- }
885
- }
886
-
887
- async function volcanoSubmitTask(volc: VolcanoConfig, taskId: string, audioUrl: string, format: string): Promise<void> {
888
- const body = {
889
- user: { uid: 'voicenote' },
890
- audio: { url: audioUrl, format },
891
- request: {
892
- model_name: 'bigmodel',
893
- enable_itn: true,
894
- enable_punc: true,
895
- enable_ddc: true,
896
- enable_speaker_info: true,
897
- show_utterances: true,
898
- ...(volc.language ? { language: volc.language } : {}),
899
- },
900
- }
901
- const res = await fetch('https://openspeech.bytedance.com/api/v3/auc/bigmodel/submit', {
902
- method: 'POST',
903
- headers: volcanoAuthHeaders(volc, taskId, true),
904
- body: JSON.stringify(body),
905
- })
906
- const status = res.headers.get('X-Api-Status-Code') || ''
907
- const message = res.headers.get('X-Api-Message') || ''
908
- if (status !== '20000000') {
909
- const text = await res.text().catch(() => '')
910
- throw new Error(`Volcano submit failed: status=${status} message=${message} body=${text.slice(0, 500)}`)
911
- }
912
- }
913
-
914
- type VolcanoUtterance = {
915
- text?: string
916
- start_time?: number
917
- end_time?: number
918
- speaker_id?: number | string
919
- additions?: { speaker_id?: number | string; speaker?: string | number }
920
- }
921
-
922
- type VolcanoQueryResult = {
923
- status: string
924
- message: string
925
- result?: { text?: string; utterances?: VolcanoUtterance[] }
926
- audio_info?: { duration?: number }
927
- }
928
-
929
- async function volcanoQueryResult(volc: VolcanoConfig, taskId: string): Promise<VolcanoQueryResult> {
930
- const res = await fetch('https://openspeech.bytedance.com/api/v3/auc/bigmodel/query', {
931
- method: 'POST',
932
- headers: volcanoAuthHeaders(volc, taskId, false),
933
- body: '{}',
934
- })
935
- const status = res.headers.get('X-Api-Status-Code') || ''
936
- const message = res.headers.get('X-Api-Message') || ''
937
- const text = await res.text().catch(() => '')
938
- let parsed: any = null
939
- if (text) { try { parsed = JSON.parse(text) } catch { parsed = null } }
940
- return { status, message, result: parsed?.result, audio_info: parsed?.audio_info }
941
- }
942
-
943
- function volcanoSpeakerLabel(u: VolcanoUtterance): string {
944
- const id = u.speaker_id ?? u.additions?.speaker_id ?? u.additions?.speaker
945
- if (id == null || id === '') return 'Speaker A'
946
- const n = Number(id)
947
- if (Number.isFinite(n) && n >= 0 && n < 26) return `Speaker ${String.fromCharCode(65 + n)}`
948
- return `Speaker ${String(id)}`
949
- }
950
-
951
- function volcanoFormatTranscript(result: { text?: string; utterances?: VolcanoUtterance[] }): string {
952
- const utterances = result.utterances || []
953
- if (!utterances.length) return (result.text || '').trim()
954
- const lines = utterances
955
- .map(u => {
956
- const text = String(u.text || '').trim()
957
- if (!text) return ''
958
- const start = formatSeconds(Math.round((u.start_time || 0) / 1000))
959
- const end = formatSeconds(Math.round((u.end_time || 0) / 1000))
960
- return `[${start}-${end}] ${volcanoSpeakerLabel(u)}: ${text}`
961
- })
962
- .filter(Boolean)
963
- return lines.join('\n')
964
- }
965
-
966
- async function volcanoTranscribeAudio(volc: VolcanoConfig, audioPath: string, rec: Recording): Promise<string> {
967
- const ext = extname(audioPath).toLowerCase() || '.mp3'
968
- const format = ext.replace(/^\./, '')
969
- const contentType = volcanoContentTypeFromExt(ext)
970
- const { month } = dateParts(rec.recordedAt)
971
- const key = `voicenote/${month}/${rec.sourceId}-${Date.now()}${ext}`
972
- const object = tosObject(volc.tos, key)
973
- console.log(`Volcano: upload audio to TOS as ${key}`)
974
- await withHeartbeat('upload audio to TOS', () => object.write(Bun.file(audioPath), { type: contentType }), 30)
975
- let cleanedUp = false
976
- const cleanup = async () => {
977
- if (cleanedUp || volc.tos.keep) return
978
- cleanedUp = true
979
- await object.delete().catch(e => warnSideEffect(`delete TOS object ${key}`, e))
980
- }
981
- try {
982
- const audioUrl = object.presign({ method: 'GET', expiresIn: 6 * 3600 })
983
- const taskId = randomUUID()
984
- console.log(`Volcano: submit ASR task ${taskId} (resource=${volc.resourceId}, format=${format})`)
985
- await volcanoSubmitTask(volc, taskId, audioUrl, format)
986
- const started = Date.now()
987
- const expectedSeconds = rec.durationSeconds || 0
988
- const maxWaitMs = Math.max(20 * 60 * 1000, Math.ceil(expectedSeconds * 1000 * 1.5))
989
- let lastStatusLog = 0
990
- let lastStatus = ''
991
- // Tolerate transient failures while polling: by this point the audio is
992
- // uploaded and the ASR task is submitted (money spent) — one dropped
993
- // socket or an HTTP-level error (gateway 5xx returns no X-Api-Status-Code
994
- // header, so q.status comes back empty) must not fail the whole job and
995
- // trigger a full re-upload + re-submit on the next tick. Only give up
996
- // after many failures in a row; throws when the budget or deadline is hit.
997
- let queryFailures = 0
998
- const transientQueryFailure = (desc: string): void => {
999
- queryFailures++
1000
- if (queryFailures >= 10) throw new Error(`Volcano query failed ${queryFailures}x in a row: ${desc}`)
1001
- if (Date.now() - started > maxWaitMs) throw new Error(`Volcano: timeout after ${formatElapsed(Date.now() - started)} (last error: ${desc})`)
1002
- // console.error (not log) so wireDailyLog tags it [ERROR] and `vn errors`
1003
- // surfaces it — matching chatCompleteViaPi's transient-retry logging.
1004
- // A repeatedly-near-threshold ASR wobble is exactly what ops wants to see.
1005
- console.error(`… Volcano: transient query failure (attempt ${queryFailures}/10, will retry): ${desc}`)
1006
- }
1007
- for (;;) {
1008
- await new Promise(res => setTimeout(res, 8000))
1009
- let q: VolcanoQueryResult
1010
- try {
1011
- q = await volcanoQueryResult(volc, taskId)
1012
- } catch (e: any) {
1013
- transientQueryFailure(String(e?.message || e))
1014
- continue
1015
- }
1016
- if (!q.status) {
1017
- transientQueryFailure(`empty status header (HTTP-level error, body: ${q.message || 'none'})`)
1018
- continue
1019
- }
1020
- queryFailures = 0
1021
- if (q.status === '20000000' && q.result) {
1022
- console.log(`✓ Volcano: ASR done in ${formatElapsed(Date.now() - started)}; audio_duration=${q.audio_info?.duration ?? 'unknown'}ms`)
1023
- return volcanoFormatTranscript(q.result)
1024
- }
1025
- if (q.status === '20000001' || q.status === '20000002') {
1026
- if (q.status !== lastStatus || Date.now() - lastStatusLog > 60_000) {
1027
- const label = q.status === '20000002' ? 'queued' : 'processing'
1028
- console.log(`… Volcano: ${label} (status=${q.status}, ${formatElapsed(Date.now() - started)} elapsed)`)
1029
- lastStatusLog = Date.now()
1030
- lastStatus = q.status
1031
- }
1032
- if (Date.now() - started > maxWaitMs) throw new Error(`Volcano: timeout after ${formatElapsed(Date.now() - started)} (last status=${q.status})`)
1033
- continue
1034
- }
1035
- if (q.status === '20000003') throw new Error('Volcano: 20000003 silent audio (no speech detected)')
1036
- throw new Error(`Volcano query failed: status=${q.status} message=${q.message}`)
1037
- }
1038
- } finally {
1039
- await cleanup()
1040
- }
1041
- }
1042
-
1043
- async function transcribeAudio(config: Config, audioPath: string, rec: Recording): Promise<string> {
1044
- if (!config.volcano) throw new Error('Volcano ASR not configured. Set VOLCANO_ASR_KEY / VOLCANO_TOS_* in config.json.')
1045
- return volcanoTranscribeAudio(config.volcano, audioPath, rec)
1046
- }
1047
-
1048
-
1049
- function speakerContextBlock(speakers: SpeakersConfig): string {
1050
- const selfPart = speakers.self.name
1051
- ? `The user: ${speakers.self.name}${speakers.self.aliases.length ? ` (aliases: ${speakers.self.aliases.join(', ')})` : ''}`
1052
- : "The user's name is not configured."
1053
- const knownPart = speakers.known.length
1054
- ? speakers.known.map(k => `- ${k.name}${k.aliases?.length ? ` (aliases: ${k.aliases.join(', ')})` : ''}${k.relationship ? `, ${k.relationship}` : ''}`).join('\n')
1055
- : '(no other known speakers)'
1056
- return `Speaker context (use it to map Speaker A/B/C back to real names, but only when the evidence is solid):\n- ${selfPart}\n- Other known speakers:\n${knownPart}\n\nRules:\n- If the recording has a single speaker and the user's name is configured, treat Speaker A as the user.\n- In multi-speaker conversations, if a speaker is addressed by the user's name/alias, that speaker is the user.\n- In multi-speaker conversations, if a speaker is addressed by a known speaker's name/alias, that speaker is that known person.\n- Otherwise keep Speaker A/B/C as-is; never guess.`
1057
- }
1058
-
1059
-
1060
- function summaryMessages(config: Config, transcript: string, rec: Recording, localAudioPath: string): { role: 'system' | 'user'; content: string }[] {
1061
- const readerName = config.speakers.self.name?.trim() || 'the user'
1062
- const system = `You are ${readerName}'s personal semantic note-taking assistant, not a generic meeting-minutes template generator.
1063
-
1064
- Your goal is not to reproduce a "meeting minutes" format, but to turn a recording into the most efficient understanding material: let ${readerName} quickly grasp what the discussion was really about, why it matters, what ideas/judgments/items it contains, what deserves attention, and what to do next.
1065
-
1066
- Important: do not output only compressed "conclusions". Much of a recording's value lies in how views were raised, challenged, argued, and revised, and how consensus or disagreement formed. Without mechanically copying the transcript, reconstruct the key speakers' views, reasoning, debates, decision evolution, and how consensus emerged.
1067
-
1068
- Core principles:
1069
- 1. Structure is entirely determined by content. Do not apply any fixed template or emit fixed sections for form's sake.
1070
- 2. Prioritize semantic value over paragraph-by-paragraph retelling; but do not flatten the process into conclusions. Important thinking, debate, validation, concession, rebuttal, and consensus-building are themselves semantic value.
1071
- 3. Multi-person conversations must be reconstructed as much as possible: each side's initial concerns/positions, their reasons and examples, who raised challenges or rebuttals, how the discussion pivoted, which views were revised, what consensus formed, and which disagreements remain open.
1072
- 4. Solo thinking must also have its reasoning path reconstructed: how the question was raised, how hypotheses were tested, why some options were ruled out, which experience/analogies supported the judgment, and why the current conclusion formed.
1073
- 5. Freely choose the form: short memo, strategy memo, question tree, decision record, action list, mind-map-style hierarchy, phase review, debate review, study notes, product/technical analysis, etc.; pick whichever fits the content best.
1074
- 6. If the discussion is conceptual/exploratory, focus on helping the reader understand the train of thought, key concepts, reasoning chains, shifts in views, and passages worth revisiting; do not force-extract to-dos.
1075
- 7. If the discussion is execution/project-oriented, then besides conclusions, items, owners, risks, and next steps, also explain how those conclusions were reached: what constraints applied, which options were compared, and why the current path was chosen.
1076
- 8. If the discussion is short, output only the minimal useful content; if long, you may start with a reading guide and then expand. For long content, err on the side of length rather than dropping key reasoning and debates.
1077
- 9. Avoid filler, boilerplate, and formalistic headings. Every heading should carry information.
1078
- 10. If real names appear in the transcript (see Speaker context below), use them directly; keep Speaker A/B/C only when unsure.
1079
- 11. Explicitly flag uncertain or likely mis-transcribed words; do not treat them as facts.
1080
- 12. Default is Integrated notes mode: the input transcript may not have been separately cleaned. Before generating content, internally perform necessary cleanup: fix obvious typos, unify terminology, restore speakers, merge verbal repetition, fix punctuation and sentence breaks; but never invent information not in the source, and never scrub away the genuine thinking process.
1081
-
1082
- Write all output content (title, markdown, structured fields) in the dominant language of the transcript.
1083
-
1084
- Output must be valid JSON, no markdown fences.
1085
-
1086
- ${speakerContextBlock(config.speakers)}`
1087
-
1088
- const user = `Generate a "semantic notes" document from the transcript below.
1089
-
1090
- Processing mode: Integrated notes mode (no separate transcript cleanup pass; perform necessary cleanup, error correction, organization, and speaker restoration while generating the notes)
1091
-
1092
- The reading scenario you serve:
1093
- - When ${readerName} opens these notes later, they should immediately know: what is worth reading in this recording, what the core ideas/items are, how those views were discussed/argued, what needs understanding, which questions remain open, and what to do next.
1094
- - Do not assume this is a "meeting"; it may be thinking aloud, product ideation, a technical discussion, a business judgment, study notes, an idea capture, a phone call, or task execution.
1095
- - Do not follow Feishu/generic meeting-minutes structures. The markdown structure is determined by the content's semantics.
1096
- - For multi-person discussions, the notes should help ${readerName} review the process: who raised what question, who held what view, who challenged what, how it was answered, where the turning points were, and how consensus formed or disagreements remained.
1097
- - If the transcript clearly contains discussion, debate, joint reasoning, option comparison, or evolving views, the markdown body must include a section that carries this "process reconstruction" (title up to you, e.g. "How the discussion unfolded", "How the views evolved", "Debate and consensus"); a bare conclusion list is not acceptable.
1098
-
1099
- Recording info:
1100
- - Source file: ${rec.sourcePath}
1101
- - Local audio: ${localAudioPath}
1102
- - Time inferred from filename: ${rec.recordedAt.toISOString()}
1103
- - File size: ${rec.sizeBytes} bytes
1104
- - Duration: ${rec.durationSeconds} seconds
1105
-
1106
- Output JSON with these fields:
1107
- {
1108
- "title": "A title in the transcript's language that captures the real topic and value; avoid generic 'meeting minutes' phrasing",
1109
- "date": "YYYY-MM-DD",
1110
- "start_time": "HH:mm|null",
1111
- "end_time": "HH:mm|null",
1112
- "participants": ["Only actually identified real names (including the user's); never Speaker A/B"],
1113
- "organizations": ["string"],
1114
- "projects": ["string"],
1115
- "markdown": "Full markdown body. Must start with an # H1 title. Structure is entirely yours based on the semantics; do not include the trailing source details block, the system appends it.",
1116
- "discussion_flow": [{"stage": "discussion stage/topic", "what_happened": "what happened in this stage", "speaker_positions": [{"speaker": "real name or Speaker label", "position": "view/concern/reasoning"}], "turning_point": "key pivot or change of view|null", "outcome": "stage consensus/disagreement/open|null"}],
1117
- "consensus_points": [{"point": "consensus reached", "how_reached": "how this consensus formed through discussion/argument|null"}],
1118
- "disagreements": [{"issue": "point of disagreement", "positions": [{"speaker": "real name or Speaker label", "position": "stance and reasoning"}], "status": "resolved|unresolved|partially_resolved|null"}],
1119
- "action_items": [{"task": "string", "owner": "string|null", "due_date": "YYYY-MM-DD|null", "priority": "high|medium|low|null", "note": "string|null"}],
1120
- "decisions": [{"decision": "string", "reason": "string|null", "owner": "string|null", "date": "YYYY-MM-DD|null", "how_reached": "how this decision was reached|null"}],
1121
- "open_questions": [{"question": "string", "next_step": "string|null"}],
1122
- "key_quotes_or_details": ["string"],
1123
- "transcription_uncertainties": ["string"]
1124
- }
1125
-
1126
- Markdown quality requirements:
1127
- - The first screen must have a high signal-to-noise ratio: the reader should know why this content is worth keeping without reading the full transcript.
1128
- - No empty sections; no placeholder content like "no clear record / unknown / unidentified".
1129
- - Do not force headings like "Summary, To-dos, Smart sections, Key decisions, Quotes"; use them only when semantically warranted.
1130
- - If there are action items, use concrete actionable language; if there are none, do not fabricate any.
1131
- - If there are ideas/judgments, write out the reasoning chain, not just conclusions.
1132
- - If there was discussion, debate, or joint reasoning, preserve the key process: view raised → challenge/addition → response/rebuttal → revision/pivot → consensus/disagreement. Do not compress it into a single "in the end they concluded…".
1133
- - The markdown body should primarily reconstruct the process in natural language; do not just fill discussion_flow/consensus_points/disagreements as metadata and stop — those structured fields only aid your thinking and indexing.
1134
- - For important consensus, explain how it was reached; for important disagreements, state who held what view, why, and whether it was resolved.
1135
- - If a conclusion went through option comparison or trade-offs, write out the compared options, the criteria, and why one was dropped or chosen.
1136
- - For long meetings, review by topic/stage rather than as a running log, but keep each stage's key turning points and representative speakers' views.
1137
- - Clearly flag controversies, risks, and unverified assumptions.
1138
- - Timestamps may be used sparingly when they help revisit key passages; do not build a full timeline for form's sake.
1139
- - If the transcript has uncertain words, surface them in context as reminders; do not treat them as facts.
1140
- - In Integrated notes mode, especially avoid carrying stutters, repetitions, and typos from the raw transcript into the notes; the body should present cleaned, organized content while preserving the genuine reasoning, debates, and evolution of views.
1141
-
1142
- Transcript:
1143
- ${transcript}`
1144
- return [{ role: 'system', content: system }, { role: 'user', content: user }]
1145
- }
1146
-
1147
- // ───────────────────────────────────────────────────────────────────────
1148
- // Summary via pi. Provider and credentials are pi's own configuration. The
1149
- // optional VOICENOTE_PI_MODEL pins a model; otherwise pi's selected model writes
1150
- // the notes. VoiceNote does not implement a provider fallback chain.
1151
- // ───────────────────────────────────────────────────────────────────────
1152
-
1153
- // pi can't be `bun build --compile`'d (it reads data files from disk), so the
1154
- // bundled GUI ships pi as plain JS and runs it under a bundled bun. When
1155
- // `pi.cli` is set, `pi.bin` is the runtime (bun) and the cli.js is prepended to
1156
- // pi's args — `<bun> <cli.js> <args>`, no wrapper script and no shell (critical
1157
- // on Windows, where pi args include a huge --system-prompt that a .cmd/%*
1158
- // wrapper would mangle). CLI users with a real `pi` on PATH leave it unset.
1159
- function piInvocation(pi: PiConfig, args: string[]): { bin: string; args: string[] } {
1160
- return pi.cli ? { bin: pi.bin, args: [pi.cli, ...args] } : { bin: pi.bin, args }
1161
- }
1162
-
1163
- // ───────────────────────────────────────────────────────────────────────
1164
- // ChatGPT (OpenAI Codex) OAuth login. The browser callback is the default;
1165
- // --device-code is available for accounts that opted into that flow. This
1166
- // exposes pi's login as a plain command for non-TUI and GUI users.
1167
- // We reuse pi's own OAuth implementation (@earendil-works/pi-ai) and persist
1168
- // to pi's auth.json in the exact shape it reads: { type: 'oauth', ...creds }.
1169
- // ───────────────────────────────────────────────────────────────────────
1170
-
1171
- async function persistPiOAuth(authPath: string, providerId: string, creds: Record<string, unknown>): Promise<void> {
1172
- await mkdir(dirname(authPath), { recursive: true })
1173
- let existing: Json = {}
1174
- if (existsSync(authPath)) {
1175
- try { existing = JSON.parse(await readFile(authPath, 'utf8')) as Json } catch (e) { warnSideEffect(`parse ${authPath}`, e) }
1176
- }
1177
- existing[providerId] = { type: 'oauth', ...creds }
1178
- const tmp = `${authPath}.tmp-${process.pid}`
1179
- await writeFile(tmp, JSON.stringify(existing, null, 2) + '\n', { mode: 0o600 })
1180
- await rename(tmp, authPath)
1181
- }
1182
-
1183
- async function loginChatGPT(opts: { json?: boolean; deviceCode?: boolean; emit?: (o: Record<string, unknown>) => void }): Promise<void> {
1184
- // OpenAI's OAuth endpoint is geo-blocked in some regions; getConfig() resolves
1185
- // the proxy into this process's env before any request goes out.
1186
- const authPath = getConfig().pi.authPath
1187
- const json = !!opts.json
1188
- const emit = opts.emit ?? ((o: Record<string, unknown>) => { if (json) console.log(JSON.stringify(o)) })
1189
- try {
1190
- const oauth = await import('@earendil-works/pi-ai/oauth')
1191
- let creds: Record<string, unknown>
1192
- if (opts.deviceCode) {
1193
- // Device-code flow: no localhost server, but the account must first enable
1194
- // "device code authorization for Codex" in ChatGPT > Settings > Security.
1195
- creds = await oauth.loginOpenAICodexDeviceCode({
1196
- onDeviceCode: (info) => {
1197
- if (json) emit({ event: 'device_code', userCode: info.userCode, verificationUri: info.verificationUri, intervalSeconds: info.intervalSeconds, expiresInSeconds: info.expiresInSeconds })
1198
- else {
1199
- console.log('\nTo sign in to ChatGPT (device code):')
1200
- console.log(` 1. Open ${info.verificationUri}`)
1201
- console.log(` 2. Enter code: ${info.userCode}`)
1202
- console.log('\nIf you see "Enable device code authorization", turn it on in')
1203
- console.log('ChatGPT > Settings > Security — or just rerun `vn login` (browser flow).')
1204
- console.log('\nWaiting for authorization…')
1205
- }
1206
- },
1207
- }) as Record<string, unknown>
1208
- } else {
1209
- // Default: browser-callback flow (same as pi `/login` and the official Codex
1210
- // CLI). Spins up localhost:1455/auth/callback; no account setting required.
1211
- creds = await oauth.loginOpenAICodex({
1212
- onAuth: ({ url }) => {
1213
- if (json) emit({ event: 'auth_url', url })
1214
- else {
1215
- console.log('\nOpening your browser to sign in to ChatGPT…')
1216
- console.log(`If it doesn't open, paste this into a browser on THIS machine:\n ${url}`)
1217
- }
1218
- // Best-effort auto-open; the URL is printed/emitted above as fallback.
1219
- void openPath(url)
1220
- },
1221
- onPrompt: async ({ message }) => {
1222
- // Only reached if the localhost:1455 callback can't complete (port busy,
1223
- // or browser on another machine). Fail loudly rather than hang.
1224
- throw new Error(`${message} — automatic callback failed (is localhost:1455 free, and is your browser on this machine?). Retry, or use --device-code.`)
1225
- },
1226
- }) as Record<string, unknown>
1227
- }
1228
- await persistPiOAuth(authPath, oauth.openaiCodexOAuthProvider.id, creds)
1229
- if (json) emit({ event: 'success', provider: oauth.openaiCodexOAuthProvider.id })
1230
- else console.log(`\n✓ Signed in. Credentials saved to ${authPath}. Verify with: vn doctor`)
1231
- } catch (e: any) {
1232
- let message = String(e?.message || e)
1233
- if (/unsupported_country_region_territory|\b403\b/.test(message)) {
1234
- message += ' — OpenAI blocks this region without a proxy. Set LOCAL_PROXY_HOST/LOCAL_PROXY_PORT (or http_proxy) and retry; Volcano stays direct.'
1235
- }
1236
- if (json) emit({ event: 'error', message })
1237
- else console.error(`\nLogin failed: ${message}`)
1238
- process.exitCode = 1
1239
- }
1240
- }
1241
-
1242
- // ───────────────────────────────────────────────────────────────────────
1243
- // File-based config (~/.config/voicenote/config.json) — written by the GUI
1244
- // via `vn config set`, read by loadEnvConfig(). ENV config uses ENV_KEYS;
1245
- // identity lives under the same file's `speakers` object.
1246
- // ───────────────────────────────────────────────────────────────────────
1247
-
1248
- function readStdin(): Promise<string> {
1249
- return new Promise((resolve) => {
1250
- let data = ''
1251
- process.stdin.setEncoding('utf8')
1252
- process.stdin.on('data', d => { data += d })
1253
- process.stdin.on('end', () => resolve(data))
1254
- process.stdin.on('error', () => resolve(data))
1255
- })
1256
- }
1257
-
1258
- function configFileEnv(raw = loadConfigJson()): Record<string, string> {
1259
- const env: Record<string, string> = {}
1260
- for (const k of ENV_KEYS) if (typeof raw[k] === 'string') env[k] = raw[k] as string
1261
- return env
1262
- }
1263
-
1264
- function configGetData(): { path: string; env: Record<string, string>; self: { name: string | null; aliases: string[] } } {
1265
- const current = loadConfigJson()
1266
- const speakers = normalizeSpeakers(current.speakers ?? DEFAULT_SPEAKERS)
1267
- return {
1268
- path: CONFIG_ENV_PATH,
1269
- env: configFileEnv(current),
1270
- self: { name: speakers.self.name, aliases: speakers.self.aliases },
1271
- }
1272
- }
1273
-
1274
- function configGet(): void { console.log(JSON.stringify(configGetData(), null, 2)) }
1275
-
1276
- type ConfigSetPayload = { env?: Record<string, unknown>; self?: { name?: string | null; aliases?: string[] } }
1277
-
1278
- async function writeConfigJson(value: Record<string, unknown>): Promise<void> {
1279
- await mkdir(CONFIG_DIR, { recursive: true })
1280
- const tmp = `${CONFIG_ENV_PATH}.tmp-${process.pid}`
1281
- await writeFile(tmp, JSON.stringify(value, null, 2) + '\n', { mode: 0o600 })
1282
- await rename(tmp, CONFIG_ENV_PATH)
1283
- }
1284
-
1285
- async function configSetData(payload: ConfigSetPayload): Promise<{ ok: true; path: string; ignoredKeys?: string[] }> {
1286
- if (!payload || typeof payload !== 'object' || Array.isArray(payload)) throw new Error('Config payload must be a JSON object')
1287
- const current = loadConfigJson()
1288
- const known = ENV_KEYS as readonly string[]
1289
- const ignored: string[] = []
1290
- if (payload.env) {
1291
- for (const [key, value] of Object.entries(payload.env)) {
1292
- if (!known.includes(key)) { ignored.push(key); continue }
1293
- if (value === null) delete current[key]
1294
- else if (typeof value === 'string') current[key] = value
1295
- else throw new Error(`Config value ${key} must be a string or null`)
1296
- }
1297
- }
1298
- if (payload.self) {
1299
- const speakers = normalizeSpeakers(current.speakers ?? DEFAULT_SPEAKERS)
1300
- if (payload.self.name !== undefined) {
1301
- if (payload.self.name !== null && typeof payload.self.name !== 'string') throw new Error('self.name must be a string or null')
1302
- speakers.self.name = payload.self.name
1303
- }
1304
- if (payload.self.aliases !== undefined) {
1305
- if (!Array.isArray(payload.self.aliases) || payload.self.aliases.some(alias => typeof alias !== 'string')) throw new Error('self.aliases must contain only strings')
1306
- speakers.self.aliases = payload.self.aliases
1307
- }
1308
- current.speakers = speakers
1309
- }
1310
- await writeConfigJson(current)
1311
- return { ok: true, path: CONFIG_ENV_PATH, ...(ignored.length ? { ignoredKeys: ignored } : {}) }
1312
- }
1313
-
1314
- async function configSet(): Promise<void> {
1315
- let payload: ConfigSetPayload
1316
- try { payload = JSON.parse(await readStdin()) }
1317
- catch (e: any) { console.error(`Invalid JSON on stdin: ${e?.message || e}`); process.exitCode = 1; return }
1318
- // Every other key is re-read by the agent on each run, but VOICENOTE_PI_BIN
1319
- // is snapshotted into the scheduler as a resolved absolute path at install
1320
- // time (launchd's fixed PATH can't find it otherwise). The GUI reinstalls on
1321
- // save; the CLI path must be told — but only when the value actually CHANGES.
1322
- // A GUI-style client resubmits every field on every save, so `in payload`
1323
- // alone would nag on every unrelated edit.
1324
- const PI_BIN = 'VOICENOTE_PI_BIN'
1325
- const before = String(loadConfigJson()[PI_BIN] ?? '')
1326
- console.log(JSON.stringify(await configSetData(payload)))
1327
- const piBinChanged = payload.env && PI_BIN in payload.env && String(payload.env[PI_BIN] ?? '') !== before
1328
- if (piBinChanged) {
1329
- console.error(`Note: ${PI_BIN} changed — re-run \`vn install-launch-agent\` to apply it to the background scheduler.`)
1330
- }
1331
- }
1332
-
1333
- function extractFirstJsonObject(text: string): string {
1334
- const raw = text.trim()
1335
- // Models often wrap JSON in a ```json fence; strip it before looking inside.
1336
- const trimmed = raw.match(/^```(?:json)?\s*([\s\S]*?)\s*```\s*$/i)?.[1]?.trim() ?? raw
1337
- if (trimmed.startsWith('{') && trimmed.endsWith('}')) return trimmed
1338
- // Find the first balanced {...}
1339
- let depth = 0, start = -1, inString = false, escape = false
1340
- for (let i = 0; i < trimmed.length; i++) {
1341
- const ch = trimmed[i]!
1342
- if (escape) { escape = false; continue }
1343
- if (inString) {
1344
- if (ch === '\\') { escape = true; continue }
1345
- if (ch === '"') inString = false
1346
- continue
1347
- }
1348
- if (ch === '"') { inString = true; continue }
1349
- if (ch === '{') { if (depth === 0) start = i; depth++ }
1350
- else if (ch === '}') { depth--; if (depth === 0 && start !== -1) return trimmed.slice(start, i + 1) }
1351
- }
1352
- return trimmed
1353
- }
1354
-
1355
- type PiRunOptions = {
1356
- systemPrompt: string
1357
- userPrompt: string
1358
- timeoutMs?: number
1359
- thinking?: string
1360
- tools?: string // e.g. 'read,grep'; empty/undefined = --no-tools
1361
- appendSystemPrompt?: string
1362
- cwd?: string // agent working dir: the knowledge base, so read/grep/find default there
1363
- }
1364
-
1365
- async function runPi(config: Config, opts: PiRunOptions): Promise<string> {
1366
- const args = [
1367
- '-p',
1368
- '--mode', 'text',
1369
- '--no-extensions', '--no-skills', '--no-context-files', '--no-session', '--no-prompt-templates', '--no-themes',
1370
- '--system-prompt', opts.systemPrompt,
1371
- ]
1372
- // Unset means pi's own default model and provider. There is no second
1373
- // provider to fall back to either way.
1374
- if (config.pi.model) args.push('--model', config.pi.model)
1375
- if (opts.thinking) args.push('--thinking', opts.thinking)
1376
- if (opts.tools && opts.tools.trim()) args.push('--tools', opts.tools.trim())
1377
- else args.push('--no-tools')
1378
- if (opts.appendSystemPrompt) args.push('--append-system-prompt', opts.appendSystemPrompt)
1379
- return new Promise<string>((resolve, reject) => {
1380
- const inv = piInvocation(config.pi, args)
1381
- const child = spawn(inv.bin, inv.args, { stdio: ['pipe', 'pipe', 'pipe'], cwd: opts.cwd, windowsHide: true, env: { ...process.env, ...config.childEnv } })
1382
- let stdout = '', stderr = ''
1383
- const timer = opts.timeoutMs ? setTimeout(() => child.kill('SIGKILL'), opts.timeoutMs) : null
1384
- child.stdout.on('data', d => stdout += String(d))
1385
- child.stderr.on('data', d => stderr += String(d))
1386
- child.on('error', err => { if (timer) clearTimeout(timer); reject(err) })
1387
- child.on('close', code => {
1388
- if (timer) clearTimeout(timer)
1389
- if (code !== 0) return reject(new Error(`pi exited ${code}: ${(stderr || stdout).slice(0, 800)}`))
1390
- const text = stdout.trim()
1391
- if (!text) return reject(new Error('pi returned empty output'))
1392
- resolve(text)
1393
- })
1394
- // A pi that dies before draining stdin (bad flags, crash on startup) closes the
1395
- // pipe mid-write. Without this handler the EPIPE is an unhandled 'error' event
1396
- // that kills the whole run, hiding pi's actual error; 'close' below reports it.
1397
- child.stdin.on('error', (e: NodeJS.ErrnoException) => {
1398
- if (e.code !== 'EPIPE') warnSideEffect('write prompt to pi stdin', e)
1399
- })
1400
- child.stdin.end(opts.userPrompt)
1401
- })
1402
- }
1403
-
1404
- // A transient pi failure (proxy reset, dropped socket, upstream 5xx/429) is
1405
- // retried: a momentary blip must not cost a run its notes. Quota/auth/4xx are NOT
1406
- // transient — retrying them only wastes time, so they fail the summary at once.
1407
- function isTransientPiError(e: any): boolean {
1408
- const msg = String(e?.message || e).toLowerCase()
1409
- if (/quota|unauthorized|invalid.*(key|token|credential)|forbidden|\b40[0-4]\b/.test(msg)) return false
1410
- return /socket connection was closed|socket hang up|econnreset|etimedout|esockettimedout|enetunreach|econnrefused|eai_again|fetch failed|network error|timed ?out|temporarily|overloaded|\b(429|500|502|503|504)\b/.test(msg)
1411
- }
1412
-
1413
- async function chatCompleteViaPi(config: Config, opts: PiRunOptions): Promise<string> {
1414
- const maxAttempts = config.pi.retries
1415
- for (let attempt = 1; ; attempt++) {
1416
- try {
1417
- return await runPi(config, opts)
1418
- } catch (e: any) {
1419
- if (attempt >= maxAttempts || !isTransientPiError(e)) throw e
1420
- const backoffMs = Math.min(30000, 2000 * 2 ** (attempt - 1))
1421
- console.error(`pi transient error (attempt ${attempt}/${maxAttempts}); retrying in ${backoffMs}ms: ${e?.message || e}`)
1422
- await new Promise(res => setTimeout(res, backoffMs))
1423
- }
1424
- }
1425
- }
1426
-
1427
- function piSummaryToolsHint(contextDir: string): string {
1428
- return `Before writing the notes you have two read-only tools: read and grep. Your current working directory (cwd) is \`${contextDir}\` (the configured notes/reference directory); use relative paths for grep/read.\n\nGoal: use existing context to align names, speakers, client/project names, product names, and domain terms in this note; do not maintain or assume a separate glossary.\n\nSuggested flow:\n- First extract the most likely client/project/product keywords from the title, filename, and transcript.\n- If a clear topic matches, prefer grep/read on related index pages, project docs, status records, or the 3-5 most recent related notes in the same directory; use them to identify Speaker B/C/F etc., common aliases, product names, and term spellings.\n- If no clear topic matches, grep the current directory with keywords and read only the few most relevant files.\n- Before output, do one names/terms lint pass: eliminate leftover Speaker A/B/C, obviously misheard names, product-name variants, and outdated names; when context is insufficient, keep the uncertainty — never guess.\n\nConstraints:\n- At most 10 tool calls total; if the transcript alone is sufficient, make none.\n- Read only within \`${contextDir}\`; skip directories that clearly involve personal privacy/credentials/finance (e.g. identity / credentials / finance).\n- Found information is only for consistency and background calibration; never write content absent from this transcript into the notes as new meeting facts.\n- Do not attempt to write files or call bash (those tools are not enabled).`
1429
- }
1430
-
1431
- // Summary runs through pi. The agent's working dir IS the knowledge
1432
- // base, so read/grep/find operate there directly. If a configured context dir is
1433
- // missing, say so loudly and run without tools rather than searching the wrong
1434
- // tree (tools, the cwd hint, and the spawn cwd move together).
1435
- async function chatComplete(opts: { systemPrompt: string; userPrompt: string; config: Config }): Promise<string> {
1436
- const { pi } = opts.config
1437
- const ctx = pi.tools ? pi.contextDir : undefined
1438
- const ctxExists = ctx ? existsSync(ctx) : false
1439
- if (ctx && !ctxExists) console.error(`Warning: context dir ${ctx} does not exist; summary agent runs WITHOUT read/grep cross-reference.`)
1440
- const toolsActive = !!ctx && ctxExists
1441
- return chatCompleteViaPi(opts.config, {
1442
- systemPrompt: opts.systemPrompt,
1443
- userPrompt: opts.userPrompt,
1444
- timeoutMs: 60 * 60 * 1000,
1445
- thinking: pi.thinking,
1446
- tools: toolsActive ? pi.tools : undefined,
1447
- appendSystemPrompt: toolsActive ? piSummaryToolsHint(ctx!) : undefined,
1448
- cwd: toolsActive ? ctx : undefined,
1449
- })
1450
- }
1451
-
1452
- async function summarizeTranscript(config: Config, transcript: string, rec: Recording, localAudioPath: string): Promise<Json> {
1453
- const messages = summaryMessages(config, transcript, rec, localAudioPath)
1454
- const systemPrompt = String(messages[0]!.content)
1455
- const userPrompt = String(messages[1]!.content)
1456
- const text = await chatComplete({ systemPrompt, userPrompt, config })
1457
- const jsonText = extractFirstJsonObject(text)
1458
- try {
1459
- return JSON.parse(jsonText || '{}') as Json
1460
- } catch (e: any) {
1461
- throw new Error(`summary returned non-JSON output (${e?.message || e}). First 400 chars: ${text.slice(0, 400)}`)
1462
- }
1463
- }
1464
-
1465
- // ────────────────────────────────────────────────────────────────────────────
1466
- // Metadata + markdown
1467
- // ────────────────────────────────────────────────────────────────────────────
1468
-
1469
- function isSpeakerLabel(text: string): boolean {
1470
- return /^\s*speaker\s+[a-z]\s*$/i.test(text) || /^\s*说话人\s*[A-ZA-Za-za-z一二三四五六七八九十0-9]+\s*$/.test(text)
1471
- }
1472
-
1473
- function normalizeMetadata(meta: Json, rec: Recording): Json {
1474
- const d = rec.recordedAt
1475
- meta.date ||= `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`
1476
- meta.start_time ||= `${pad(d.getHours())}:${pad(d.getMinutes())}`
1477
- meta.end_time ??= null
1478
- for (const key of ['participants', 'organizations', 'projects', 'discussion_flow', 'consensus_points', 'disagreements', 'action_items', 'decisions', 'open_questions', 'key_quotes_or_details', 'transcription_uncertainties']) {
1479
- if (!Array.isArray(meta[key])) meta[key] = []
1480
- }
1481
- meta.participants = meta.participants.filter((p: any) => typeof p === 'string' && p.trim() && !isSpeakerLabel(p))
1482
- return meta
1483
- }
1484
-
1485
- const SOURCE_MARKER = '<!-- voicenote:source -->'
1486
- function sourceDetails(audioPath: string, transcriptPath: string): string {
1487
- return `${SOURCE_MARKER}\n<details>\n<summary>Source</summary>\n\n- Generated by: voicenote automatic transcription\n- Original audio: \`${audioPath}\`\n- Full transcript: \`${transcriptPath}\`\n\n</details>`
1488
- }
1489
-
1490
- function markdownNotes(meta: Json, audioPath: string, transcriptPath: string): string {
1491
- let body = typeof meta.markdown === 'string' && meta.markdown.trim() ? meta.markdown.trim() : `# ${meta.title || 'Untitled recording notes'}\n`
1492
- if (!body.startsWith('#')) body = `# ${meta.title || 'Untitled recording notes'}\n\n${body}`
1493
- if (!body.includes(SOURCE_MARKER)) body = `${body.trim()}\n\n${sourceDetails(audioPath, transcriptPath)}`
1494
- return `${body.trim()}\n`
1495
- }
1496
-
1497
- async function markdownToPdf(markdownPath: string): Promise<string> {
1498
- const pdfPath = markdownPath.replace(/\.md$/i, '.pdf')
1499
- const tempBase = join(os.tmpdir(), `voicenote-pdf-${Date.now()}-${Math.random().toString(36).slice(2)}`)
1500
- const htmlPath = `${tempBase}.html`
1501
- const cssPath = `${tempBase}.css`
1502
- const css = `
1503
- :root { color-scheme: light; }
1504
- body { font-family: -apple-system, BlinkMacSystemFont, "PingFang SC", "Hiragino Sans GB", "Microsoft YaHei", "Noto Sans CJK SC", sans-serif; line-height: 1.68; color: #1f2328; max-width: 860px; margin: 40px auto; padding: 0 32px; font-size: 15px; }
1505
- h1, h2, h3 { line-height: 1.32; margin-top: 1.8em; color: #111827; }
1506
- h1 { font-size: 28px; border-bottom: 1px solid #e5e7eb; padding-bottom: 12px; }
1507
- h2 { font-size: 22px; border-bottom: 1px solid #eef2f7; padding-bottom: 6px; }
1508
- h3 { font-size: 18px; }
1509
- p, ul, ol, blockquote, table { margin: 0.9em 0; }
1510
- blockquote { border-left: 4px solid #d0d7de; padding-left: 16px; color: #57606a; }
1511
- code { font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, monospace; background: #f6f8fa; padding: 0.15em 0.35em; border-radius: 4px; }
1512
- table { border-collapse: collapse; width: 100%; }
1513
- th, td { border: 1px solid #d0d7de; padding: 8px 10px; vertical-align: top; }
1514
- th { background: #f6f8fa; }
1515
- details { margin-top: 2em; color: #57606a; font-size: 13px; }
1516
- @page { size: A4; margin: 18mm 16mm; }
1517
- @media print { body { margin: 0; padding: 0; max-width: none; } h1, h2, h3 { break-after: avoid; } table, blockquote { break-inside: avoid; } }
1518
- `
1519
- await writeFile(cssPath, css, 'utf8')
1520
- try {
1521
- const title = basename(markdownPath, extname(markdownPath))
1522
- const pandoc = await runCommand('pandoc', [markdownPath, '--from', 'markdown+smart', '--to', 'html5', '--standalone', '--metadata', `title=${title}`, '--css', cssPath, '-o', htmlPath], 120000)
1523
- if (pandoc.code !== 0) throw new Error(`pandoc failed: ${pandoc.stderr || pandoc.stdout}`)
1524
- const chromePath = existsSync('/Applications/Google Chrome.app/Contents/MacOS/Google Chrome') ? '/Applications/Google Chrome.app/Contents/MacOS/Google Chrome' : 'google-chrome'
1525
- const chrome = await runCommand(chromePath, ['--headless', '--disable-gpu', '--no-pdf-header-footer', `--print-to-pdf=${pdfPath}`, pathToFileURL(htmlPath).href], 120000)
1526
- if (chrome.code !== 0 || !existsSync(pdfPath)) throw new Error(`chrome pdf failed: ${chrome.stderr || chrome.stdout}`)
1527
- return pdfPath
1528
- } finally {
1529
- await unlink(htmlPath).catch(() => {})
1530
- await unlink(cssPath).catch(() => {})
1531
- }
1532
- }
1533
-
1534
- function transcriptMarkdown(config: Config, rec: Recording, transcript: string, opts: { mode?: RunMode } = {}): string {
1535
- const transcribeBackend = `Volcano Doubao (resource ${config.volcano?.resourceId || 'volc.seedasr.auc'})`
1536
- return `# Transcript: ${basename(rec.sourcePath)}\n\n- Source file: \`${rec.sourcePath}\`\n- Transcription backend: ${transcribeBackend}\n- Mode: ${opts.mode || 'notes'}\n- Recorded at: ${rec.recordedAt.toISOString()}\n- File size: ${rec.sizeBytes} bytes\n- Duration: ${rec.durationSeconds ?? 'unknown'} seconds\n- Transcribed at: ${nowIso()}\n\n---\n\n${RAW_TRANSCRIPT_MARKER}${transcript.trim()}`
1537
- }
1538
-
1539
- // ────────────────────────────────────────────────────────────────────────────
1540
- // Pipeline
1541
- // ────────────────────────────────────────────────────────────────────────────
1542
-
1543
- async function processRecording(config: Config, rec: Recording, opts: any): Promise<Json> {
1544
- const jobStarted = Date.now()
1545
- let files = (opts.resumeFromTranscriptFiles as LocalFiles | null) || layout(config, rec)
1546
- const mode = normalizeRunMode(opts)
1547
- const needsNotes = mode === 'notes'
1548
- const resumeSummary = needsNotes && Boolean(opts.resumeFromTranscriptFiles)
1549
- const transcribeBackendLabel = `volcano:${config.volcano?.resourceId || 'volc.seedasr.auc'}`
1550
- const llmBackendLabel = needsNotes && !opts.dryRun ? 'pi' : null
1551
- const plan = resumeSummary
1552
- ? 'reuse saved transcript → integrated semantic notes → write metadata/index (no auto move)'
1553
- : `copy audio → transcribe → write transcript${needsNotes ? ' → integrated semantic notes' : ''} → write metadata/index (no auto move)`
1554
-
1555
- console.log(`\n=== voicenote job: ${basename(rec.sourcePath)} ===`)
1556
- console.log(`Source: ${rec.sourcePath}`)
1557
- console.log(`Audio: duration=${rec.durationSeconds == null ? 'unknown' : formatSeconds(rec.durationSeconds)}, size=${formatBytes(rec.sizeBytes)}, mode=${mode}, asr=${transcribeBackendLabel}${llmBackendLabel ? `, llm=${llmBackendLabel}` : ''}`)
1558
- console.log(`Plan: ${plan}`)
1559
- if (opts.dryRun) return { source_path: rec.sourcePath, source_id: rec.sourceId, would_copy_to: files.audio, resume_from_transcript: resumeSummary ? files.transcript : null, size_bytes: rec.sizeBytes, duration_seconds: rec.durationSeconds, mode }
1560
-
1561
- const totalSteps = resumeSummary ? 3 : needsNotes ? 4 : 3
1562
- let stepNo = 0
1563
- const nextStep = () => ++stepNo
1564
-
1565
- let transcript = ''
1566
- let meta: Json = {
1567
- title: basename(rec.sourcePath, extname(rec.sourcePath)),
1568
- markdown: '',
1569
- }
1570
-
1571
- if (resumeSummary) {
1572
- progressStep(nextStep(), totalSteps, 'Reuse saved transcript', files.transcript)
1573
- transcript = await readSavedTranscript(files.transcript)
1574
- console.log(`✓ Reusing transcript: ${files.transcript}`)
1575
- if (!existsSync(files.audio)) {
1576
- await mkdir(dirname(files.audio), { recursive: true })
1577
- await copyFile(rec.sourcePath, files.audio)
1578
- console.log(`✓ Local audio restored: ${files.audio}`)
1579
- }
1580
- } else {
1581
- progressStep(nextStep(), totalSteps, 'Copy audio to workspace', files.audio)
1582
- await mkdir(dirname(files.audio), { recursive: true })
1583
- await copyFile(rec.sourcePath, files.audio)
1584
- console.log(`✓ Local audio ready: ${files.audio}`)
1585
-
1586
- progressStep(nextStep(), totalSteps, 'Transcribe audio', transcribeBackendLabel)
1587
- transcript = await withHeartbeat('transcribe audio', () => transcribeAudio(config, files.audio, rec), 90)
1588
-
1589
- // Persist transcript IMMEDIATELY so an expensive ASR result is never lost
1590
- // if a later step (summary) blows up. We use the initial (untitled) path;
1591
- // if summary succeeds we'll move it to the titled path below.
1592
- await mkdir(dirname(files.transcript), { recursive: true })
1593
- // Atomic: "transcript exists on disk" is what makes a later run skip ASR, so
1594
- // a run killed mid-write must not leave a truncated file behind. The raw
1595
- // marker sits near the top, so a partial write would still pass
1596
- // readSavedTranscript()'s checks and get summarised as if complete.
1597
- await writeFileAtomic(files.transcript, transcriptMarkdown(config, rec, transcript, { mode }))
1598
- console.log(`✓ Transcript saved: ${files.transcript}`)
1599
- }
1600
-
1601
- let summaryError: any = null
1602
- if (needsNotes) {
1603
- progressStep(nextStep(), totalSteps, 'Generate integrated semantic notes', `via pi, model=${config.pi.model || "pi's own default"}`)
1604
- try {
1605
- meta = await withHeartbeat('generate integrated semantic notes', () => summarizeTranscript(config, transcript, rec, files.audio), 60)
1606
- } catch (e: any) {
1607
- summaryError = e
1608
- console.error(`Summary step failed; transcript is preserved. Error: ${e?.message || e}`)
1609
- console.error(`Hint: fix LLM auth/credits, then re-run with: vn run --latest`)
1610
- }
1611
- }
1612
-
1613
- meta = normalizeMetadata(meta, rec)
1614
- meta.processing_mode = mode
1615
- meta.source_audio_path = rec.sourcePath
1616
- meta.source_id = rec.sourceId
1617
- meta.source_size_bytes = rec.sizeBytes
1618
- meta.source_modified_at = rec.modifiedAt
1619
- meta.duration_seconds = rec.durationSeconds
1620
- meta.asr_provider = 'volcano'
1621
- meta.transcribe_model = config.volcano?.resourceId || 'volc.seedasr.auc'
1622
- // pi picks the model, so we cannot name it here. Null when no summary ran —
1623
- // summary_error says why.
1624
- meta.llm_backend = needsNotes && !summaryError ? 'pi' : null
1625
- meta.processed_at = nowIso()
1626
- if (summaryError) meta.summary_error = String(summaryError?.message || summaryError)
1627
-
1628
- progressStep(nextStep(), totalSteps, 'Write outputs and index')
1629
- let failedStubPathToRemove: string | null = null
1630
- if (needsNotes && !summaryError) {
1631
- const titled = layout(config, rec, meta.title)
1632
- await promoteOutputs(files, titled)
1633
- // The stub note of a failed attempt is removed only after the real note is
1634
- // written, so a failure in between still leaves the user a pointer to the
1635
- // saved transcript.
1636
- if (files.notes !== titled.notes) failedStubPathToRemove = files.notes
1637
- files = titled
1638
- }
1639
- await mkdir(dirname(files.notes), { recursive: true })
1640
- await mkdir(dirname(files.metadata), { recursive: true })
1641
-
1642
- if (needsNotes && !summaryError) {
1643
- await writeFile(files.notes, markdownNotes(meta, files.audio, files.transcript), 'utf8')
1644
- console.log(`✓ Notes: ${files.notes}`)
1645
- if (failedStubPathToRemove) await removeFailedSummaryStub(failedStubPathToRemove)
1646
- if (opts.pdf) {
1647
- const pdf = await withHeartbeat('render notes PDF', () => markdownToPdf(files.notes), 30)
1648
- meta.local_paths = { ...files, pdf }
1649
- console.log(`✓ PDF: ${pdf}`)
1650
- }
1651
- } else if (needsNotes && summaryError) {
1652
- // No unconditional "just re-run" promise: after MAX_ATTEMPTS the job is
1653
- // `gave_up` and further runs skip it, so the note has to name both ways out.
1654
- const stubBody = `# Pending summary: ${basename(rec.sourcePath)}\n\n> ⚠ Transcription completed and saved, but the summary stage failed; retry needed.\n\n- Transcript file: \`${files.transcript}\`\n- Original audio: \`${rec.sourcePath}\`\n- Failure reason: ${meta.summary_error}\n- Retry: the next \`vn run\` reuses the saved transcript automatically (no new transcription cost). After ${MAX_ATTEMPTS} failed attempts it stops retrying — run \`vn forget ${basename(rec.sourcePath)}\` to queue it again.\n`
1655
- await writeFile(files.notes, stubBody, 'utf8')
1656
- console.log(`⚠ Stub notes (summary failed): ${files.notes}`)
1657
- } else if (opts.pdf) {
1658
- console.log('PDF skipped: --pdf only applies to --mode notes.')
1659
- }
1660
-
1661
- meta.local_paths = { ...files, ...(meta.local_paths?.pdf ? { pdf: meta.local_paths.pdf } : {}) }
1662
- meta.final_paths = {
1663
- audio: files.audio,
1664
- transcript: files.transcript,
1665
- notes: needsNotes ? files.notes : null,
1666
- metadata: files.metadata,
1667
- ...(meta.local_paths?.pdf ? { pdf: meta.local_paths.pdf } : {}),
1668
- }
1669
-
1670
- if (summaryError) {
1671
- meta.status = SUMMARY_FAILED_STATUS
1672
- } else if (needsNotes) {
1673
- meta.status = 'completed'
1674
- } else {
1675
- meta.status = 'transcript_only'
1676
- }
1677
-
1678
- await writeJson(files.metadata, meta)
1679
- await appendJsonl(await notesIndexPath(config), meta)
1680
- console.log(`✓ Completed: ${meta.title || basename(rec.sourcePath)} (${formatElapsed(Date.now() - jobStarted)} total)`)
1681
- if (needsNotes) console.log(`Final notes: ${files.notes}`)
1682
- else console.log(`Final transcript: ${files.transcript}`)
1683
- return meta
1684
- }
1685
-
1686
- // ────────────────────────────────────────────────────────────────────────────
1687
- // Job state — `vn run` is the only writer; every view is a pure read of this.
1688
- // ────────────────────────────────────────────────────────────────────────────
1689
-
1690
- // Named for what it holds: every recording's job state, not just the processed
1691
- // ones. (Pre-0.18 this was `processed.json` with two reason-keyed buckets.)
1692
- const statePathFor = (config: Config) => join(config.workspace, '_state', 'jobs.json')
1693
-
1694
- const legacyStatePathFor = (config: Config) => join(config.workspace, '_state', 'processed.json')
1695
-
1696
- /**
1697
- * Read-only load. On an un-migrated workspace this converts in memory and does
1698
- * NOT write: `vn jobs` and the GUI's poll both come through here without the run
1699
- * lock, and a write from a view could race a live `vn run`. Persisting the
1700
- * conversion is migrateStateOnDisk()'s job, under the lock.
1701
- */
1702
- // The legacy read is the one irreversible read in the codebase, so it gets the
1703
- // same strictness as the new format — `readJson` swallows a parse failure and
1704
- // returns `{}`, which here would mean "nothing was ever processed" and re-pay
1705
- // for every recording's ASR.
1706
- async function readLegacyState(config: Config): Promise<Json> {
1707
- const path = legacyStatePathFor(config)
1708
- return parseStrictJson(await readFile(path, 'utf8'), path) as Json
1709
- }
1710
-
1711
- async function loadState(config: Config): Promise<StateFile> {
1712
- const path = statePathFor(config)
1713
- if (!existsSync(path) && existsSync(legacyStatePathFor(config))) {
1714
- return migrateLegacyState(await readLegacyState(config), nowIso())
1715
- }
1716
- const store = existsSync(path) ? parseStateFile(await readFile(path, 'utf8'), path) : emptyState()
1717
- lastSavedState = JSON.stringify(store)
1718
- return store
1719
- }
1720
-
1721
- // Inline rather than a repo script: most installs are the GUI's compiled
1722
- // sidecar, which has no checkout to run a script from — and starting from empty
1723
- // is not an option, it would re-transcribe everything and pay for ASR twice.
1724
- // Call only with the run lock held.
1725
- async function migrateStateOnDisk(config: Config): Promise<void> {
1726
- const path = statePathFor(config)
1727
- const legacy = legacyStatePathFor(config)
1728
- if (existsSync(path) || !existsSync(legacy)) return
1729
- const store = migrateLegacyState(await readLegacyState(config), nowIso())
1730
- await writeJson(path, store)
1731
- lastSavedState = JSON.stringify(store)
1732
- await rename(legacy, `${legacy}.v1.bak`).catch(e => warnSideEffect('archive pre-0.18 state', e))
1733
- console.log(`Converted ${basename(legacy)} → ${basename(path)} (${Object.keys(store.jobs).length} records; old file kept as .v1.bak)`)
1734
- }
1735
-
1736
- // Workspaces are often synced folders; skip writes when a run did not change
1737
- // the state.
1738
- let lastSavedState = ''
1739
- async function saveState(config: Config, store: StateFile): Promise<void> {
1740
- const serialized = JSON.stringify(store)
1741
- if (serialized === lastSavedState) return
1742
- await writeJson(statePathFor(config), store)
1743
- lastSavedState = serialized
1744
- }
1745
-
1746
- /** Upsert the scan-time facts; never touches lifecycle fields. */
1747
- function recordFor(store: StateFile, rec: Recording): JobRecord {
1748
- const existing = store.jobs[rec.sourceId]
1749
- const next: JobRecord = existing ?? {
1750
- name: basename(rec.sourcePath), source_path: rec.sourcePath, content_hash: rec.contentHash, recorded_at: localIso(rec.recordedAt),
1751
- size_bytes: rec.sizeBytes, duration_seconds: rec.durationSeconds,
1752
- state: 'queued', code: null, detail: null, attempts: 0, updated_at: nowIso(), title: null, paths: null,
1753
- ...(rec.imported ? { origin: 'import' as const } : {}),
1754
- }
1755
- next.source_path = rec.sourcePath
1756
- next.content_hash = rec.contentHash
1757
- next.size_bytes = rec.sizeBytes
1758
- next.duration_seconds = rec.durationSeconds
1759
- next.origin = rec.imported ? 'import' : undefined
1760
- store.jobs[rec.sourceId] = next
1761
- return next
1762
- }
1763
-
1764
- // The live job, declared by the run itself. Lives next to run.lock (machine
1765
- // state, not workspace data) and carries the pid so a reader can tell a live
1766
- // job from one whose process was killed.
1767
- const CURRENT_PATH = join(STATE_DIR, 'current.json')
1768
-
1769
- function writeCurrent(sourceId: string, step: string, startedAt: string): void {
1770
- try {
1771
- mkdirSync(STATE_DIR, { recursive: true })
1772
- // tmp+rename, same rule as writeFileAtomic: this file's existence and
1773
- // contents are the live-job signal, and progressStep rewrites it at every
1774
- // step. A truncated write would read back as null and show a running job
1775
- // as queued.
1776
- const tmp = `${CURRENT_PATH}.tmp`
1777
- writeFileSync(tmp, JSON.stringify({ pid: process.pid, source_id: sourceId, step, started_at: startedAt } satisfies CurrentJob))
1778
- renameSync(tmp, CURRENT_PATH)
1779
- } catch (e) { warnSideEffect('write current job', e) }
1780
- }
1781
-
1782
- function clearCurrent(): void {
1783
- try { unlinkSync(CURRENT_PATH) } catch (e: any) { if (e?.code !== 'ENOENT') warnSideEffect('clear current job', e) }
1784
- }
1785
-
1786
- // Step reporting from inside the pipeline: a job is only "the current job" for
1787
- // as long as this run says so, so the step is written, never guessed from logs.
1788
- let currentJobId: string | null = null
1789
- let currentJobStartedAt = ''
1790
- function reportStep(step: string): void {
1791
- if (currentJobId) writeCurrent(currentJobId, step, currentJobStartedAt)
1792
- }
1793
-
1794
- function readCurrent(): CurrentJob | null {
1795
- let raw: string
1796
- try { raw = readFileSync(CURRENT_PATH, 'utf8') } catch (e: any) {
1797
- if (e?.code !== 'ENOENT') warnSideEffect('read current job', e)
1798
- return null
1799
- }
1800
- // A damaged file means a live job shows up as queued; treating it as "no job"
1801
- // is the safe read, but it must not be silent.
1802
- try {
1803
- const c = JSON.parse(raw)
1804
- if (Number.isFinite(c?.pid) && typeof c?.source_id === 'string') return c
1805
- warnSideEffect('read current job', new Error(`${CURRENT_PATH} has no pid/source_id`))
1806
- } catch (e) { warnSideEffect('read current job', e) }
1807
- return null
1808
- }
1809
-
1810
- function pidAlive(pid: number): boolean {
1811
- if (!(pid > 0)) return false
1812
- try { process.kill(pid, 0); return true } catch (e: any) { return e?.code === 'EPERM' }
1813
- }
1814
-
1815
- async function runPipeline(file: string | undefined, opts: any): Promise<void> {
1816
- wireDailyLog()
1817
- const config = getConfig()
1818
- opts = { ...opts, file }
1819
- const lock = await acquireRunLock()
1820
- if (!lock) {
1821
- console.log('voicenote pipeline already running; skip')
1822
- return
1823
- }
1824
- try {
1825
- await runPipelineLocked(config, opts)
1826
- } finally {
1827
- await lock.release()
1828
- }
1829
- }
1830
-
1831
- async function runPipelineLocked(config: Config, opts: any): Promise<void> {
1832
- await ensureDirs(config)
1833
- // --dry-run is a zero-side-effect diagnostic; the migration renames the legacy
1834
- // file and permanently drops its `error:*` entries. loadState converts in
1835
- // memory, so a dry run still sees the right picture.
1836
- if (!opts.dryRun) await migrateStateOnDisk(config)
1837
- const store = await loadState(config)
1838
- // We hold the run lock, so nothing else can own a `running` record: any that
1839
- // survive are debris from a killed run. Their attempt was already counted, so
1840
- // this is what makes the retry cap cover crashes as well as thrown errors.
1841
- const interrupted = reconcileInterrupted(store.jobs, nowIso())
1842
- if (interrupted.length) console.log(`Reclaimed ${interrupted.length} job(s) left running by an interrupted run: ${interrupted.slice(0, 3).map(j => j.name).join(', ')}`)
1843
-
1844
- // Explicit file: process exactly that path, wherever it lives. Nothing is
1845
- // scanned, so the listing is never "complete" (no pruning), and the recorder
1846
- // filters (age/size/duration) don't apply — the user named the file.
1847
- const single = opts.file ? resolve(String(opts.file)) : null
1848
- if (single && !statSync(single, { throwIfNoEntry: false })?.isFile()) throw new Error(`Not a file: ${single}`)
1849
- if (single && !isCandidateFile(single)) throw new Error(`Unsupported audio file. Use: ${[...AUDIO_EXTENSIONS].join(', ')}`)
1850
- const recorderPresent = existsSync(config.recordDir)
1851
- const { recordings, complete: scanComplete } = single
1852
- ? { recordings: [await toRecording(config, single)], complete: false }
1853
- : await scanRecordings(config)
1854
- if (!single && !recorderPresent && !recordings.length) {
1855
- if (shouldLogIdleStatus(`missing:${config.recordDir}`)) {
1856
- console.log(`Idle: recorder not mounted and no manual imports are queued: ${config.recordDir} (repeated idle logs suppressed for 30m)`)
1857
- }
1858
- return
1859
- }
1860
- const mode = normalizeRunMode(opts)
1861
- const force = Boolean(opts.force)
1862
- const eligible: Recording[] = []
1863
- const skipCounts: Record<string, number> = {}
1864
- const skipSamples: Record<string, string[]> = {}
1865
- // An explicitly named file that gets skipped must say why, not fall into the
1866
- // idle-suppressed silence meant for the 60s scheduler tick.
1867
- const verboseSkips = Boolean(opts.verbose || opts.dryRun || single)
1868
- const seen = new Set<string>()
1869
- const automaticLimits = { maxAgeHours: config.maxAgeHours, minBytes: config.minBytes, minDurationSeconds: config.minDurationSeconds }
1870
- const manualLimits = { maxAgeHours: 0, minBytes: 0, minDurationSeconds: 0 }
1871
- for (const rec of recordings) {
1872
- seen.add(rec.sourceId)
1873
- const completedDuplicate = rec.imported && !store.jobs[rec.sourceId]
1874
- ? completedJobByHash(store, rec.contentHash)
1875
- : undefined
1876
- if (completedDuplicate) {
1877
- const name = basename(rec.sourcePath)
1878
- skipCounts.already_done = (skipCounts.already_done || 0) + 1
1879
- ;(skipSamples.already_done ||= []).push(name)
1880
- if (!opts.dryRun) await removeImportedSource(rec.sourcePath)
1881
- if (verboseSkips) console.log(` Skip: ${name} (already_done)`)
1882
- continue
1883
- }
1884
- const entry = recordFor(store, rec)
1885
- const verdict = classify(rec, store.jobs[rec.sourceId], single || rec.imported ? manualLimits : automaticLimits, { force, notesMode: mode === 'notes', now: Date.now() })
1886
- if (verdict.run) { eligible.push(rec); continue }
1887
- skipCounts[verdict.code] = (skipCounts[verdict.code] || 0) + 1
1888
- ;(skipSamples[verdict.code] ||= []).push(entry.name)
1889
- if (verdict.persist) patchJob(entry, { state: 'filtered', code: verdict.code, detail: verdict.detail }, nowIso())
1890
- if (rec.imported && verdict.code === 'already_done' && !opts.dryRun) await removeImportedSource(rec.sourcePath)
1891
- if (verboseSkips) console.log(` Skip: ${entry.name} (${verdict.code}${verdict.detail ? `: ${verdict.detail}` : ''})`)
1892
- }
1893
- // Only prune against a listing we believe to be complete: if the recorder went
1894
- // away mid-glob the scan is partial, and pruning would wipe live queue entries
1895
- // (they'd return on the next scan, but their retry counters would not).
1896
- const dropped = pruneUnseen(store.jobs, seen, !single && scanComplete && existsSync(config.recordDir))
1897
- // The only routine path that deletes state — never do it silently.
1898
- if (dropped.length) console.log(`Forgot ${dropped.length} record(s) whose source is no longer on the recorder: ${dropped.slice(0, 3).map(j => j.name).join(', ')}${dropped.length > 3 ? `…(+${dropped.length - 3})` : ''}`)
1899
- const skipSummary = Object.entries(skipCounts).map(([reason, count]) => `${reason}=${count}`).join(', ') || 'none'
1900
- const scanLine = `Scan summary: found=${recordings.length}; eligible=${eligible.length}; skipped=${recordings.length - eligible.length} (${skipSummary})`
1901
- const samplesLine = !verboseSkips && Object.keys(skipSamples).length
1902
- ? `Skipped samples: ${Object.entries(skipSamples).map(([reason, names]) => `${reason}: ${names.slice(0, 3).join(', ')}${names.length > 3 ? `…(+${names.length - 3})` : ''}`).join(' | ')}`
1903
- : ''
1904
- const latestOnly = Boolean(opts.latest)
1905
- const targets = latestOnly ? eligible.slice(-1) : eligible
1906
- // Preflight: if there is work but the run cannot complete, skip BEFORE spending
1907
- // ASR money, rather than failing per-recording on every 60s StartInterval tick.
1908
- // Idle-suppressed so a misconfigured daemon doesn't spam logs. Skipped for
1909
- // --dry-run, which is a zero-side-effect diagnostic and should still print the
1910
- // plan even on an unconfigured machine.
1911
- if (targets.length && !opts.dryRun) {
1912
- const needsAsr = targets.some(rec => !resumableTranscriptFiles(config, rec, store, mode, force))
1913
- if (needsAsr && !config.volcano) {
1914
- if (shouldLogIdleStatus(`asr-misconfig:${config.recordDir}`)) console.error('ASR not configured: Volcano needs VOLCANO_ASR_KEY / VOLCANO_TOS_*. Skipping; run `vn doctor`, fix config, then re-run.')
1915
- return
1916
- }
1917
- }
1918
- if (!targets.length) {
1919
- if (verboseSkips || shouldLogIdleStatus(`idle:${config.recordDir}:${recordings.length}:${skipSummary}:${samplesLine}`)) {
1920
- console.log(scanLine)
1921
- if (samplesLine) console.log(samplesLine)
1922
- console.log('Idle: no new recordings to process. (repeated idle logs suppressed for 30m)')
1923
- }
1924
- } else {
1925
- console.log(scanLine)
1926
- if (samplesLine) console.log(samplesLine)
1927
- console.log(`Queue: processing ${targets.length} recording(s)${latestOnly ? ' (--latest)' : ''}. Remaining after this run: ${Math.max(0, eligible.length - targets.length)}`)
1928
- }
1929
- if (opts.dryRun) {
1930
- // Print the plan and touch nothing: no attempt counted, no state written.
1931
- for (const rec of targets) {
1932
- const plan = await processRecording(config, rec, { ...opts, resumeFromTranscriptFiles: resumableTranscriptFiles(config, rec, store, mode, force) })
1933
- console.log(JSON.stringify(plan, null, 2))
1934
- }
1935
- return
1936
- }
1937
- await saveState(config, store)
1938
-
1939
- for (const [targetIndex, rec] of targets.entries()) {
1940
- const entry = store.jobs[rec.sourceId]!
1941
- let importedDone = false
1942
- // --force means "start over", so it refunds the retry budget too. Without
1943
- // this it only skips one refusal: a spent record would be back at `gave_up`
1944
- // the moment this attempt failed.
1945
- if (force) patchJob(entry, { attempts: 0 }, nowIso())
1946
- currentJobId = rec.sourceId
1947
- currentJobStartedAt = nowIso()
1948
- startAttempt(entry, nowIso())
1949
- await saveState(config, store)
1950
- writeCurrent(rec.sourceId, 'starting', currentJobStartedAt)
1951
- try {
1952
- const resumeFromTranscriptFiles = resumableTranscriptFiles(config, rec, store, mode, force)
1953
- const result = await processRecording(config, rec, { ...opts, resumeFromTranscriptFiles })
1954
- applyOutcome(entry, result.status === SUMMARY_FAILED_STATUS
1955
- ? { kind: 'summary_failed', title: result.title ?? null, paths: result.final_paths ?? null, message: String(result.summary_error ?? 'summary failed; transcript saved') }
1956
- : { kind: 'done', title: result.title ?? null, paths: result.final_paths ?? null }, nowIso())
1957
- importedDone = rec.imported && result.status !== SUMMARY_FAILED_STATUS
1958
- } catch (e: any) {
1959
- const message = String(e?.message || e)
1960
- console.error(`ERROR processing ${rec.sourcePath}: ${message}`)
1961
- // Source vanished mid-run (recorder unplugged, file deleted) AND nothing
1962
- // was produced: that's not a failed job, it's a job that no longer exists.
1963
- // Drop it so it can't linger as a permanent "failed" row. A record that
1964
- // already owns output is history — same rule pruneUnseen follows — and
1965
- // deleting it would re-pay for ASR when the recorder comes back.
1966
- if (!existsSync(rec.sourcePath) && !ownsOutput(entry)) {
1967
- console.log(`Forgot ${entry.name}: source left the recorder before it produced anything`)
1968
- delete store.jobs[rec.sourceId]
1969
- } else {
1970
- applyOutcome(entry, { kind: 'failed', message }, nowIso())
1971
- }
1972
- } finally {
1973
- currentJobId = null
1974
- clearCurrent()
1975
- await saveState(config, store) // per job, not per batch: a kill -9 costs one job, not the batch
1976
- }
1977
- if (importedDone) await removeImportedSource(rec.sourcePath)
1978
- // Whole recorder went away — every remaining recorder target would fail the
1979
- // same way. Local imports do not depend on the recorder and keep running.
1980
- if (!single && !existsSync(config.recordDir) && targets.slice(targetIndex + 1).some(target => !target.imported)) {
1981
- console.error(`Recorder disappeared mid-run (${config.recordDir}); stopping. Remaining recordings stay queued.`)
1982
- break
1983
- }
1984
- }
1985
- }
1986
-
1987
- // ────────────────────────────────────────────────────────────────────────────
1988
- // LaunchAgent
1989
- // ────────────────────────────────────────────────────────────────────────────
1990
-
1991
- // Bun standalone executables embed source in a virtual FS, so import.meta.url is
1992
- // NOT a real on-disk path: "/$bunfs/..." on mac/Linux, "B:\~BUN\root\..." on
1993
- // Windows. Either marker means we're the compiled exe (run it directly via
1994
- // process.execPath); otherwise we're bun + cli.ts on disk. NOTE: matching only
1995
- // $bunfs (the old check) misfired on Windows and leaked the virtual path into the
1996
- // scheduled task's arguments.
1997
- function resolveCli(): { cliPath: string; compiled: boolean } {
1998
- const cliPath = fileURLToPath(import.meta.url)
1999
- return { cliPath, compiled: /\$bunfs|~BUN/i.test(cliPath) }
2000
- }
2001
-
2002
- function plistPath(): string {
2003
- return join(os.homedir(), 'Library', 'LaunchAgents', `${LAUNCH_AGENT_LABEL}.plist`)
2004
- }
2005
-
2006
- function xmlEscape(s: string): string {
2007
- return s.replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;').replace(/"/g, '&quot;').replace(/'/g, '&apos;')
2008
- }
2009
-
2010
- // Scheduled runs read all business settings from config.json. The plist only
2011
- // carries a fixed PATH and desktop-bundled runtime paths that do not exist in
2012
- // that file.
2013
- async function launchAgentEnv(config: Config): Promise<Record<string, string>> {
2014
- const env: Record<string, string> = {
2015
- PATH: `${os.homedir()}/.local/bin:/opt/homebrew/bin:/opt/homebrew/sbin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin`,
2016
- }
2017
- // Provenance matters here, so this reads the raw sources rather than Config:
2018
- // only paths the GUI injected into our environment (and that config.json does
2019
- // not already carry) have to be written into the plist.
2020
- const fileEnv = configFileEnv()
2021
- for (const key of ['VOICENOTE_PI_CLI', 'VOICENOTE_FFPROBE_BIN'] as const) {
2022
- if (process.env[key] && process.env[key] !== fileEnv[key]) env[key] = process.env[key]!
2023
- }
2024
- const configuredPi = config.pi.bin
2025
- if (configuredPi.startsWith('/')) env.VOICENOTE_PI_BIN = configuredPi
2026
- else {
2027
- const found = await runCommand(IS_WINDOWS ? 'where' : 'which', [configuredPi], 5000)
2028
- const path = found.code === 0 ? (found.stdout.trim().split(/\r?\n/)[0] || '') : ''
2029
- if (path && existsSync(path)) env.VOICENOTE_PI_BIN = path
2030
- }
2031
- return env
2032
- }
2033
-
2034
- async function installLaunchAgent(opts: { load?: boolean } = {}): Promise<void> {
2035
- const { cliPath, compiled } = resolveCli()
2036
- const programArgs = compiled
2037
- ? [process.execPath, 'run']
2038
- : [existsSync('/opt/homebrew/bin/bun') ? '/opt/homebrew/bin/bun' : process.execPath, cliPath, 'run']
2039
- const programArgsXml = programArgs.map(a => ` <string>${xmlEscape(a)}</string>`).join('\n')
2040
- const plist = plistPath()
2041
- await mkdir(dirname(plist), { recursive: true })
2042
- await mkdir(LOG_DIR, { recursive: true })
2043
- const env = await launchAgentEnv(getConfig())
2044
- const envEntries = Object.entries(env)
2045
- .map(([k, v]) => ` <key>${xmlEscape(k)}</key>\n <string>${xmlEscape(v)}</string>`).join('\n')
2046
- const content = `<?xml version="1.0" encoding="UTF-8"?>
2047
- <!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
2048
- <plist version="1.0">
2049
- <dict>
2050
- <key>Label</key>
2051
- <string>${LAUNCH_AGENT_LABEL}</string>
2052
- <key>ProgramArguments</key>
2053
- <array>
2054
- ${programArgsXml}
2055
- </array>
2056
- <key>RunAtLoad</key>
2057
- <true/>
2058
- <key>StartInterval</key>
2059
- <integer>60</integer>
2060
- <key>StandardOutPath</key>
2061
- <string>${LOG_DIR}/launchd.out.log</string>
2062
- <key>StandardErrorPath</key>
2063
- <string>${LOG_DIR}/launchd.err.log</string>
2064
- <key>WorkingDirectory</key>
2065
- <string>${os.homedir()}</string>
2066
- <key>EnvironmentVariables</key>
2067
- <dict>
2068
- ${envEntries}
2069
- </dict>
2070
- </dict>
2071
- </plist>
2072
- `
2073
- await writeFile(plist, content, 'utf8')
2074
- // Keep scheduler details private and tighten permissions on older plists.
2075
- await chmod(plist, 0o600)
2076
- const summary = Object.keys(env).join(', ')
2077
- console.log(`LaunchAgent written: ${plist}`)
2078
- console.log(`Embedded env keys: ${summary}`)
2079
- if (opts.load) {
2080
- const uid = process.getuid?.()
2081
- // Remove the legacy-label agent so old installs don't double-run vn.
2082
- const legacyPlist = join(os.homedir(), 'Library', 'LaunchAgents', `${LAUNCH_AGENT_LABEL_LEGACY}.plist`)
2083
- if (existsSync(legacyPlist)) {
2084
- await runCommand('launchctl', ['bootout', `gui/${uid}/${LAUNCH_AGENT_LABEL_LEGACY}`], 10000)
2085
- await unlink(legacyPlist).catch(e => warnSideEffect(`remove legacy LaunchAgent ${legacyPlist}`, e))
2086
- }
2087
- await runCommand('launchctl', ['bootout', `gui/${uid}`, plist], 10000) // ignore if not loaded
2088
- const r = await runCommand('launchctl', ['bootstrap', `gui/${uid}`, plist], 10000)
2089
- await runCommand('launchctl', ['enable', `gui/${uid}/${LAUNCH_AGENT_LABEL}`], 10000)
2090
- if (r.code === 0) console.log('LaunchAgent loaded (launchctl bootstrap).')
2091
- else console.error(`bootstrap exit ${r.code}: ${(r.stderr || r.stdout).trim().slice(0, 200)}`)
2092
- } else {
2093
- console.log(`Enable with: launchctl bootstrap gui/$(id -u) ${plist}`)
2094
- }
2095
- }
2096
-
2097
- async function uninstallLaunchAgent(): Promise<void> {
2098
- await runCommand('launchctl', ['bootout', `gui/${process.getuid?.()}`, plistPath()], 10000)
2099
- console.log(`Bootout attempted: ${plistPath()}`)
2100
- }
2101
-
2102
- // ────────────────────────────────────────────────────────────────────────────
2103
- // Windows Task Scheduler (parallel to the mac LaunchAgent above)
2104
- // ────────────────────────────────────────────────────────────────────────────
2105
-
2106
- function taskXmlPath(): string { return join(STATE_DIR, 'task.xml') }
2107
- function taskVbsPath(): string { return join(STATE_DIR, 'run-hidden.vbs') }
2108
-
2109
- // Run via the interpreter currently executing us: process.execPath is the
2110
- // absolute bun.exe (or the compiled vn.exe). Mirrors installLaunchAgent's
2111
- // compiled-vs-script detection.
2112
- function schedulerProgramArgs(): { command: string; argLine: string } {
2113
- const { cliPath, compiled } = resolveCli()
2114
- const args = compiled ? ['run'] : [cliPath, 'run']
2115
- const argLine = args.map(a => (/\s/.test(a) ? `"${a}"` : a)).join(' ')
2116
- return { command: process.execPath, argLine }
2117
- }
2118
-
2119
- async function installScheduledTask(opts: { load?: boolean } = {}): Promise<void> {
2120
- await mkdir(STATE_DIR, { recursive: true })
2121
- await mkdir(LOG_DIR, { recursive: true })
2122
- // The task carries no env (Task Scheduler has no per-task env block), so the
2123
- // bundled CLI paths the GUI injected via process env (pi runtime + cli.js +
2124
- // ffprobe) must be persisted to config.json, which `vn run` reads on startup.
2125
- // (On mac these ride in the LaunchAgent plist instead.)
2126
- const persist: Record<string, string> = {}
2127
- for (const k of ['VOICENOTE_PI_BIN', 'VOICENOTE_PI_CLI', 'VOICENOTE_FFPROBE_BIN'] as const) {
2128
- if (process.env[k]) persist[k] = process.env[k]!
2129
- }
2130
- if (Object.keys(persist).length) {
2131
- await mkdir(CONFIG_DIR, { recursive: true })
2132
- const current = loadConfigJson()
2133
- Object.assign(current, persist)
2134
- await writeConfigJson(current)
2135
- }
2136
- const { command, argLine } = schedulerProgramArgs()
2137
- // bun.exe / vn.exe are console-subsystem: an InteractiveToken task flashes a
2138
- // console window on every tick. Launch through wscript with window style 0
2139
- // (hidden). wait=True keeps wscript alive for the duration of `vn run` so
2140
- // IgnoreNew still prevents overlap, and WScript.Quit propagates vn's exit
2141
- // code so the task's Last Run Result stays meaningful. UTF-16 BOM so
2142
- // non-ASCII paths survive (wscript reads BOM-less files as ANSI).
2143
- const fullCmd = `"${command}" ${argLine}`
2144
- const vbs = `WScript.Quit CreateObject("WScript.Shell").Run("${fullCmd.replace(/"/g, '""')}", 0, True)\r\n`
2145
- await writeFile(taskVbsPath(), '\ufeff' + vbs, 'utf16le')
2146
- const wscript = join(process.env.SystemRoot || 'C:\\Windows', 'System32', 'wscript.exe')
2147
- // Register the task as the current user (DOMAIN\user; DOMAIN == machine name for
2148
- // local accounts). Without an explicit <UserId>, `schtasks /create /xml` can't tell
2149
- // who to register as and a standard (non-admin) user gets "Access is denied".
2150
- const taskUser = process.env.USERDOMAIN && process.env.USERNAME
2151
- ? `${process.env.USERDOMAIN}\\${process.env.USERNAME}`
2152
- : (process.env.USERNAME || os.userInfo().username)
2153
- // Local-time StartBoundary for the TimeTrigger (Task Scheduler wants no zone).
2154
- const n = new Date()
2155
- const startBoundary = `${n.getFullYear()}-${pad(n.getMonth() + 1)}-${pad(n.getDate())}T${pad(n.getHours())}:${pad(n.getMinutes())}:${pad(n.getSeconds())}`
2156
- // The task just runs `vn run`; config comes from config.json (vn config set /
2157
- // the GUI), so unlike the mac plist there's no env to embed. A TimeTrigger that
2158
- // repeats every PT1M (mirrors the working `schtasks /sc minute /mo 1` form; a
2159
- // LogonTrigger gave "Access is denied" for standard users) + IgnoreNew is the
2160
- // StartInterval(60)+flock equivalent.
2161
- const xml = `<?xml version="1.0" encoding="UTF-16"?>
2162
- <Task version="1.2" xmlns="http://schemas.microsoft.com/windows/2004/02/mit/task">
2163
- <RegistrationInfo>
2164
- <Description>VoiceNote: watch the recorder and process new recordings.</Description>
2165
- </RegistrationInfo>
2166
- <Triggers>
2167
- <TimeTrigger>
2168
- <StartBoundary>${startBoundary}</StartBoundary>
2169
- <Enabled>true</Enabled>
2170
- <Repetition>
2171
- <Interval>PT1M</Interval>
2172
- <StopAtDurationEnd>false</StopAtDurationEnd>
2173
- </Repetition>
2174
- </TimeTrigger>
2175
- </Triggers>
2176
- <Principals>
2177
- <Principal id="Author">
2178
- <UserId>${xmlEscape(taskUser)}</UserId>
2179
- <LogonType>InteractiveToken</LogonType>
2180
- <RunLevel>LeastPrivilege</RunLevel>
2181
- </Principal>
2182
- </Principals>
2183
- <Settings>
2184
- <MultipleInstancesPolicy>IgnoreNew</MultipleInstancesPolicy>
2185
- <DisallowStartIfOnBatteries>false</DisallowStartIfOnBatteries>
2186
- <StopIfGoingOnBatteries>false</StopIfGoingOnBatteries>
2187
- <StartWhenAvailable>true</StartWhenAvailable>
2188
- <ExecutionTimeLimit>PT2H</ExecutionTimeLimit>
2189
- <AllowHardTerminate>true</AllowHardTerminate>
2190
- <Enabled>true</Enabled>
2191
- <Hidden>false</Hidden>
2192
- </Settings>
2193
- <Actions Context="Author">
2194
- <Exec>
2195
- <Command>${xmlEscape(wscript)}</Command>
2196
- <Arguments>${xmlEscape(`//B //Nologo "${taskVbsPath()}"`)}</Arguments>
2197
- </Exec>
2198
- </Actions>
2199
- </Task>
2200
- `
2201
- const xmlPath = taskXmlPath()
2202
- // schtasks /xml wants UTF-16; prepend a BOM so non-ASCII paths survive.
2203
- await writeFile(xmlPath, '\ufeff' + xml, 'utf16le')
2204
- const r = await runCommand('schtasks', ['/create', '/tn', TASK_NAME, '/xml', xmlPath, '/f'], 15000)
2205
- if (r.code !== 0) {
2206
- console.error(`schtasks /create failed (exit ${r.code}): ${(r.stderr || r.stdout).trim()}`)
2207
- process.exitCode = 1
2208
- return
2209
- }
2210
- console.log(`Scheduled task '${TASK_NAME}' installed — runs \`vn run\` every 60s at/after logon.`)
2211
- console.log(`Command: ${command} ${argLine} (launched hidden via wscript)`)
2212
- console.log('Note: the task reads config from config.json — set it with `vn config set` (or the GUI) so the background run is configured.')
2213
- if (opts.load) await runCommand('schtasks', ['/run', '/tn', TASK_NAME], 10000)
2214
- }
2215
-
2216
- async function uninstallScheduledTask(): Promise<void> {
2217
- const r = await runCommand('schtasks', ['/delete', '/tn', TASK_NAME, '/f'], 10000)
2218
- // Remove our artifacts too: the VBS is the task's actual entry point, and a
2219
- // leftover copy could make schedulerIsCurrent misjudge a future install.
2220
- // Only when the task is actually gone — deleting the VBS while the task is
2221
- // still registered would turn every tick into a silent wscript failure.
2222
- if (r.code === 0) {
2223
- for (const p of [taskVbsPath(), taskXmlPath()]) {
2224
- try { unlinkSync(p) } catch (e: any) { if (e?.code !== 'ENOENT') warnSideEffect(`remove scheduler artifact ${p}`, e) }
2225
- }
2226
- }
2227
- console.log(r.code === 0 ? `Scheduled task '${TASK_NAME}' removed.` : `schtasks /delete: ${(r.stderr || r.stdout).trim()}`)
2228
- }
2229
-
2230
- // ── Cross-platform scheduler dispatch ──
2231
- function installScheduler(opts: { load?: boolean } = {}): Promise<void> {
2232
- return IS_WINDOWS ? installScheduledTask(opts) : installLaunchAgent(opts)
2233
- }
2234
- function uninstallScheduler(): Promise<void> {
2235
- return IS_WINDOWS ? uninstallScheduledTask() : uninstallLaunchAgent()
2236
- }
2237
- async function printSchedulerStatus(): Promise<void> {
2238
- if (IS_WINDOWS) {
2239
- const r = await runCommand('schtasks', ['/query', '/tn', TASK_NAME, '/v', '/fo', 'LIST'], 10000)
2240
- process.stdout.write(r.stdout || r.stderr || `Task '${TASK_NAME}' not found.\n`)
2241
- return
2242
- }
2243
- const r = await runCommand('launchctl', ['print', `gui/${process.getuid?.()}/${LAUNCH_AGENT_LABEL}`], 10000)
2244
- process.stdout.write(r.stdout || r.stderr)
2245
- }
2246
-
2247
- // ────────────────────────────────────────────────────────────────────────────
2248
- // Browse / debug commands
2249
- // ────────────────────────────────────────────────────────────────────────────
2250
-
2251
- async function listMeetings(opts: { month?: string }): Promise<void> {
2252
- const config = getConfig()
2253
- const month = opts.month || `${new Date().getFullYear()}-${pad(new Date().getMonth() + 1)}`
2254
- const dir = join(config.workspace, month)
2255
- if (!existsSync(dir)) {
2256
- console.log(`No notes in ${dir}`)
2257
- return
2258
- }
2259
- const entries = (await readdir(dir)).filter(f => f.endsWith('.md')).sort()
2260
- if (!entries.length) {
2261
- console.log(`No notes in ${dir}`)
2262
- return
2263
- }
2264
- for (const name of entries) {
2265
- console.log(join(dir, name))
2266
- }
2267
- }
2268
-
2269
- // One-time migration: pre-0.15.4 wrote _index/meetings.jsonl. Rename it to the new
2270
- // canonical notes.jsonl on first access so all history stays in a single file.
2271
- async function notesIndexPath(config: Config): Promise<string> {
2272
- const p = join(config.workspace, '_index', 'notes.jsonl')
2273
- const legacy = join(config.workspace, '_index', 'meetings.jsonl')
2274
- if (!existsSync(p) && existsSync(legacy)) await rename(legacy, p).catch(e => warnSideEffect(`rename ${legacy}`, e))
2275
- return p
2276
- }
2277
-
2278
- async function lastMeeting(): Promise<void> {
2279
- const config = getConfig()
2280
- const indexPath = await notesIndexPath(config)
2281
- if (!existsSync(indexPath)) {
2282
- console.log('No notes indexed yet.')
2283
- return
2284
- }
2285
- const lines = (await readFile(indexPath, 'utf8')).trim().split('\n').filter(Boolean)
2286
- const last = lines[lines.length - 1]
2287
- if (!last) {
2288
- console.log('No notes indexed yet.')
2289
- return
2290
- }
2291
- let obj: Json
2292
- try { obj = JSON.parse(last) } catch { console.log(last); return }
2293
- console.log(`Title: ${obj.title}`)
2294
- console.log(`Date: ${obj.date} ${obj.start_time || ''}-${obj.end_time || ''}`)
2295
- console.log(`Status: ${obj.status || 'unknown'}`)
2296
- console.log(`Notes: ${obj.final_paths?.notes || obj.local_paths?.notes}`)
2297
- console.log(`Transcript: ${obj.final_paths?.transcript || obj.local_paths?.transcript}`)
2298
- console.log(`Audio: ${obj.final_paths?.audio || obj.local_paths?.audio}`)
2299
- }
2300
-
2301
-
2302
- async function openTarget(arg?: string): Promise<void> {
2303
- const config = getConfig()
2304
- let target = config.workspace
2305
- if (arg === 'config') {
2306
- target = CONFIG_DIR
2307
- } else if (arg === 'logs') {
2308
- target = LOG_DIR
2309
- } else if (arg) {
2310
- // Try matching most recent file in current month containing arg.
2311
- const month = `${new Date().getFullYear()}-${pad(new Date().getMonth() + 1)}`
2312
- const dir = join(config.workspace, month)
2313
- if (existsSync(dir)) {
2314
- const matches = (await readdir(dir)).filter(f => f.includes(arg) && f.endsWith('.md'))
2315
- if (matches.length) target = join(dir, matches[matches.length - 1]!)
2316
- }
2317
- }
2318
- await openPath(target)
2319
- console.log(`open ${target}`)
2320
- }
2321
-
2322
- function completedJobByHash(store: StateFile, digest: string): [string, JobRecord] | undefined {
2323
- return Object.entries(store.jobs).find(([, job]) => job.state === 'done' && job.content_hash === digest)
2324
- }
2325
-
2326
- type ImportResult = {
2327
- status: 'queued' | 'running' | 'gave_up' | 'already_done'
2328
- id: string
2329
- name: string
2330
- title: string | null
2331
- notes: string | null
2332
- }
2333
-
2334
- async function importRecording(file: string, opts: { json?: boolean }): Promise<void> {
2335
- const source = resolve(file)
2336
- const sourceStat = await stat(source).catch(() => null)
2337
- if (!sourceStat?.isFile()) throw new Error(`Not a file: ${source}`)
2338
- if (!isCandidateFile(source)) throw new Error(`Unsupported audio file. Use: ${[...AUDIO_EXTENSIONS].join(', ')}`)
2339
-
2340
- const config = getConfig()
2341
- await ensureDirs(config)
2342
- const digest = await sha256File(source)
2343
- const id = `import:${digest}`
2344
- const store = await loadState(config)
2345
- const entry = store.jobs[id]
2346
- const doneMatch = entry?.state === 'done'
2347
- ? [id, entry] as const
2348
- : entry ? undefined : completedJobByHash(store, digest)
2349
- let result: ImportResult
2350
-
2351
- if (doneMatch) {
2352
- const [doneId, done] = doneMatch
2353
- result = { status: 'already_done', id: doneId, name: done.name, title: done.title, notes: done.paths?.notes ?? null }
2354
- } else {
2355
- const digestDir = join(inboxPathFor(config), digest)
2356
- const queuedName = (await readdir(digestDir).catch(() => []))
2357
- .find(name => isCandidateFile(name) && statSync(join(digestDir, name), { throwIfNoEntry: false })?.isFile())
2358
- const existingSource = entry?.source_path && existsSync(entry.source_path) ? entry.source_path : null
2359
- const inboxFile = existingSource ?? (queuedName ? join(digestDir, queuedName) : join(digestDir, basename(source)))
2360
- if (!existsSync(inboxFile)) {
2361
- await mkdir(dirname(inboxFile), { recursive: true })
2362
- const tmp = join(dirname(inboxFile), `.${basename(inboxFile)}.tmp-${process.pid}`)
2363
- try {
2364
- await copyFile(source, tmp)
2365
- await utimes(tmp, sourceStat.atime, sourceStat.mtime)
2366
- await rename(tmp, inboxFile)
2367
- } finally {
2368
- await unlink(tmp).catch(() => {})
2369
- }
2370
- }
2371
- result = {
2372
- status: entry?.state === 'running' ? 'running' : entry?.state === 'gave_up' ? 'gave_up' : 'queued',
2373
- id, name: entry?.name ?? basename(source), title: entry?.title ?? null, notes: entry?.paths?.notes ?? null,
2374
- }
2375
- }
2376
-
2377
- if (opts.json) console.log(JSON.stringify(result))
2378
- else if (result.status === 'already_done') console.log(`already processed: ${result.title || result.name}`)
2379
- else console.log(`${result.status}: ${result.name}`)
2380
- }
2381
-
2382
- async function removeImportedSource(path: string): Promise<void> {
2383
- try { await unlink(path) }
2384
- catch (e: any) { if (e?.code !== 'ENOENT') { warnSideEffect(`remove imported source ${path}`, e); return } }
2385
- await rmdir(dirname(path)).catch((e: any) => {
2386
- if (e?.code !== 'ENOENT' && e?.code !== 'ENOTEMPTY') warnSideEffect(`remove empty import dir ${dirname(path)}`, e)
2387
- })
2388
- }
2389
-
2390
- async function forgetRecording(needle: string): Promise<void> {
2391
- const config = getConfig()
2392
- // Under the run lock: `vn run` holds the state file in memory for the length
2393
- // of a batch and re-saves after every job, so an unlocked delete here would be
2394
- // silently resurrected by the next save.
2395
- const lock = await acquireRunLock()
2396
- if (!lock) { console.error('A voicenote run is in progress, so the state file is busy. Re-run this once it finishes (`vn jobs` shows what it is working on).'); process.exitCode = 1; return }
2397
- try {
2398
- await migrateStateOnDisk(config)
2399
- const store = await loadState(config)
2400
- let removed = 0
2401
- for (const [id, entry] of Object.entries(store.jobs)) {
2402
- if (id === needle || entry.source_path.includes(needle) || entry.name.includes(needle)) {
2403
- delete store.jobs[id]
2404
- removed++
2405
- }
2406
- }
2407
- await saveState(config, store)
2408
- console.log(`forgot ${removed} record(s)`)
2409
- } finally { await lock.release() }
2410
- }
2411
-
2412
- async function retryRecording(id: string): Promise<void> {
2413
- const config = getConfig()
2414
- const lock = await acquireRunLock()
2415
- if (!lock) throw new Error('A voicenote run is in progress. Retry once it finishes.')
2416
- try {
2417
- await migrateStateOnDisk(config)
2418
- const store = await loadState(config)
2419
- const entry = store.jobs[id]
2420
- if (!entry) throw new Error('Recording no longer exists in the processing list.')
2421
- if (!requeueFailed(entry, nowIso())) throw new Error(`Cannot retry a recording in state '${entry.state}'.`)
2422
- await saveState(config, store)
2423
- console.log(`queued ${entry.name} for retry`)
2424
- } finally { await lock.release() }
2425
- }
2426
-
2427
- async function showLog(opts: { lines?: number; follow?: boolean; err?: boolean; date?: string }): Promise<void> {
2428
- const lines = Number(opts.lines || 30)
2429
- const wanted = [opts.date ? join(LOG_DIR, `${opts.date}.log`) : dailyLogPath()]
2430
- if (opts.err) wanted.push(join(LOG_DIR, 'launchd.err.log'))
2431
- const files = wanted.filter(f => existsSync(f))
2432
- if (!files.length) {
2433
- console.log(`No log file: ${wanted.join(', ')}`)
2434
- return
2435
- }
2436
- await tailFiles(files, lines, !!opts.follow)
2437
- }
2438
-
2439
- async function showErrors(opts: { lines?: number }): Promise<void> {
2440
- if (!existsSync(LOG_DIR)) {
2441
- console.log('No logs.')
2442
- return
2443
- }
2444
- // Only the daily rolling logs (YYYY-MM-DD.log) carry timestamped [ERROR] lines;
2445
- // launchd.out.log/launchd.err.log are raw, never-truncated stdout/stderr mirrors
2446
- // that sort after dated files alphabetically ('l' > digit) and would otherwise
2447
- // crowd out the real recent logs in the slice(-3) below.
2448
- const files = (await readdir(LOG_DIR)).filter(f => /^\d{4}-\d{2}-\d{2}\.log$/.test(f)).sort().slice(-3)
2449
- if (!files.length) {
2450
- // Distinguish "no dated logs yet" (fresh install) from "scanned, no errors".
2451
- console.log('No logs.')
2452
- return
2453
- }
2454
- const lineCount = Number(opts.lines || 20)
2455
- const errors: string[] = []
2456
- for (const f of files) {
2457
- const content = await readFile(join(LOG_DIR, f), 'utf8').catch(() => '')
2458
- for (const line of content.split('\n')) {
2459
- if (line.includes('[ERROR]') || line.includes('ERROR processing')) errors.push(line)
2460
- }
2461
- }
2462
- for (const line of errors.slice(-lineCount)) console.log(line)
2463
- }
2464
-
2465
- async function upgradeSelf(): Promise<void> {
2466
- // The registry fetch needs the configured proxy: `bun add -g` only sees it if
2467
- // we pass it, because the proxy lives in config.json, not in the shell.
2468
- const env = { ...process.env, ...getConfig().childEnv }
2469
- // Plain `bun` from PATH: vn is started by bun (`#!/usr/bin/env bun`), so an
2470
- // interactive upgrade always has it. If it is somehow missing, the spawn error
2471
- // below says so instead of the command silently "failing".
2472
- // `bun add -g` upgrades in place: verified no dependency loop on npm→npm re-add
2473
- // (the steady-state upgrade path) nor on replacing an old git-ref install. No
2474
- // remove-first, so a failed add leaves the running vn intact.
2475
- console.log('$ bun add -g @fastagent-sh/voicenote')
2476
- const addCode = await new Promise<number>(res =>
2477
- spawn('bun', ['add', '-g', '@fastagent-sh/voicenote'], { stdio: 'inherit', shell: IS_WINDOWS, env })
2478
- .on('close', c => res(c ?? 1))
2479
- .on('error', (e: Error) => { console.error(`Cannot run bun: ${e.message}`); res(1) }))
2480
- if (addCode !== 0) {
2481
- console.error(`Upgrade failed: \`bun add -g @fastagent-sh/voicenote\` exited ${addCode}. Your current install is unchanged; retry later.`)
2482
- process.exitCode = 1
2483
- return
2484
- }
2485
- // Refresh the background scheduler so it points at the upgraded version. This
2486
- // process is still the OLD code in memory, so invoke the freshly installed binary
2487
- // to regenerate.
2488
- if (IS_WINDOWS) {
2489
- const installed = (await runCommand('schtasks', ['/query', '/tn', TASK_NAME], 10000)).code === 0
2490
- if (installed) {
2491
- const code = await new Promise<number>(res =>
2492
- spawn('vn', ['install-launch-agent'], { stdio: 'inherit', shell: true })
2493
- .on('close', c => res(c ?? 1)).on('error', () => res(1)))
2494
- console.log(code === 0 ? 'Scheduled task refreshed.' : 'Warning: `vn install-launch-agent` failed; re-register manually.')
2495
- }
2496
- return
2497
- }
2498
- if (existsSync(plistPath())) {
2499
- console.log('Refreshing LaunchAgent plist for the upgraded version…')
2500
- const code = await new Promise<number>(res =>
2501
- spawn('vn', ['install-launch-agent'], { stdio: 'inherit' })
2502
- .on('close', c => res(c ?? 1)).on('error', () => res(1)))
2503
- if (code !== 0) {
2504
- console.error(`Warning: \`vn install-launch-agent\` failed (exit ${code}); the LaunchAgent still points at the previous version. Ensure vn is on PATH and re-run \`vn install-launch-agent\`.`)
2505
- return
2506
- }
2507
- const uid = process.getuid?.()
2508
- await runCommand('launchctl', ['bootout', `gui/${uid}`, plistPath()], 10000) // ok if not currently loaded
2509
- const bs = await runCommand('launchctl', ['bootstrap', `gui/${uid}`, plistPath()], 10000)
2510
- if (bs.code !== 0) {
2511
- console.error(`Warning: launchctl bootstrap failed: ${(bs.stderr || bs.stdout).trim()}. Reload manually: launchctl bootstrap gui/$(id -u) ${plistPath()}`)
2512
- return
2513
- }
2514
- console.log('LaunchAgent reloaded.')
2515
- }
2516
- }
2517
-
2518
- // ────────────────────────────────────────────────────────────────────────────
2519
- // Doctor
2520
- // ────────────────────────────────────────────────────────────────────────────
2521
-
2522
- // Read up to the last `maxBytes` of a (possibly large, ever-appending) log file
2523
- // without slurping the whole thing — used to surface the agent's latest activity.
2524
- function readLogTail(path: string, maxBytes: number): string {
2525
- try {
2526
- const size = statSync(path).size
2527
- const start = Math.max(0, size - maxBytes)
2528
- const len = size - start
2529
- const fd = openSync(path, 'r')
2530
- try {
2531
- const buf = Buffer.alloc(len)
2532
- readSync(fd, buf, 0, len, start)
2533
- return buf.toString('utf8')
2534
- } finally { closeSync(fd) }
2535
- } catch { return '' }
2536
- }
2537
-
2538
- // Where the background agent's latest activity lands. mac: launchd redirects
2539
- // the agent's stdout to launchd.out.log. Windows: Task Scheduler redirects
2540
- // nothing — the agent's own daily rolling log is the only mirror of its
2541
- // output. wireDailyLog captures the log path once at process start, so a run
2542
- // spanning midnight keeps writing to its START day's file; pick the
2543
- // most-recently-modified dated log rather than today's by name, or a
2544
- // still-running cross-midnight job would look idle on the dashboard.
2545
- function agentLogPath(): string {
2546
- if (!IS_WINDOWS) return join(LOG_DIR, 'launchd.out.log')
2547
- try {
2548
- const dated = readdirSync(LOG_DIR)
2549
- .filter(f => /^\d{4}-\d{2}-\d{2}\.log$/.test(f))
2550
- .map(f => join(LOG_DIR, f))
2551
- let newest: string | null = null
2552
- let newestMs = -Infinity
2553
- for (const p of dated) {
2554
- const ms = statSync(p).mtimeMs
2555
- if (ms > newestMs) { newestMs = ms; newest = p }
2556
- }
2557
- return newest ?? dailyLogPath()
2558
- } catch { return dailyLogPath() }
2559
- }
2560
-
2561
- // Is the background scheduler installed at all (any version)? Cheaper cousin
2562
- // of schedulerIsCurrent(), used for the dashboard's installed/not-installed
2563
- // pill — mac checks the plist file, Windows must ask schtasks (there is no
2564
- // file whose existence tracks task registration).
2565
- async function schedulerInstalledAtAll(): Promise<boolean> {
2566
- if (IS_WINDOWS) return (await runCommand('schtasks', ['/query', '/tn', TASK_NAME], 10000)).code === 0
2567
- return existsSync(plistPath())
2568
- }
2569
-
2570
- // Background agent snapshot for the dashboard (LaunchAgent / Scheduled Task).
2571
- async function agentStatus() {
2572
- const logFile = agentLogPath()
2573
- let logTail: string[] = []
2574
- let logAt: string | null = null
2575
- if (existsSync(logFile)) {
2576
- try { logAt = statSync(logFile).mtime.toISOString() } catch {}
2577
- logTail = readLogTail(logFile, 16384).split('\n').map(s => s.trim()).filter(Boolean).slice(-8)
2578
- }
2579
- // `scheduler` points at the on-disk scheduler entry for `vn doctor` to show.
2580
- // mac: the plist IS the registration (its existence == installed). Windows:
2581
- // the task XML is only the staging file we wrote; registration lives in Task
2582
- // Scheduler (queried by `installed`), so the XML may lag reality — it's an
2583
- // inspection aid, not proof of registration.
2584
- return { installed: await schedulerInstalledAtAll(), scheduler: IS_WINDOWS ? taskXmlPath() : plistPath(), logAt, logTail }
2585
- }
2586
-
2587
- // Structured health/config snapshot. Single source for both `vn doctor` (text)
2588
- // and `vn doctor --json` (consumed by the GUI status dashboard).
2589
- async function collectDoctor() {
2590
- const config = getConfig()
2591
- // pi is a bun-based CLI; cold start (esp. behind a proxy) can take >5s, so
2592
- // give --version a generous timeout to avoid a false 'missing' on a healthy pi.
2593
- const piInv = piInvocation(config.pi, ['--version'])
2594
- const piCheck = await runCommand(piInv.bin, piInv.args, 15000)
2595
- const ff = await runCommand(config.ffprobeBin, ['-version'], 5000)
2596
- const v = config.volcano
2597
- const { pi } = config
2598
- return {
2599
- version: VERSION,
2600
- bun: process.versions.bun || null,
2601
- node: process.version,
2602
- recorder: { dir: config.recordDir, exists: existsSync(config.recordDir) },
2603
- workspace: config.workspace,
2604
- volcano: v
2605
- ? {
2606
- configured: true as const,
2607
- auth: 'new-console',
2608
- resourceId: v.resourceId,
2609
- tos: { bucket: v.tos.bucket, region: v.tos.region, endpoint: v.tos.endpoint, keep: v.tos.keep, accessKey: !!v.tos.accessKey, secretKey: !!v.tos.secretKey },
2610
- language: v.language ?? null,
2611
- }
2612
- : { configured: false as const },
2613
- // Provider/model/credentials are pi's own configuration; `pi.available` is
2614
- // all we can honestly report about whether a summary can run.
2615
- summary: { backend: 'pi', model: pi.model, thinking: pi.thinking, tools: pi.tools || null, contextDir: pi.tools ? pi.contextDir : null },
2616
- pi: { bin: pi.bin, version: piCheck.code === 0 ? (piCheck.stdout.trim() || piCheck.stderr.trim() || null) : null, available: piCheck.code === 0, auth: existsSync(pi.authPath), authPath: pi.authPath },
2617
- // Outbound proxy for HTTPS endpoints (updater/GitHub). The GUI reads this to
2618
- // route its own update check, so it reports the resolved value.
2619
- proxy: { url: config.childEnv.https_proxy ?? null },
2620
- identity: { self: config.speakers.self.name || null, aliases: config.speakers.self.aliases, knownCount: config.speakers.known.length },
2621
- // The thresholds that silently decide what never gets processed. Without
2622
- // them here, confirming a change to VOICENOTE_MAX_AGE_HOURS meant planting
2623
- // a test recording and watching the scan — not a reasonable way to check
2624
- // a setting.
2625
- filters: { maxAgeHours: config.maxAgeHours, minBytes: config.minBytes, minDurationSeconds: config.minDurationSeconds },
2626
- deps: { ffprobe: ff.code === 0 },
2627
- agent: await agentStatus(),
2628
- }
2629
- }
2630
-
2631
- // The dashboard/CLI view of every recording's processing status. A pure read of
2632
- // the state file `vn run` writes, grouped by jobs.ts. Nothing here rescans
2633
- // the recorder or parses logs: the queue shown IS the queue that runs, and it
2634
- // stays visible when the recorder is unplugged.
2635
- async function jobsListData(limit: number): Promise<{ items: Json[]; total: number; queued_total: number; recorder_present: boolean }> {
2636
- const config = getConfig()
2637
- const store = await loadState(config)
2638
- // One existsSync on the mount point — not the recursive glob the old pending
2639
- // section ran on every poll, and always current.
2640
- return buildJobsView(store, readCurrent(), { limit, alive: pidAlive, recorderPresent: existsSync(config.recordDir) })
2641
- }
2642
-
2643
- async function jobsList(opts: { limit?: number; json?: boolean }): Promise<void> {
2644
- let limit: number
2645
- try { limit = parseJobsLimit(opts.limit, 30) } catch (e: any) { console.error(e.message); process.exitCode = 1; return }
2646
- const data = await jobsListData(limit)
2647
- if (opts.json) { console.log(JSON.stringify(data, null, 2)); return }
2648
- if (!data.items.length) {
2649
- console.log(data.recorder_present ? 'No jobs yet.' : 'No jobs yet. (recorder not connected)')
2650
- return
2651
- }
2652
- for (const j of data.items) {
2653
- const suffix = [j.step, j.detail].filter(Boolean).join(' \u00b7 ')
2654
- console.log(`[${j.status}] ${j.title || j.name}${suffix ? ' \u00b7 ' + suffix.slice(0, 140) : ''}`)
2655
- }
2656
- // Truncation used to be silent, which is how a 126-entry backlog read as 27.
2657
- if (data.total > data.items.length) console.log(`\u2026 ${data.total - data.items.length} more (vn jobs --limit 0 to show all)`)
2658
- if (!data.recorder_present) {
2659
- console.log(data.queued_total ? `Recorder not connected \u2014 ${data.queued_total} recording(s) waiting for it.` : 'Recorder not connected.')
2660
- }
2661
- }
2662
-
2663
- async function doctor(opts: { json?: boolean } = {}): Promise<void> {
2664
- const s = await collectDoctor()
2665
- if (opts.json) { console.log(JSON.stringify(s, null, 2)); return }
2666
- console.log(`version=${s.version}`)
2667
- console.log(`bun=${s.bun || 'not-bun'}`)
2668
- console.log(`node=${s.node}`)
2669
- console.log(`recordDir=${s.recorder.dir} exists=${s.recorder.exists}`)
2670
- console.log(`workspace=${s.workspace}`)
2671
- console.log(`filters=maxAge:${s.filters.maxAgeHours > 0 ? `${s.filters.maxAgeHours}h` : 'none'} minSize:${(s.filters.minBytes / 1000).toFixed(0)}KB minDuration:${s.filters.minDurationSeconds}s`)
2672
- if (s.volcano.configured) {
2673
- console.log(`volcano.auth=${s.volcano.auth}`)
2674
- console.log(`volcano.resourceId=${s.volcano.resourceId}`)
2675
- console.log(`volcano.tos=bucket:${s.volcano.tos.bucket} region:${s.volcano.tos.region} endpoint:${s.volcano.tos.endpoint} keep:${s.volcano.tos.keep}`)
2676
- console.log(`volcano.tos.accessKey=${s.volcano.tos.accessKey ? 'loaded' : 'missing'} secretKey=${s.volcano.tos.secretKey ? 'loaded' : 'missing'}`)
2677
- if (s.volcano.language) console.log(`volcano.language=${s.volcano.language}`)
2678
- } else {
2679
- console.log(`volcano=not configured`)
2680
- }
2681
- console.log(`summaryBackend=${s.summary.backend}`)
2682
- console.log(`pi.bin=${s.pi.bin} model=${s.summary.model || "<pi's own default>"}`)
2683
- console.log(`pi.thinking=${s.summary.thinking}`)
2684
- console.log(`pi.summaryTools=${s.summary.tools || '<disabled>'}`)
2685
- if (s.summary.contextDir) console.log(`pi.contextDir=${s.summary.contextDir} (summary agent cwd + read/grep cross-reference root)`)
2686
- console.log(`pi.version=${s.pi.version || 'missing'}`)
2687
- // Neutral fact, not an instruction: an API-key user has no auth.json and needs
2688
- // nothing fixed.
2689
- console.log(`pi.auth=${s.pi.authPath} ${s.pi.auth ? '(present)' : '(missing — fine if a provider API key is set)'}`)
2690
- console.log(`defaultMode=notes`)
2691
- console.log(`proxy=${s.proxy.url || '<unset>'}`)
2692
- console.log(`speakers.self=${s.identity.self || '<unset>'}`)
2693
- console.log(`speakers.known=${s.identity.knownCount}`)
2694
- console.log(`scheduler=${s.agent.scheduler}`)
2695
- console.log(`ffprobe=${s.deps.ffprobe ? 'ok' : 'missing'}`)
2696
- }
2697
-
2698
- // Is the background scheduler installed and pointing at this binary?
2699
- async function schedulerIsCurrent(): Promise<boolean> {
2700
- const exe = process.execPath
2701
- if (IS_WINDOWS) {
2702
- if ((await runCommand('schtasks', ['/query', '/tn', TASK_NAME], 10000)).code !== 0) return false
2703
- // The task XML points at wscript; the actual CLI path lives in the VBS.
2704
- try { return readFileSync(taskVbsPath(), 'utf16le').includes(exe) } catch { return false }
2705
- }
2706
- try { return readFileSync(plistPath(), 'utf8').includes(exe) } catch { return false }
2707
- }
2708
-
2709
- async function ensureScheduler(force: boolean): Promise<{ ok: true; skipped?: boolean }> {
2710
- if (!force && await schedulerIsCurrent()) return { ok: true, skipped: true }
2711
- await installScheduler({ load: true })
2712
- return { ok: true }
2713
- }
2714
-
2715
- // ────────────────────────────────────────────────────────────────────────────
2716
- // CLI commands
2717
- // ────────────────────────────────────────────────────────────────────────────
2718
-
2719
- const cli = cac('vn')
2720
-
2721
- cli.command('run [file]', 'Scan recorder and process recordings, or process one audio file by path (Volcano ASR + pi notes)')
2722
- .option('--mode <mode>', 'Output mode: notes (default) | transcript', { default: 'notes' })
2723
- .option('--latest', 'Only process newest eligible recording')
2724
- .option('--force', 'Reprocess already processed recordings')
2725
- .option('--dry-run', 'Do not copy / transcribe / write files')
2726
- .option('--pdf', 'Also render notes to PDF (only meaningful for --mode notes)')
2727
- .option('--verbose', 'Print per-file skip details during scan')
2728
- .action(runPipeline)
2729
-
2730
- cli.command('list', 'List notes in a month')
2731
- .option('--month <YYYY-MM>', 'Month to list (default: current month)')
2732
- .action(listMeetings)
2733
-
2734
- cli.command('last', 'Print summary of most recent processed recording').action(lastMeeting)
2735
- cli.command('jobs', 'Show every recording\'s processing status (running, queued, done, failed, gave up, filtered)')
2736
- .option('--limit <n>', 'How many to list', { default: 30 })
2737
- .option('--json', 'Output as JSON (for the GUI)')
2738
- .action((opts: { limit?: number; json?: boolean }) => jobsList(opts))
2739
-
2740
-
2741
- cli.command('open [target]', 'Open notes dir, config dir (`config`), logs dir (`logs`), or a note matching the slug').action((target?: string) => openTarget(target))
2742
-
2743
- cli.command('import <file>', 'Copy one audio file into the durable manual-import queue')
2744
- .option('--json', 'Output structured status (for the GUI)')
2745
- .action((file: string, opts: { json?: boolean }) => importRecording(file, opts))
2746
- cli.command('forget <key>', 'Drop a recording\'s job record so it is queued again (a saved transcript on disk is still reused)').action((key: string) => forgetRecording(key))
2747
- cli.command('retry <id>', 'Requeue one failed recording while retaining saved outputs').action((id: string) => retryRecording(id))
2748
-
2749
- cli.command('log', 'Print the daily log (today by default)')
2750
- .option('--lines <n>', 'How many trailing lines to print', { default: 30 })
2751
- .option('-f, --follow', 'Follow the log live (tail -F)')
2752
- .option('--err', 'Also include launchd.err.log')
2753
- .option('--date <YYYY-MM-DD>', 'Show a specific day instead of today')
2754
- .action(showLog)
2755
-
2756
- cli.command('errors', 'Show recent ERROR lines from daily logs').option('--lines <n>', 'How many lines to print', { default: 20 }).action(showErrors)
2757
-
2758
- cli.command('upgrade', 'Upgrade to the latest published version via bun add -g').action(upgradeSelf)
2759
-
2760
- cli.command('doctor', 'Check environment')
2761
- .option('--json', 'Output structured status as JSON (for the GUI)')
2762
- .action((opts: { json?: boolean }) => doctor(opts))
2763
- cli.command('login', 'Sign in to ChatGPT (Codex OAuth) for the pi summary backend')
2764
- .option('--json', 'Emit machine-readable JSON events (for the GUI client)')
2765
- .option('--device-code', 'Use the device-code flow instead of the browser callback (needs the ChatGPT security-settings opt-in)')
2766
- .action((opts: { json?: boolean; deviceCode?: boolean }) => loginChatGPT(opts))
2767
- cli.command('config <action>', 'Read/write file-based config. action: get (print JSON) | set (write from stdin JSON)')
2768
- .action((action: string) => {
2769
- if (action === 'set') return configSet()
2770
- if (action === 'get') return configGet()
2771
- console.error(`Unknown config action '${action}'. Use: vn config get | vn config set`)
2772
- process.exitCode = 1
2773
- })
2774
- cli.command('install-launch-agent', 'Install background scheduler (mac LaunchAgent / Windows Task Scheduler)')
2775
- .option('--load', 'Also (re)load/start it immediately')
2776
- .action((opts: { load?: boolean }) => installScheduler(opts))
2777
- cli.command('ensure-launch-agent', 'Install the background scheduler when missing or stale')
2778
- .option('--force', 'Reinstall even when the scheduler is current')
2779
- .action((opts: { force?: boolean }) => ensureScheduler(!!opts.force))
2780
- cli.command('uninstall-launch-agent', 'Remove the background scheduler').action(uninstallScheduler)
2781
- cli.command('status', 'Print background scheduler status').action(printSchedulerStatus)
2782
-
2783
- cli.help()
2784
- cli.version(VERSION)
2785
- // Run the command ourselves so a thrown error (bad config, unreadable state
2786
- // file) reaches the user as the one line it is, not as a bun stack trace.
2787
- const parsed = cli.parse(process.argv, { run: false })
2788
- // cac prints --help/--version itself and then reports no matched command; any
2789
- // OTHER unmatched invocation is a typo, which it would ignore in silence.
2790
- if (!cli.matchedCommand && !parsed.options.help && !parsed.options.version) {
2791
- if (parsed.args.length) console.error(`vn: unknown command '${parsed.args[0]}'`)
2792
- cli.outputHelp()
2793
- process.exit(parsed.args.length ? 1 : 0)
2794
- }
2795
- try {
2796
- await cli.runMatchedCommand()
2797
- } catch (e: any) {
2798
- console.error(`vn: ${e?.message || e}`)
2799
- process.exit(1)
2800
- }