@dxos/plugin-transcription 0.7.5-main.b19bfc8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +8 -0
- package/README.md +15 -0
- package/dist/lib/browser/TranscriptContainer-G7NUUBPJ.mjs +10 -0
- package/dist/lib/browser/TranscriptContainer-G7NUUBPJ.mjs.map +7 -0
- package/dist/lib/browser/app-graph-builder-ZSSJIFY7.mjs +7 -0
- package/dist/lib/browser/app-graph-builder-ZSSJIFY7.mjs.map +7 -0
- package/dist/lib/browser/chunk-7V6ZINS5.mjs +8 -0
- package/dist/lib/browser/chunk-7V6ZINS5.mjs.map +7 -0
- package/dist/lib/browser/chunk-D4ZMHRVK.mjs +241 -0
- package/dist/lib/browser/chunk-D4ZMHRVK.mjs.map +7 -0
- package/dist/lib/browser/chunk-QIHUPVQD.mjs +80 -0
- package/dist/lib/browser/chunk-QIHUPVQD.mjs.map +7 -0
- package/dist/lib/browser/chunk-ZMES3U2D.mjs +78 -0
- package/dist/lib/browser/chunk-ZMES3U2D.mjs.map +7 -0
- package/dist/lib/browser/index.mjs +652 -0
- package/dist/lib/browser/index.mjs.map +7 -0
- package/dist/lib/browser/intent-resolver-EAR2TFK2.mjs +32 -0
- package/dist/lib/browser/intent-resolver-EAR2TFK2.mjs.map +7 -0
- package/dist/lib/browser/meta.json +1 -0
- package/dist/lib/browser/react-surface-IXF7SETX.mjs +26 -0
- package/dist/lib/browser/react-surface-IXF7SETX.mjs.map +7 -0
- package/dist/lib/browser/types/index.mjs +17 -0
- package/dist/lib/browser/types/index.mjs.map +7 -0
- package/dist/lib/node/TranscriptContainer-IKMCPYQ2.cjs +31 -0
- package/dist/lib/node/TranscriptContainer-IKMCPYQ2.cjs.map +7 -0
- package/dist/lib/node/app-graph-builder-YFBX3OTP.cjs +26 -0
- package/dist/lib/node/app-graph-builder-YFBX3OTP.cjs.map +7 -0
- package/dist/lib/node/chunk-JBN3P6ZF.cjs +97 -0
- package/dist/lib/node/chunk-JBN3P6ZF.cjs.map +7 -0
- package/dist/lib/node/chunk-SQHF7N3B.cjs +40 -0
- package/dist/lib/node/chunk-SQHF7N3B.cjs.map +7 -0
- package/dist/lib/node/chunk-UIHEGU6Q.cjs +103 -0
- package/dist/lib/node/chunk-UIHEGU6Q.cjs.map +7 -0
- package/dist/lib/node/chunk-Y2GCR535.cjs +267 -0
- package/dist/lib/node/chunk-Y2GCR535.cjs.map +7 -0
- package/dist/lib/node/index.cjs +661 -0
- package/dist/lib/node/index.cjs.map +7 -0
- package/dist/lib/node/intent-resolver-3KJZGGUQ.cjs +44 -0
- package/dist/lib/node/intent-resolver-3KJZGGUQ.cjs.map +7 -0
- package/dist/lib/node/meta.json +1 -0
- package/dist/lib/node/react-surface-L4NQQWN4.cjs +49 -0
- package/dist/lib/node/react-surface-L4NQQWN4.cjs.map +7 -0
- package/dist/lib/node/types/index.cjs +39 -0
- package/dist/lib/node/types/index.cjs.map +7 -0
- package/dist/lib/node-esm/TranscriptContainer-HJTGGLU5.mjs +11 -0
- package/dist/lib/node-esm/TranscriptContainer-HJTGGLU5.mjs.map +7 -0
- package/dist/lib/node-esm/app-graph-builder-77GQG6UF.mjs +9 -0
- package/dist/lib/node-esm/app-graph-builder-77GQG6UF.mjs.map +7 -0
- package/dist/lib/node-esm/chunk-F4J7FV6Q.mjs +10 -0
- package/dist/lib/node-esm/chunk-F4J7FV6Q.mjs.map +7 -0
- package/dist/lib/node-esm/chunk-KEG3ME4O.mjs +82 -0
- package/dist/lib/node-esm/chunk-KEG3ME4O.mjs.map +7 -0
- package/dist/lib/node-esm/chunk-TNJUQCWM.mjs +242 -0
- package/dist/lib/node-esm/chunk-TNJUQCWM.mjs.map +7 -0
- package/dist/lib/node-esm/chunk-XYU6B73V.mjs +80 -0
- package/dist/lib/node-esm/chunk-XYU6B73V.mjs.map +7 -0
- package/dist/lib/node-esm/index.mjs +653 -0
- package/dist/lib/node-esm/index.mjs.map +7 -0
- package/dist/lib/node-esm/intent-resolver-UGLYRD4X.mjs +33 -0
- package/dist/lib/node-esm/intent-resolver-UGLYRD4X.mjs.map +7 -0
- package/dist/lib/node-esm/meta.json +1 -0
- package/dist/lib/node-esm/react-surface-TX4ULXGK.mjs +27 -0
- package/dist/lib/node-esm/react-surface-TX4ULXGK.mjs.map +7 -0
- package/dist/lib/node-esm/types/index.mjs +18 -0
- package/dist/lib/node-esm/types/index.mjs.map +7 -0
- package/dist/types/src/TranscriptionPlugin.d.ts +2 -0
- package/dist/types/src/TranscriptionPlugin.d.ts.map +1 -0
- package/dist/types/src/capabilities/app-graph-builder.d.ts +180 -0
- package/dist/types/src/capabilities/app-graph-builder.d.ts.map +1 -0
- package/dist/types/src/capabilities/capabilities.d.ts +4 -0
- package/dist/types/src/capabilities/capabilities.d.ts.map +1 -0
- package/dist/types/src/capabilities/index.d.ts +181 -0
- package/dist/types/src/capabilities/index.d.ts.map +1 -0
- package/dist/types/src/capabilities/intent-resolver.d.ts +4 -0
- package/dist/types/src/capabilities/intent-resolver.d.ts.map +1 -0
- package/dist/types/src/capabilities/react-surface.d.ts +4 -0
- package/dist/types/src/capabilities/react-surface.d.ts.map +1 -0
- package/dist/types/src/components/Transcript/Transcript.d.ts +7 -0
- package/dist/types/src/components/Transcript/Transcript.d.ts.map +1 -0
- package/dist/types/src/components/Transcript/Transcript.stories.d.ts +9 -0
- package/dist/types/src/components/Transcript/Transcript.stories.d.ts.map +1 -0
- package/dist/types/src/components/Transcript/index.d.ts +2 -0
- package/dist/types/src/components/Transcript/index.d.ts.map +1 -0
- package/dist/types/src/components/TranscriptContainer.d.ts +7 -0
- package/dist/types/src/components/TranscriptContainer.d.ts.map +1 -0
- package/dist/types/src/components/index.d.ts +6 -0
- package/dist/types/src/components/index.d.ts.map +1 -0
- package/dist/types/src/components/transcription.stories.d.ts +14 -0
- package/dist/types/src/components/transcription.stories.d.ts.map +1 -0
- package/dist/types/src/hooks/index.d.ts +6 -0
- package/dist/types/src/hooks/index.d.ts.map +1 -0
- package/dist/types/src/hooks/useAudioFile.d.ts +7 -0
- package/dist/types/src/hooks/useAudioFile.d.ts.map +1 -0
- package/dist/types/src/hooks/useAudioTrack.d.ts +5 -0
- package/dist/types/src/hooks/useAudioTrack.d.ts.map +1 -0
- package/dist/types/src/hooks/useIsSpeaking.d.ts +2 -0
- package/dist/types/src/hooks/useIsSpeaking.d.ts.map +1 -0
- package/dist/types/src/hooks/useTranscriber.d.ts +10 -0
- package/dist/types/src/hooks/useTranscriber.d.ts.map +1 -0
- package/dist/types/src/hooks/useVoiceInput.d.ts +11 -0
- package/dist/types/src/hooks/useVoiceInput.d.ts.map +1 -0
- package/dist/types/src/index.d.ts +7 -0
- package/dist/types/src/index.d.ts.map +1 -0
- package/dist/types/src/meta.d.ts +11 -0
- package/dist/types/src/meta.d.ts.map +1 -0
- package/dist/types/src/sanity.test.d.ts +2 -0
- package/dist/types/src/sanity.test.d.ts.map +1 -0
- package/dist/types/src/transcriber/audio-recorder.d.ts +47 -0
- package/dist/types/src/transcriber/audio-recorder.d.ts.map +1 -0
- package/dist/types/src/transcriber/index.d.ts +4 -0
- package/dist/types/src/transcriber/index.d.ts.map +1 -0
- package/dist/types/src/transcriber/media-stream-recorder.d.ts +26 -0
- package/dist/types/src/transcriber/media-stream-recorder.d.ts.map +1 -0
- package/dist/types/src/transcriber/transcriber.d.ts +46 -0
- package/dist/types/src/transcriber/transcriber.d.ts.map +1 -0
- package/dist/types/src/transcriber/transcriber.test.d.ts +2 -0
- package/dist/types/src/transcriber/transcriber.test.d.ts.map +1 -0
- package/dist/types/src/translations.d.ts +40 -0
- package/dist/types/src/translations.d.ts.map +1 -0
- package/dist/types/src/types/index.d.ts +3 -0
- package/dist/types/src/types/index.d.ts.map +1 -0
- package/dist/types/src/types/schema.d.ts +63 -0
- package/dist/types/src/types/schema.d.ts.map +1 -0
- package/dist/types/src/types/types.d.ts +28 -0
- package/dist/types/src/types/types.d.ts.map +1 -0
- package/dist/types/src/util/array.d.ts +2 -0
- package/dist/types/src/util/array.d.ts.map +1 -0
- package/dist/types/src/util/get-time-str.d.ts +2 -0
- package/dist/types/src/util/get-time-str.d.ts.map +1 -0
- package/dist/types/src/util/index.d.ts +5 -0
- package/dist/types/src/util/index.d.ts.map +1 -0
- package/dist/types/src/util/monitor-audio-level.d.ts +6 -0
- package/dist/types/src/util/monitor-audio-level.d.ts.map +1 -0
- package/dist/types/src/util/random-queue-dxn.d.ts +3 -0
- package/dist/types/src/util/random-queue-dxn.d.ts.map +1 -0
- package/dist/types/tsconfig.tsbuildinfo +1 -0
- package/package.json +93 -0
- package/src/TranscriptionPlugin.tsx +57 -0
- package/src/capabilities/app-graph-builder.ts +7 -0
- package/src/capabilities/capabilities.ts +12 -0
- package/src/capabilities/index.ts +11 -0
- package/src/capabilities/intent-resolver.ts +24 -0
- package/src/capabilities/react-surface.tsx +21 -0
- package/src/components/Transcript/Transcript.stories.tsx +65 -0
- package/src/components/Transcript/Transcript.tsx +69 -0
- package/src/components/Transcript/index.ts +5 -0
- package/src/components/TranscriptContainer.tsx +167 -0
- package/src/components/index.ts +10 -0
- package/src/components/transcription.stories.tsx +198 -0
- package/src/hooks/index.ts +9 -0
- package/src/hooks/useAudioFile.ts +62 -0
- package/src/hooks/useAudioTrack.ts +35 -0
- package/src/hooks/useIsSpeaking.ts +77 -0
- package/src/hooks/useTranscriber.ts +62 -0
- package/src/hooks/useVoiceInput.ts +66 -0
- package/src/index.ts +11 -0
- package/src/meta.ts +17 -0
- package/src/sanity.test.ts +11 -0
- package/src/transcriber/audio-recorder.ts +56 -0
- package/src/transcriber/index.ts +7 -0
- package/src/transcriber/media-stream-recorder.ts +109 -0
- package/src/transcriber/transcriber.test.ts +230 -0
- package/src/transcriber/transcriber.ts +247 -0
- package/src/translations.ts +28 -0
- package/src/types/index.ts +6 -0
- package/src/types/schema.ts +60 -0
- package/src/types/types.ts +34 -0
- package/src/util/array.ts +14 -0
- package/src/util/get-time-str.ts +12 -0
- package/src/util/index.ts +8 -0
- package/src/util/monitor-audio-level.ts +58 -0
- package/src/util/random-queue-dxn.ts +9 -0
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2024 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import { pipe } from 'effect';
|
|
6
|
+
import React, { useCallback, useMemo, useState, type FC } from 'react';
|
|
7
|
+
|
|
8
|
+
import { chain, createIntent, LayoutAction, useIntentDispatcher } from '@dxos/app-framework';
|
|
9
|
+
import { Message } from '@dxos/artifact';
|
|
10
|
+
import { AIServiceClientImpl, DEFAULT_LLM_MODEL, MixedStreamParser, type AIServiceClient } from '@dxos/assistant';
|
|
11
|
+
import { create, getSpace, makeRef } from '@dxos/client/echo';
|
|
12
|
+
import { QueueImpl } from '@dxos/echo-db';
|
|
13
|
+
import { createStatic, isInstanceOf } from '@dxos/echo-schema';
|
|
14
|
+
import { invariant } from '@dxos/invariant';
|
|
15
|
+
import { DXN } from '@dxos/keys';
|
|
16
|
+
import { log } from '@dxos/log';
|
|
17
|
+
import { DocumentType } from '@dxos/plugin-markdown/types';
|
|
18
|
+
import { CollectionType, SpaceAction } from '@dxos/plugin-space/types';
|
|
19
|
+
import { useConfig } from '@dxos/react-client';
|
|
20
|
+
import { useEdgeClient, useQueue, type EdgeHttpClient } from '@dxos/react-edge-client';
|
|
21
|
+
import { IconButton, Toolbar, useTranslation } from '@dxos/react-ui';
|
|
22
|
+
import { ScrollContainer } from '@dxos/react-ui-components';
|
|
23
|
+
import { StackItem } from '@dxos/react-ui-stack';
|
|
24
|
+
import { TextType } from '@dxos/schema';
|
|
25
|
+
|
|
26
|
+
import { Transcript } from './Transcript';
|
|
27
|
+
import { TRANSCRIPTION_PLUGIN } from '../meta';
|
|
28
|
+
import { TranscriptBlock, type TranscriptType } from '../types';
|
|
29
|
+
|
|
30
|
+
export const TranscriptionContainer: FC<{ transcript: TranscriptType }> = ({ transcript }) => {
|
|
31
|
+
const { t } = useTranslation(TRANSCRIPTION_PLUGIN);
|
|
32
|
+
const edge = useEdgeClient();
|
|
33
|
+
const aiService = useAiServiceClient();
|
|
34
|
+
const { dispatchPromise: dispatch } = useIntentDispatcher();
|
|
35
|
+
|
|
36
|
+
const queue = useQueue<TranscriptBlock>(edge, transcript.queue ? DXN.parse(transcript.queue) : undefined, {
|
|
37
|
+
pollInterval: 1_000,
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
// TODO(dmaretskyi): Pending state and errors should be handled by the framework!!!
|
|
41
|
+
const [isSummarizing, setIsSummarizing] = useState(false);
|
|
42
|
+
const handleSummarize = useCallback(async () => {
|
|
43
|
+
setIsSummarizing(true);
|
|
44
|
+
try {
|
|
45
|
+
const document = await summarizeTranscript(edge, aiService, transcript);
|
|
46
|
+
const space = getSpace(transcript);
|
|
47
|
+
const target = space?.properties[CollectionType.typename]?.target;
|
|
48
|
+
await dispatch(
|
|
49
|
+
pipe(
|
|
50
|
+
createIntent(SpaceAction.AddObject, { object: document, target }),
|
|
51
|
+
chain(LayoutAction.Open, { part: 'main' }),
|
|
52
|
+
),
|
|
53
|
+
);
|
|
54
|
+
} finally {
|
|
55
|
+
setIsSummarizing(false);
|
|
56
|
+
}
|
|
57
|
+
}, [transcript, edge]);
|
|
58
|
+
|
|
59
|
+
// TODO(dmaretskyi): Move action to menu.
|
|
60
|
+
return (
|
|
61
|
+
<StackItem.Content toolbar={true}>
|
|
62
|
+
<StackItem.Heading>
|
|
63
|
+
<Toolbar.Root classNames='flex gap-2'>
|
|
64
|
+
<IconButton
|
|
65
|
+
icon='ph--subtitles--regular'
|
|
66
|
+
iconOnly
|
|
67
|
+
size={5}
|
|
68
|
+
disabled={isSummarizing}
|
|
69
|
+
onClick={handleSummarize}
|
|
70
|
+
label={t('summary button')}
|
|
71
|
+
/>
|
|
72
|
+
{isSummarizing && <div className='text-sm'>{t('summarizing label')}</div>}
|
|
73
|
+
</Toolbar.Root>
|
|
74
|
+
</StackItem.Heading>
|
|
75
|
+
<ScrollContainer>
|
|
76
|
+
<Transcript blocks={queue?.items} />
|
|
77
|
+
</ScrollContainer>
|
|
78
|
+
</StackItem.Content>
|
|
79
|
+
);
|
|
80
|
+
};
|
|
81
|
+
|
|
82
|
+
export default TranscriptionContainer;
|
|
83
|
+
|
|
84
|
+
// TODO(dmaretskyi): Extract to the new file once transcript refactoring has settled.
|
|
85
|
+
const summarizeTranscript = async (
|
|
86
|
+
edgeClient: EdgeHttpClient,
|
|
87
|
+
aiService: AIServiceClient,
|
|
88
|
+
transcript: TranscriptType,
|
|
89
|
+
) => {
|
|
90
|
+
invariant(transcript.queue, 'No queue found for transcript');
|
|
91
|
+
|
|
92
|
+
const queue = new QueueImpl(edgeClient, DXN.parse(transcript.queue));
|
|
93
|
+
await queue.refresh();
|
|
94
|
+
const content = queue.items
|
|
95
|
+
.filter((block) => isInstanceOf(TranscriptBlock, block))
|
|
96
|
+
.map((block) => `${block.author}: ${block.segments.map((segment) => segment.text).join('\n')}`)
|
|
97
|
+
.join('\n\n');
|
|
98
|
+
|
|
99
|
+
const parser = new MixedStreamParser();
|
|
100
|
+
|
|
101
|
+
log.info('summarizing transcript', { blockCount: queue.items.length });
|
|
102
|
+
const output = await parser.parse(
|
|
103
|
+
await aiService.generate({
|
|
104
|
+
model: DEFAULT_LLM_MODEL,
|
|
105
|
+
systemPrompt: SUMMARIZE_PROMPT,
|
|
106
|
+
history: [createStatic(Message, { role: 'user', content: [{ type: 'text', text: content }] })],
|
|
107
|
+
}),
|
|
108
|
+
);
|
|
109
|
+
|
|
110
|
+
log.info('transcript summary', { output });
|
|
111
|
+
invariant(output[0].content[0].type === 'text', 'Expected text content');
|
|
112
|
+
const summary = output[0].content[0].text;
|
|
113
|
+
|
|
114
|
+
// TODO(dmaretskyi): .started is missing.
|
|
115
|
+
const name = `Summary ${(transcript.started ?? new Date()).toLocaleString('en-US', {
|
|
116
|
+
year: 'numeric',
|
|
117
|
+
month: 'long',
|
|
118
|
+
day: 'numeric',
|
|
119
|
+
hour: '2-digit',
|
|
120
|
+
minute: '2-digit',
|
|
121
|
+
})}`;
|
|
122
|
+
|
|
123
|
+
return create(DocumentType, {
|
|
124
|
+
name,
|
|
125
|
+
content: makeRef(create(TextType, { content: summary })),
|
|
126
|
+
threads: [],
|
|
127
|
+
});
|
|
128
|
+
};
|
|
129
|
+
|
|
130
|
+
// TODO(dmaretskyi): Add example to set consistent structure for the summary.
|
|
131
|
+
const SUMMARIZE_PROMPT = `
|
|
132
|
+
# Goal
|
|
133
|
+
Create a markdown summary of the transcript provided.
|
|
134
|
+
|
|
135
|
+
# Formatting
|
|
136
|
+
- Format the summary as a markdown document without extra comments like "Here is the summary of the transcript:".
|
|
137
|
+
- Use markdown formatting for headings and bullet points.
|
|
138
|
+
- Format the summary as a list of key points and takeaways.
|
|
139
|
+
- All names of people should be in bold.
|
|
140
|
+
|
|
141
|
+
# Note Taking
|
|
142
|
+
- Correlate items in the summary with the person of origin to build a coherent narrative.
|
|
143
|
+
- Include short quotes verbatim where appropriate. Especially when concerned with design decisions and problem descriptions.
|
|
144
|
+
|
|
145
|
+
# Tasks
|
|
146
|
+
At the end of the summary include tasks.
|
|
147
|
+
Extract only the tasks that are:
|
|
148
|
+
- Directly actionable
|
|
149
|
+
- Clearly assigned to a person or team (or can easily be inferred)
|
|
150
|
+
- Strongly implied by the conversation and/or user note (no speculative tasks)
|
|
151
|
+
- Specific enough that someone reading them would know exactly what to do next
|
|
152
|
+
|
|
153
|
+
Format all tasks as markdown checkboxes using the syntax:
|
|
154
|
+
- [ ] Task description
|
|
155
|
+
|
|
156
|
+
Additional information can be included (indented).
|
|
157
|
+
|
|
158
|
+
If no actionable tasks are found, omit this tasks section.
|
|
159
|
+
`;
|
|
160
|
+
|
|
161
|
+
// TODO(dmaretskyi): Extract?
|
|
162
|
+
// Also this conflicts with plugin automation providing the ai client via a capability. Need to reconcile.
|
|
163
|
+
const useAiServiceClient = (): AIServiceClient => {
|
|
164
|
+
const config = useConfig();
|
|
165
|
+
const endpoint = config.values.runtime?.services?.ai?.server ?? 'http://localhost:8788';
|
|
166
|
+
return useMemo(() => new AIServiceClientImpl({ endpoint }), [endpoint]);
|
|
167
|
+
};
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2025 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import '@dxos-theme';
|
|
6
|
+
|
|
7
|
+
import { type StoryObj, type Meta } from '@storybook/react';
|
|
8
|
+
import React, {
|
|
9
|
+
type Dispatch,
|
|
10
|
+
type FC,
|
|
11
|
+
type SetStateAction,
|
|
12
|
+
useCallback,
|
|
13
|
+
useEffect,
|
|
14
|
+
useMemo,
|
|
15
|
+
useRef,
|
|
16
|
+
useState,
|
|
17
|
+
} from 'react';
|
|
18
|
+
|
|
19
|
+
import { createStatic } from '@dxos/echo-schema';
|
|
20
|
+
import { type DXN } from '@dxos/keys';
|
|
21
|
+
import { log } from '@dxos/log';
|
|
22
|
+
import { Config } from '@dxos/react-client';
|
|
23
|
+
import { withClientProvider } from '@dxos/react-client/testing';
|
|
24
|
+
import { useEdgeClient, useQueue } from '@dxos/react-edge-client';
|
|
25
|
+
import { IconButton, Toolbar } from '@dxos/react-ui';
|
|
26
|
+
import { ScrollContainer } from '@dxos/react-ui-components';
|
|
27
|
+
import { withTheme, withLayout } from '@dxos/storybook-utils';
|
|
28
|
+
|
|
29
|
+
import { Transcript } from './Transcript';
|
|
30
|
+
import { useTranscriber, useAudioFile, useAudioTrack } from '../hooks';
|
|
31
|
+
import { type TranscriberParams } from '../transcriber';
|
|
32
|
+
import { TranscriptBlock } from '../types';
|
|
33
|
+
import { randomQueueDxn } from '../util';
|
|
34
|
+
|
|
35
|
+
const UX: FC<{
|
|
36
|
+
playing: boolean;
|
|
37
|
+
setPlaying: Dispatch<SetStateAction<boolean>>;
|
|
38
|
+
blocks?: TranscriptBlock[];
|
|
39
|
+
}> = ({ playing, setPlaying, blocks }) => {
|
|
40
|
+
return (
|
|
41
|
+
<div className='flex flex-col w-[30rem]'>
|
|
42
|
+
<Toolbar.Root>
|
|
43
|
+
<IconButton
|
|
44
|
+
iconOnly
|
|
45
|
+
icon={playing ? 'ph--pause--regular' : 'ph--play--regular'}
|
|
46
|
+
label={playing ? 'Pause' : 'Play'}
|
|
47
|
+
onClick={() => setPlaying((playing) => !playing)}
|
|
48
|
+
/>
|
|
49
|
+
</Toolbar.Root>
|
|
50
|
+
<ScrollContainer>
|
|
51
|
+
<Transcript blocks={blocks} />
|
|
52
|
+
</ScrollContainer>
|
|
53
|
+
</div>
|
|
54
|
+
);
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
const Microphone = () => {
|
|
58
|
+
const [playing, setPlaying] = useState(false);
|
|
59
|
+
|
|
60
|
+
// Audio.
|
|
61
|
+
const track = useAudioTrack(playing);
|
|
62
|
+
|
|
63
|
+
// Queue.
|
|
64
|
+
const queueDxn = useMemo(() => randomQueueDxn(), []);
|
|
65
|
+
const echoClient = useEdgeClient();
|
|
66
|
+
const queue = useQueue<TranscriptBlock>(echoClient, queueDxn, { pollInterval: 500 });
|
|
67
|
+
|
|
68
|
+
// Transcriber.
|
|
69
|
+
const handleSegments = useCallback<TranscriberParams['onSegments']>(
|
|
70
|
+
async (segments) => {
|
|
71
|
+
const block = createStatic(TranscriptBlock, { segments });
|
|
72
|
+
queue?.append([block]);
|
|
73
|
+
},
|
|
74
|
+
[queue],
|
|
75
|
+
);
|
|
76
|
+
const transcriber = useTranscriber({ audioStreamTrack: track, onSegments: handleSegments });
|
|
77
|
+
useEffect(() => {
|
|
78
|
+
void transcriber?.open();
|
|
79
|
+
return () => {
|
|
80
|
+
void transcriber?.close();
|
|
81
|
+
};
|
|
82
|
+
}, [transcriber]);
|
|
83
|
+
|
|
84
|
+
// Manage transcription state.
|
|
85
|
+
useEffect(() => {
|
|
86
|
+
if (playing && transcriber?.isOpen) {
|
|
87
|
+
void transcriber?.startChunksRecording();
|
|
88
|
+
} else if (!playing) {
|
|
89
|
+
void transcriber?.stopChunksRecording();
|
|
90
|
+
}
|
|
91
|
+
}, [transcriber, playing, transcriber?.isOpen]);
|
|
92
|
+
|
|
93
|
+
return <UX playing={playing} setPlaying={setPlaying} blocks={queue?.items} />;
|
|
94
|
+
};
|
|
95
|
+
|
|
96
|
+
const AudioFile = ({ queueDxn, audioUrl }: { queueDxn: DXN; audioUrl: string; transcriptUrl: string }) => {
|
|
97
|
+
const [playing, setPlaying] = useState(false);
|
|
98
|
+
|
|
99
|
+
// Audio.
|
|
100
|
+
const audioElement = useRef<HTMLAudioElement>(null);
|
|
101
|
+
const { audio, stream, track } = useAudioFile(audioUrl);
|
|
102
|
+
useEffect(() => {
|
|
103
|
+
if (stream && audioElement.current) {
|
|
104
|
+
audioElement.current.srcObject = stream;
|
|
105
|
+
}
|
|
106
|
+
}, [stream, audioElement.current]);
|
|
107
|
+
|
|
108
|
+
useEffect(() => {
|
|
109
|
+
if (!audio) {
|
|
110
|
+
log.warn('no audio');
|
|
111
|
+
return;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
if (playing) {
|
|
115
|
+
void audio.play();
|
|
116
|
+
} else {
|
|
117
|
+
void audio.pause();
|
|
118
|
+
}
|
|
119
|
+
}, [audio, playing]);
|
|
120
|
+
|
|
121
|
+
// Transcriber.
|
|
122
|
+
const echoClient = useEdgeClient();
|
|
123
|
+
const queue = useQueue<TranscriptBlock>(echoClient, queueDxn, { pollInterval: 500 });
|
|
124
|
+
const handleSegments = useCallback<TranscriberParams['onSegments']>(
|
|
125
|
+
async (segments) => {
|
|
126
|
+
const block = createStatic(TranscriptBlock, { author: 'test', segments });
|
|
127
|
+
queue?.append([block]);
|
|
128
|
+
},
|
|
129
|
+
[queue],
|
|
130
|
+
);
|
|
131
|
+
|
|
132
|
+
const transcriber = useTranscriber({
|
|
133
|
+
audioStreamTrack: track,
|
|
134
|
+
onSegments: handleSegments,
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
useEffect(() => {
|
|
138
|
+
if (transcriber && playing) {
|
|
139
|
+
void transcriber.open();
|
|
140
|
+
} else if (!playing) {
|
|
141
|
+
transcriber?.stopChunksRecording();
|
|
142
|
+
void transcriber?.close();
|
|
143
|
+
}
|
|
144
|
+
}, [transcriber, playing]);
|
|
145
|
+
|
|
146
|
+
useEffect(() => {
|
|
147
|
+
if (track?.readyState === 'live' && transcriber?.isOpen) {
|
|
148
|
+
log.info('starting transcription');
|
|
149
|
+
transcriber.startChunksRecording();
|
|
150
|
+
} else if (track?.readyState !== 'live' && transcriber) {
|
|
151
|
+
log.info('stopping transcription');
|
|
152
|
+
transcriber.stopChunksRecording();
|
|
153
|
+
}
|
|
154
|
+
}, [transcriber, track?.readyState, transcriber?.isOpen]);
|
|
155
|
+
|
|
156
|
+
return <UX playing={playing} setPlaying={setPlaying} blocks={queue?.items} />;
|
|
157
|
+
};
|
|
158
|
+
|
|
159
|
+
const meta: Meta<typeof AudioFile> = {
|
|
160
|
+
title: 'plugins/plugin-transcription/transcription',
|
|
161
|
+
decorators: [
|
|
162
|
+
withClientProvider({
|
|
163
|
+
config: new Config({
|
|
164
|
+
runtime: {
|
|
165
|
+
client: { edgeFeatures: { signaling: true } },
|
|
166
|
+
services: {
|
|
167
|
+
edge: { url: 'https://edge.dxos.workers.dev/' },
|
|
168
|
+
iceProviders: [{ urls: 'https://edge.dxos.workers.dev/ice' }],
|
|
169
|
+
},
|
|
170
|
+
},
|
|
171
|
+
}),
|
|
172
|
+
}),
|
|
173
|
+
withTheme,
|
|
174
|
+
withLayout({
|
|
175
|
+
tooltips: true,
|
|
176
|
+
fullscreen: true,
|
|
177
|
+
classNames: 'justify-center',
|
|
178
|
+
}),
|
|
179
|
+
],
|
|
180
|
+
};
|
|
181
|
+
|
|
182
|
+
export default meta;
|
|
183
|
+
|
|
184
|
+
type Story = StoryObj<typeof AudioFile>;
|
|
185
|
+
|
|
186
|
+
export const Default: Story = {
|
|
187
|
+
render: Microphone,
|
|
188
|
+
};
|
|
189
|
+
|
|
190
|
+
export const File: Story = {
|
|
191
|
+
render: AudioFile,
|
|
192
|
+
args: {
|
|
193
|
+
queueDxn: randomQueueDxn(),
|
|
194
|
+
// https://learnenglish.britishcouncil.org/general-english/audio-zone/living-london
|
|
195
|
+
transcriptUrl: 'https://dxos.network/audio-london.txt',
|
|
196
|
+
audioUrl: 'https://dxos.network/audio-london.m4a',
|
|
197
|
+
},
|
|
198
|
+
};
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2025 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import { useEffect, useState } from 'react';
|
|
6
|
+
|
|
7
|
+
import { log } from '@dxos/log';
|
|
8
|
+
|
|
9
|
+
export type UseAudioState = {
|
|
10
|
+
audio?: HTMLAudioElement;
|
|
11
|
+
stream?: MediaStream;
|
|
12
|
+
track?: MediaStreamTrack;
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
export const useAudioFile = (audioUrl: string): UseAudioState => {
|
|
16
|
+
const [{ audio, stream, track }, setStream] = useState<UseAudioState>({});
|
|
17
|
+
useEffect(() => {
|
|
18
|
+
const t = setTimeout(async () => {
|
|
19
|
+
const response = await fetch(audioUrl);
|
|
20
|
+
const blob = await response.blob();
|
|
21
|
+
const audio = new Audio();
|
|
22
|
+
audio.src = URL.createObjectURL(blob);
|
|
23
|
+
await new Promise<void>((resolve, reject) => {
|
|
24
|
+
audio.addEventListener(
|
|
25
|
+
'error',
|
|
26
|
+
(err) => {
|
|
27
|
+
log.error('error', { err });
|
|
28
|
+
reject(err);
|
|
29
|
+
},
|
|
30
|
+
{ once: true },
|
|
31
|
+
);
|
|
32
|
+
audio.addEventListener(
|
|
33
|
+
'canplay',
|
|
34
|
+
() => {
|
|
35
|
+
log.info('starting...');
|
|
36
|
+
resolve();
|
|
37
|
+
},
|
|
38
|
+
{ once: true },
|
|
39
|
+
);
|
|
40
|
+
audio.load();
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
const ctx = new AudioContext();
|
|
44
|
+
|
|
45
|
+
const destination = ctx.createMediaStreamDestination();
|
|
46
|
+
destination.channelCount = 1;
|
|
47
|
+
|
|
48
|
+
const source = ctx.createMediaElementSource(audio);
|
|
49
|
+
source.connect(destination);
|
|
50
|
+
|
|
51
|
+
setStream({
|
|
52
|
+
audio,
|
|
53
|
+
stream: destination.stream,
|
|
54
|
+
track: destination.stream.getAudioTracks()[0],
|
|
55
|
+
});
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
return () => clearTimeout(t);
|
|
59
|
+
}, [audioUrl]);
|
|
60
|
+
|
|
61
|
+
return { audio, stream, track };
|
|
62
|
+
};
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2025 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import { useEffect, useState } from 'react';
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
*
|
|
9
|
+
*/
|
|
10
|
+
// TODO(burdon): Reconcile with react-ui-sfx and plugin-calls.
|
|
11
|
+
export const useAudioTrack = (active?: boolean): MediaStreamTrack | undefined => {
|
|
12
|
+
const [track, setTrack] = useState<MediaStreamTrack>();
|
|
13
|
+
useEffect(() => {
|
|
14
|
+
if (!active) {
|
|
15
|
+
track?.stop();
|
|
16
|
+
setTrack(undefined);
|
|
17
|
+
return;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
const initAudio = async () => {
|
|
21
|
+
const audio = new Audio();
|
|
22
|
+
audio.srcObject = await navigator.mediaDevices.getUserMedia({ audio: true });
|
|
23
|
+
const [track] = audio.srcObject.getAudioTracks();
|
|
24
|
+
if (track) {
|
|
25
|
+
setTrack(track);
|
|
26
|
+
}
|
|
27
|
+
};
|
|
28
|
+
|
|
29
|
+
if (active) {
|
|
30
|
+
void initAudio();
|
|
31
|
+
}
|
|
32
|
+
}, [active]);
|
|
33
|
+
|
|
34
|
+
return track;
|
|
35
|
+
};
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2025 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import { useEffect, useRef, useState } from 'react';
|
|
6
|
+
|
|
7
|
+
import { monitorAudioLevel } from '../util';
|
|
8
|
+
|
|
9
|
+
export const useIsSpeaking = (mediaStreamTrack?: MediaStreamTrack) => {
|
|
10
|
+
const [isSpeaking, setIsSpeaking] = useState(false);
|
|
11
|
+
|
|
12
|
+
// The audio level is monitored very rapidly, and we don't want
|
|
13
|
+
// react involved in tracking the state because it causes way
|
|
14
|
+
// too many re-renders. To work around this, we use the isSpeakingRef
|
|
15
|
+
// to track the state and then sync it using another effect.
|
|
16
|
+
const isSpeakingRef = useRef(isSpeaking);
|
|
17
|
+
|
|
18
|
+
// this effect syncs the state on a 50ms interval
|
|
19
|
+
useEffect(() => {
|
|
20
|
+
if (!mediaStreamTrack) {
|
|
21
|
+
setIsSpeaking(false);
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
isSpeakingRef.current = isSpeaking;
|
|
26
|
+
const interval = window.setInterval(() => {
|
|
27
|
+
// state is already in sync do nothing
|
|
28
|
+
if (isSpeaking === isSpeakingRef.current) {
|
|
29
|
+
return;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// sync state
|
|
33
|
+
setIsSpeaking(isSpeakingRef.current);
|
|
34
|
+
}, 50);
|
|
35
|
+
|
|
36
|
+
return () => {
|
|
37
|
+
clearInterval(interval);
|
|
38
|
+
};
|
|
39
|
+
}, [isSpeaking, isSpeakingRef, mediaStreamTrack]);
|
|
40
|
+
|
|
41
|
+
useEffect(() => {
|
|
42
|
+
if (!mediaStreamTrack) {
|
|
43
|
+
return;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
let timeout = -1;
|
|
47
|
+
const cleanup = monitorAudioLevel({
|
|
48
|
+
mediaStreamTrack,
|
|
49
|
+
onMeasure: (vol) => {
|
|
50
|
+
// Once the user has been determined to be speaking, we want
|
|
51
|
+
// to lower the threshold because speech patterns don't always
|
|
52
|
+
// kick up above 0.05
|
|
53
|
+
const audioLevelAboveThreshold = vol > (isSpeakingRef.current ? 0.02 : 0.05);
|
|
54
|
+
if (audioLevelAboveThreshold) {
|
|
55
|
+
// user is still speaking, clear timeout & reset.
|
|
56
|
+
clearTimeout(timeout);
|
|
57
|
+
timeout = -1;
|
|
58
|
+
// track state.
|
|
59
|
+
isSpeakingRef.current = true;
|
|
60
|
+
} else if (timeout === -1) {
|
|
61
|
+
// user is not speaking and timeout is not set.
|
|
62
|
+
timeout = window.setTimeout(() => {
|
|
63
|
+
isSpeakingRef.current = false;
|
|
64
|
+
// reset timeout
|
|
65
|
+
timeout = -1;
|
|
66
|
+
}, 1_000);
|
|
67
|
+
}
|
|
68
|
+
},
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
return () => {
|
|
72
|
+
cleanup();
|
|
73
|
+
};
|
|
74
|
+
}, [isSpeaking, mediaStreamTrack]);
|
|
75
|
+
|
|
76
|
+
return isSpeaking;
|
|
77
|
+
};
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2025 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import { useEffect, useMemo } from 'react';
|
|
6
|
+
|
|
7
|
+
import { MediaStreamRecorder, Transcriber, type TranscriberParams } from '../transcriber';
|
|
8
|
+
|
|
9
|
+
// TODO(burdon): Move to config?
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Length of the chunk in ms.
|
|
13
|
+
*/
|
|
14
|
+
const RECORD_INTERVAL = 200;
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Number of chunks to save before the user starts speaking.
|
|
18
|
+
*/
|
|
19
|
+
const PREFIXED_CHUNKS_AMOUNT = 10;
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Number of chunks to transcribe automatically after.
|
|
23
|
+
* Combined should be mess than 25MB or whisper would fail.
|
|
24
|
+
*/
|
|
25
|
+
const TRANSCRIBE_AFTER_CHUNKS_AMOUNT = 50;
|
|
26
|
+
|
|
27
|
+
export type UseTranscriberProps = {
|
|
28
|
+
audioStreamTrack?: MediaStreamTrack;
|
|
29
|
+
onSegments?: TranscriberParams['onSegments'];
|
|
30
|
+
};
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Records audio while user is speaking and transcribes it after user is done speaking.
|
|
34
|
+
*/
|
|
35
|
+
export const useTranscriber = ({ audioStreamTrack, onSegments }: UseTranscriberProps) => {
|
|
36
|
+
// Initialize audio transcription.
|
|
37
|
+
const transcriber = useMemo<Transcriber | undefined>(() => {
|
|
38
|
+
if (!onSegments || !audioStreamTrack) {
|
|
39
|
+
return undefined;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
return new Transcriber({
|
|
43
|
+
config: {
|
|
44
|
+
transcribeAfterChunksAmount: TRANSCRIBE_AFTER_CHUNKS_AMOUNT,
|
|
45
|
+
prefixBufferChunksAmount: PREFIXED_CHUNKS_AMOUNT,
|
|
46
|
+
},
|
|
47
|
+
recorder: new MediaStreamRecorder({
|
|
48
|
+
mediaStreamTrack: audioStreamTrack,
|
|
49
|
+
interval: RECORD_INTERVAL,
|
|
50
|
+
}),
|
|
51
|
+
onSegments,
|
|
52
|
+
});
|
|
53
|
+
}, [audioStreamTrack, onSegments]);
|
|
54
|
+
|
|
55
|
+
useEffect(() => {
|
|
56
|
+
return () => {
|
|
57
|
+
void transcriber?.close();
|
|
58
|
+
};
|
|
59
|
+
}, [transcriber]);
|
|
60
|
+
|
|
61
|
+
return transcriber;
|
|
62
|
+
};
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
//
|
|
2
|
+
// Copyright 2025 DXOS.org
|
|
3
|
+
//
|
|
4
|
+
|
|
5
|
+
import { useState, useEffect, useCallback } from 'react';
|
|
6
|
+
|
|
7
|
+
import { scheduleMicroTask } from '@dxos/async';
|
|
8
|
+
import { Context } from '@dxos/context';
|
|
9
|
+
import { log } from '@dxos/log';
|
|
10
|
+
import { useSoundEffect } from '@dxos/react-ui-sfx';
|
|
11
|
+
|
|
12
|
+
import { useAudioTrack } from './useAudioTrack';
|
|
13
|
+
import { useTranscriber } from './useTranscriber';
|
|
14
|
+
import { type TranscriberParams } from '../transcriber';
|
|
15
|
+
|
|
16
|
+
export type UseVoiceInputProps = {
|
|
17
|
+
active?: boolean;
|
|
18
|
+
onUpdate: (text: string) => void;
|
|
19
|
+
};
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Hook for voice input.
|
|
23
|
+
*/
|
|
24
|
+
export const useVoiceInput = ({ active, onUpdate }: UseVoiceInputProps) => {
|
|
25
|
+
const soundStart = useSoundEffect('StartRecording');
|
|
26
|
+
const soundStop = useSoundEffect('StopRecording');
|
|
27
|
+
|
|
28
|
+
const handleSegments = useCallback<TranscriberParams['onSegments']>(async (segments) => {
|
|
29
|
+
const text = segments.map((str) => str.text.trim().replace(/[^\w\s]/g, '')).join(' ');
|
|
30
|
+
onUpdate(text);
|
|
31
|
+
}, []);
|
|
32
|
+
|
|
33
|
+
// Audio/transcription.
|
|
34
|
+
const [recording, setRecording] = useState(false);
|
|
35
|
+
const track = useAudioTrack(active);
|
|
36
|
+
const transcriber = useTranscriber({
|
|
37
|
+
audioStreamTrack: track,
|
|
38
|
+
onSegments: handleSegments,
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
// Start/stop transcription.
|
|
42
|
+
useEffect(() => {
|
|
43
|
+
const ctx = new Context();
|
|
44
|
+
scheduleMicroTask(ctx, async () => {
|
|
45
|
+
if (active && transcriber) {
|
|
46
|
+
await transcriber.open();
|
|
47
|
+
log.info('starting...');
|
|
48
|
+
setRecording(true);
|
|
49
|
+
void soundStart.play();
|
|
50
|
+
transcriber.startChunksRecording();
|
|
51
|
+
} else if (!active && transcriber?.isOpen) {
|
|
52
|
+
transcriber?.stopChunksRecording();
|
|
53
|
+
await transcriber?.close();
|
|
54
|
+
log.info('stopped');
|
|
55
|
+
setRecording(false);
|
|
56
|
+
void soundStop.play();
|
|
57
|
+
}
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
return () => {
|
|
61
|
+
void ctx.dispose();
|
|
62
|
+
};
|
|
63
|
+
}, [active, transcriber]);
|
|
64
|
+
|
|
65
|
+
return { recording };
|
|
66
|
+
};
|