@dxos/plugin-transcription 0.8.3 → 0.8.4-main.28f8d3d
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/lib/browser/TranscriptContainer-FH7VAWYA.mjs +11 -0
- package/dist/lib/browser/{chunk-GOF7MJNV.mjs → chunk-5WH45KBY.mjs} +2 -2
- package/dist/lib/browser/{chunk-53AXUEIQ.mjs → chunk-BJWYA2BU.mjs} +332 -314
- package/dist/lib/browser/chunk-BJWYA2BU.mjs.map +7 -0
- package/dist/lib/browser/{chunk-IQ5ZRMHZ.mjs → chunk-KJIQQHJB.mjs} +45 -34
- package/dist/lib/browser/chunk-KJIQQHJB.mjs.map +7 -0
- package/dist/lib/browser/{chunk-OSISJM2F.mjs → chunk-KWOOVS5G.mjs} +4 -4
- package/dist/lib/browser/{chunk-OSISJM2F.mjs.map → chunk-KWOOVS5G.mjs.map} +2 -2
- package/dist/lib/browser/{chunk-W2KNYJD6.mjs → chunk-SREIAQSP.mjs} +3 -3
- package/dist/lib/browser/{chunk-W2KNYJD6.mjs.map → chunk-SREIAQSP.mjs.map} +1 -1
- package/dist/lib/browser/{chunk-UDCCT32F.mjs → chunk-UKUSJTQW.mjs} +3 -3
- package/dist/lib/browser/index.mjs +12 -12
- package/dist/lib/browser/index.mjs.map +3 -3
- package/dist/lib/browser/{intent-resolver-BGADCHVM.mjs → intent-resolver-AOVQ7JI4.mjs} +4 -4
- package/dist/lib/{node-esm/intent-resolver-5LVCYSJS.mjs.map → browser/intent-resolver-AOVQ7JI4.mjs.map} +2 -2
- package/dist/lib/browser/meta.json +1 -1
- package/dist/lib/browser/{react-surface-HMJS5FSS.mjs → react-surface-4MEJMKYQ.mjs} +7 -7
- package/dist/lib/browser/{transcriber-OYPLZFG5.mjs → transcriber-V77CN7XX.mjs} +6 -6
- package/dist/lib/browser/transcriber-V77CN7XX.mjs.map +7 -0
- package/dist/lib/browser/types/index.mjs +2 -2
- package/dist/lib/node-esm/{TranscriptContainer-G75LDEO5.mjs → TranscriptContainer-GEUOOYHX.mjs} +4 -4
- package/dist/lib/node-esm/{chunk-PRL64SRU.mjs → chunk-A33GNR44.mjs} +45 -34
- package/dist/lib/node-esm/chunk-A33GNR44.mjs.map +7 -0
- package/dist/lib/node-esm/{chunk-FH75X4JC.mjs → chunk-IKQMDRY6.mjs} +4 -4
- package/dist/lib/node-esm/{chunk-FH75X4JC.mjs.map → chunk-IKQMDRY6.mjs.map} +2 -2
- package/dist/lib/node-esm/{chunk-MFJO7EKD.mjs → chunk-PHP7MCAY.mjs} +3 -3
- package/dist/lib/node-esm/{chunk-MFJPEIVB.mjs → chunk-TM2ASOZA.mjs} +3 -3
- package/dist/lib/node-esm/{chunk-MFJPEIVB.mjs.map → chunk-TM2ASOZA.mjs.map} +1 -1
- package/dist/lib/node-esm/{chunk-TXMDLLBR.mjs → chunk-XMUCZBP3.mjs} +2 -2
- package/dist/lib/node-esm/{chunk-56QHOX4A.mjs → chunk-Y7PCJ34K.mjs} +332 -314
- package/dist/lib/node-esm/chunk-Y7PCJ34K.mjs.map +7 -0
- package/dist/lib/node-esm/index.mjs +12 -12
- package/dist/lib/node-esm/index.mjs.map +3 -3
- package/dist/lib/node-esm/{intent-resolver-5LVCYSJS.mjs → intent-resolver-P4XPIKQF.mjs} +4 -4
- package/dist/lib/{browser/intent-resolver-BGADCHVM.mjs.map → node-esm/intent-resolver-P4XPIKQF.mjs.map} +2 -2
- package/dist/lib/node-esm/meta.json +1 -1
- package/dist/lib/node-esm/{react-surface-JGQSYGRR.mjs → react-surface-EZUPNHUC.mjs} +7 -7
- package/dist/lib/node-esm/{transcriber-TVG2MBBN.mjs → transcriber-3SX3DNLV.mjs} +6 -6
- package/dist/lib/node-esm/transcriber-3SX3DNLV.mjs.map +7 -0
- package/dist/lib/node-esm/types/index.mjs +2 -2
- package/dist/types/src/capabilities/capabilities.d.ts +3 -3
- package/dist/types/src/capabilities/capabilities.d.ts.map +1 -1
- package/dist/types/src/capabilities/intent-resolver.d.ts.map +1 -1
- package/dist/types/src/capabilities/transcriber.d.ts.map +1 -1
- package/dist/types/src/components/Transcript/Transcript.d.ts.map +1 -1
- package/dist/types/src/components/Transcript/Transcript.stories.d.ts +1 -1
- package/dist/types/src/components/Transcript/Transcript.stories.d.ts.map +1 -1
- package/dist/types/src/components/TranscriptContainer.d.ts +4 -3
- package/dist/types/src/components/TranscriptContainer.d.ts.map +1 -1
- package/dist/types/src/components/index.d.ts +1 -4
- package/dist/types/src/components/index.d.ts.map +1 -1
- package/dist/types/src/components/stories/FileTranscription.stories.d.ts +1 -1
- package/dist/types/src/components/stories/FileTranscription.stories.d.ts.map +1 -1
- package/dist/types/src/components/stories/KeyWordRecognition.stories.d.ts +1 -1
- package/dist/types/src/components/stories/KeyWordRecognition.stories.d.ts.map +1 -1
- package/dist/types/src/components/stories/MicrophoneTranscription.stories.d.ts +1 -1
- package/dist/types/src/components/stories/MicrophoneTranscription.stories.d.ts.map +1 -1
- package/dist/types/src/hooks/useTranscriber.d.ts +1 -1
- package/dist/types/src/hooks/useVoiceInput.d.ts.map +1 -1
- package/dist/types/src/segments-normalization/normalization.d.ts +159 -66
- package/dist/types/src/segments-normalization/normalization.d.ts.map +1 -1
- package/dist/types/src/testing/testing.d.ts +1 -1
- package/dist/types/src/testing/testing.d.ts.map +1 -1
- package/dist/types/src/transcriber/index.d.ts +1 -1
- package/dist/types/src/transcriber/index.d.ts.map +1 -1
- package/dist/types/src/transcriber/media-stream-recorder.d.ts +1 -1
- package/dist/types/src/transcriber/media-stream-recorder.d.ts.map +1 -1
- package/dist/types/src/transcriber/transcriber.d.ts +1 -1
- package/dist/types/src/transcriber/transcriber.d.ts.map +1 -1
- package/dist/types/src/translations.d.ts +11 -11
- package/dist/types/src/translations.d.ts.map +1 -1
- package/dist/types/src/types/types.d.ts.map +1 -1
- package/dist/types/tsconfig.tsbuildinfo +1 -1
- package/package.json +59 -57
- package/src/TranscriptionPlugin.tsx +1 -1
- package/src/capabilities/capabilities.ts +5 -5
- package/src/capabilities/intent-resolver.ts +2 -2
- package/src/capabilities/transcriber.ts +3 -2
- package/src/components/Transcript/Transcript.stories.tsx +5 -4
- package/src/components/Transcript/Transcript.tsx +5 -4
- package/src/components/Transcript/deprecated/OldTranscript.tsx +2 -2
- package/src/components/Transcript/transcript-extension.ts +2 -2
- package/src/components/TranscriptContainer.tsx +10 -4
- package/src/components/stories/FileTranscription.stories.tsx +23 -19
- package/src/components/stories/KeyWordRecognition.stories.tsx +1 -1
- package/src/components/stories/MicrophoneTranscription.stories.tsx +18 -15
- package/src/hooks/useTranscriber.ts +5 -5
- package/src/hooks/useVoiceInput.ts +3 -2
- package/src/model/model.test.ts +9 -9
- package/src/segments-normalization/message-normalizer.ts +1 -1
- package/src/segments-normalization/normalization.test.ts +19 -21
- package/src/segments-normalization/normalization.ts +54 -44
- package/src/testing/testing.ts +11 -11
- package/src/transcriber/index.ts +1 -1
- package/src/transcriber/media-stream-recorder.ts +1 -1
- package/src/transcriber/transcriber.browser.test.ts +4 -3
- package/src/transcriber/transcriber.ts +5 -7
- package/src/transcriber/transcription-manager.ts +3 -3
- package/src/translations.ts +6 -4
- package/src/types/types.ts +2 -1
- package/dist/lib/browser/TranscriptContainer-H3AZQ4CQ.mjs +0 -11
- package/dist/lib/browser/chunk-53AXUEIQ.mjs.map +0 -7
- package/dist/lib/browser/chunk-IQ5ZRMHZ.mjs.map +0 -7
- package/dist/lib/browser/transcriber-OYPLZFG5.mjs.map +0 -7
- package/dist/lib/node/TranscriptContainer-KIG3GMAB.cjs +0 -32
- package/dist/lib/node/TranscriptContainer-KIG3GMAB.cjs.map +0 -7
- package/dist/lib/node/chunk-3PYT3BDM.cjs +0 -40
- package/dist/lib/node/chunk-3PYT3BDM.cjs.map +0 -7
- package/dist/lib/node/chunk-6LW5RA3K.cjs +0 -75
- package/dist/lib/node/chunk-6LW5RA3K.cjs.map +0 -7
- package/dist/lib/node/chunk-AFUZI63A.cjs +0 -42
- package/dist/lib/node/chunk-AFUZI63A.cjs.map +0 -7
- package/dist/lib/node/chunk-CVUN3UO7.cjs +0 -778
- package/dist/lib/node/chunk-CVUN3UO7.cjs.map +0 -7
- package/dist/lib/node/chunk-FKDJNJ57.cjs +0 -35
- package/dist/lib/node/chunk-FKDJNJ57.cjs.map +0 -7
- package/dist/lib/node/chunk-ZFAUMZXE.cjs +0 -518
- package/dist/lib/node/chunk-ZFAUMZXE.cjs.map +0 -7
- package/dist/lib/node/index.cjs +0 -133
- package/dist/lib/node/index.cjs.map +0 -7
- package/dist/lib/node/intent-resolver-LL3TTI4H.cjs +0 -45
- package/dist/lib/node/intent-resolver-LL3TTI4H.cjs.map +0 -7
- package/dist/lib/node/meta.json +0 -1
- package/dist/lib/node/react-surface-N25TBBG4.cjs +0 -56
- package/dist/lib/node/react-surface-N25TBBG4.cjs.map +0 -7
- package/dist/lib/node/transcriber-6J4FVFSZ.cjs +0 -68
- package/dist/lib/node/transcriber-6J4FVFSZ.cjs.map +0 -7
- package/dist/lib/node/types/index.cjs +0 -36
- package/dist/lib/node/types/index.cjs.map +0 -7
- package/dist/lib/node-esm/chunk-56QHOX4A.mjs.map +0 -7
- package/dist/lib/node-esm/chunk-PRL64SRU.mjs.map +0 -7
- package/dist/lib/node-esm/transcriber-TVG2MBBN.mjs.map +0 -7
- /package/dist/lib/browser/{TranscriptContainer-H3AZQ4CQ.mjs.map → TranscriptContainer-FH7VAWYA.mjs.map} +0 -0
- /package/dist/lib/browser/{chunk-GOF7MJNV.mjs.map → chunk-5WH45KBY.mjs.map} +0 -0
- /package/dist/lib/browser/{chunk-UDCCT32F.mjs.map → chunk-UKUSJTQW.mjs.map} +0 -0
- /package/dist/lib/browser/{react-surface-HMJS5FSS.mjs.map → react-surface-4MEJMKYQ.mjs.map} +0 -0
- /package/dist/lib/node-esm/{TranscriptContainer-G75LDEO5.mjs.map → TranscriptContainer-GEUOOYHX.mjs.map} +0 -0
- /package/dist/lib/node-esm/{chunk-MFJO7EKD.mjs.map → chunk-PHP7MCAY.mjs.map} +0 -0
- /package/dist/lib/node-esm/{chunk-TXMDLLBR.mjs.map → chunk-XMUCZBP3.mjs.map} +0 -0
- /package/dist/lib/node-esm/{react-surface-JGQSYGRR.mjs.map → react-surface-EZUPNHUC.mjs.map} +0 -0
package/src/model/model.test.ts
CHANGED
|
@@ -9,7 +9,7 @@ import { Obj } from '@dxos/echo';
|
|
|
9
9
|
import type { ObjectId } from '@dxos/keys';
|
|
10
10
|
import { DataType } from '@dxos/schema';
|
|
11
11
|
|
|
12
|
-
import {
|
|
12
|
+
import { type ChunkRenderer, DocumentAdapter, SerializationModel } from './model';
|
|
13
13
|
|
|
14
14
|
const blockToMarkdown: ChunkRenderer<DataType.Message> = (
|
|
15
15
|
message: DataType.Message,
|
|
@@ -18,7 +18,7 @@ const blockToMarkdown: ChunkRenderer<DataType.Message> = (
|
|
|
18
18
|
): string[] => {
|
|
19
19
|
return [
|
|
20
20
|
`###### ${message.sender.name}`,
|
|
21
|
-
...message.blocks.filter((block) => block.
|
|
21
|
+
...message.blocks.filter((block) => block._tag === 'transcript').map((block) => block.text),
|
|
22
22
|
'',
|
|
23
23
|
];
|
|
24
24
|
};
|
|
@@ -37,7 +37,7 @@ describe('SerializationModel', () => {
|
|
|
37
37
|
sender: { name: 'Alice' },
|
|
38
38
|
blocks: [
|
|
39
39
|
{
|
|
40
|
-
|
|
40
|
+
_tag: 'transcript',
|
|
41
41
|
started: createDate(),
|
|
42
42
|
text: 'Hello world!',
|
|
43
43
|
},
|
|
@@ -51,7 +51,7 @@ describe('SerializationModel', () => {
|
|
|
51
51
|
}
|
|
52
52
|
|
|
53
53
|
// Update message.
|
|
54
|
-
message.blocks.push({
|
|
54
|
+
message.blocks.push({ _tag: 'transcript', started: createDate(), text: 'Hello again!' });
|
|
55
55
|
model.updateChunk(message);
|
|
56
56
|
{
|
|
57
57
|
const text = model.doc.toString();
|
|
@@ -74,7 +74,7 @@ describe('SerializationModel', () => {
|
|
|
74
74
|
sender: { name: 'Alice' },
|
|
75
75
|
blocks: [
|
|
76
76
|
{
|
|
77
|
-
|
|
77
|
+
_tag: 'transcript',
|
|
78
78
|
started: createDate(),
|
|
79
79
|
text: 'Hello world!',
|
|
80
80
|
},
|
|
@@ -92,7 +92,7 @@ describe('SerializationModel', () => {
|
|
|
92
92
|
sender: { name: 'Bob' },
|
|
93
93
|
blocks: [
|
|
94
94
|
{
|
|
95
|
-
|
|
95
|
+
_tag: 'transcript',
|
|
96
96
|
started: createDate(),
|
|
97
97
|
text: 'Hello world!',
|
|
98
98
|
},
|
|
@@ -119,7 +119,7 @@ describe('SerializationModel', () => {
|
|
|
119
119
|
sender: { name: 'Alice' },
|
|
120
120
|
blocks: [
|
|
121
121
|
{
|
|
122
|
-
|
|
122
|
+
_tag: 'transcript',
|
|
123
123
|
started: createDate(),
|
|
124
124
|
text: 'Hello world!',
|
|
125
125
|
},
|
|
@@ -131,7 +131,7 @@ describe('SerializationModel', () => {
|
|
|
131
131
|
expect(view.state.doc.toString()).to.eq('###### Alice\nHello world!\n\n');
|
|
132
132
|
|
|
133
133
|
// Update message (add block).
|
|
134
|
-
message.blocks.push({
|
|
134
|
+
message.blocks.push({ _tag: 'transcript', started: createDate(), text: 'Hello again!' });
|
|
135
135
|
model.updateChunk(message);
|
|
136
136
|
model.sync(adapter);
|
|
137
137
|
expect(view.state.doc.toString()).to.eq('###### Alice\nHello world!\nHello again!\n\n');
|
|
@@ -150,7 +150,7 @@ describe('SerializationModel', () => {
|
|
|
150
150
|
const message = Obj.make(DataType.Message, {
|
|
151
151
|
created: createDate(),
|
|
152
152
|
sender: { name: 'Bob' },
|
|
153
|
-
blocks: [{
|
|
153
|
+
blocks: [{ _tag: 'transcript', started: createDate(), text: 'Hello again!' }],
|
|
154
154
|
});
|
|
155
155
|
model.appendChunk(message);
|
|
156
156
|
model.sync(adapter);
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
|
|
9
9
|
import { effect } from '@preact/signals-core';
|
|
10
10
|
|
|
11
|
-
import {
|
|
11
|
+
import { DeferredTask, asyncTimeout } from '@dxos/async';
|
|
12
12
|
import { LifecycleState, Resource } from '@dxos/context';
|
|
13
13
|
import { type Queue } from '@dxos/echo-db';
|
|
14
14
|
import { type FunctionExecutor } from '@dxos/functions';
|
|
@@ -5,8 +5,6 @@
|
|
|
5
5
|
import { effect } from '@preact/signals-core';
|
|
6
6
|
import { describe, test } from 'vitest';
|
|
7
7
|
|
|
8
|
-
import { EdgeAiServiceClient, OllamaAiServiceClient } from '@dxos/ai';
|
|
9
|
-
import { AI_SERVICE_ENDPOINT } from '@dxos/ai/testing';
|
|
10
8
|
import { scheduleTaskInterval } from '@dxos/async';
|
|
11
9
|
import { Context } from '@dxos/context';
|
|
12
10
|
import { Obj } from '@dxos/echo';
|
|
@@ -51,33 +49,32 @@ const messages: MessageWithRangeId[] = [
|
|
|
51
49
|
'in classical physics objects have well-defined properties such as position speed and momentum',
|
|
52
50
|
].map((string, index) =>
|
|
53
51
|
Obj.make(DataType.Message, {
|
|
54
|
-
created: new Date(Date.now() +
|
|
52
|
+
created: new Date(Date.now() + 1_000 * index).toISOString(),
|
|
55
53
|
sender,
|
|
56
|
-
blocks: [{
|
|
57
|
-
|
|
58
|
-
} as any),
|
|
54
|
+
blocks: [{ _tag: 'transcript', started: new Date(Date.now() + 1_000 * index).toISOString(), text: string }],
|
|
55
|
+
}),
|
|
59
56
|
);
|
|
60
57
|
|
|
61
|
-
const REMOTE_AI = true;
|
|
58
|
+
// const REMOTE_AI = true;
|
|
62
59
|
|
|
63
60
|
describe.skip('SentenceNormalization', () => {
|
|
64
61
|
const getExecutor = () => {
|
|
65
62
|
return new FunctionExecutor(
|
|
66
63
|
new ServiceContainer().setServices({
|
|
67
|
-
ai: {
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
},
|
|
64
|
+
// ai: {
|
|
65
|
+
// client: REMOTE_AI
|
|
66
|
+
// ? new Edge AiServiceClient({
|
|
67
|
+
// endpoint: AI_SERVICE_ENDPOINT.REMOTE,
|
|
68
|
+
// defaultGenerationOptions: {
|
|
69
|
+
// model: '@anthropic/claude-3-5-sonnet-20241022',
|
|
70
|
+
// },
|
|
71
|
+
// })
|
|
72
|
+
// : new Ollama AiServiceClient({
|
|
73
|
+
// overrides: {
|
|
74
|
+
// model: 'llama3.1:8b',
|
|
75
|
+
// },
|
|
76
|
+
// }),
|
|
77
|
+
// },
|
|
81
78
|
}),
|
|
82
79
|
);
|
|
83
80
|
};
|
|
@@ -100,6 +97,7 @@ describe.skip('SentenceNormalization', () => {
|
|
|
100
97
|
buffer = sentences.slice(activeSentenceIndex);
|
|
101
98
|
}
|
|
102
99
|
}
|
|
100
|
+
|
|
103
101
|
log.info('sentences', {
|
|
104
102
|
originalMessages: JSON.stringify(messages, null, 2),
|
|
105
103
|
sentences: JSON.stringify(sentences, null, 2),
|
|
@@ -2,15 +2,19 @@
|
|
|
2
2
|
// Copyright 2025 DXOS.org
|
|
3
3
|
//
|
|
4
4
|
|
|
5
|
+
// ISSUE(burdon): defineFunction
|
|
6
|
+
// @ts-nocheck
|
|
7
|
+
|
|
5
8
|
import { Schema } from 'effect';
|
|
6
9
|
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
10
|
+
import { AiService, DEFAULT_EDGE_MODEL } from '@dxos/ai';
|
|
11
|
+
import { AiSession } from '@dxos/assistant';
|
|
9
12
|
import { Obj } from '@dxos/echo';
|
|
10
|
-
import {
|
|
13
|
+
import { defineFunction } from '@dxos/functions';
|
|
11
14
|
import { ObjectId } from '@dxos/keys';
|
|
12
15
|
import { log } from '@dxos/log';
|
|
13
16
|
import { DataType } from '@dxos/schema';
|
|
17
|
+
import { trim } from '@dxos/util';
|
|
14
18
|
|
|
15
19
|
const MessageWithRangeId = Schema.extend(
|
|
16
20
|
DataType.Message,
|
|
@@ -39,40 +43,6 @@ export const NormalizationOutput = Schema.Struct({
|
|
|
39
43
|
});
|
|
40
44
|
export interface NormalizationOutput extends Schema.Schema.Type<typeof NormalizationOutput> {}
|
|
41
45
|
|
|
42
|
-
const prompt = `
|
|
43
|
-
You are observing a real-time transcript of a single person speaking.
|
|
44
|
-
The transcription is delivered in chunks of 10 seconds or less. As a result, individual sentences may be split across multiple messages, or multiple sentences may appear within a single message. Additionally, because this is real-time transcription, punctuation and capitalization may be incorrect or missing.
|
|
45
|
-
|
|
46
|
-
# Task Description:
|
|
47
|
-
- Your task is to detect and reconstruct broken or incomplete sentences by merging segments into coherent, grammatically correct sentences where appropriate.
|
|
48
|
-
|
|
49
|
-
# Input Format:
|
|
50
|
-
- The input is an array of messages that contain transcription blocks.
|
|
51
|
-
- Each message is a DataType.Message, as described in the output format.
|
|
52
|
-
|
|
53
|
-
# Output Format:
|
|
54
|
-
- You have been provided with the tool that defines the output format; make sure to query it.
|
|
55
|
-
- Do not output anything other than the expected format.
|
|
56
|
-
|
|
57
|
-
# Segment Handling Rules:
|
|
58
|
-
- Sort messages by timestamp.
|
|
59
|
-
- Leave ID of the first message in a sentence.
|
|
60
|
-
- Each divided sentence should be added to the output as a separate message or block within existing messages, preserving the original order.
|
|
61
|
-
- If one message contains multiple sentences, you should not split them.
|
|
62
|
-
- The last sentence may be incomplete; it should still be output as a separate block or message.
|
|
63
|
-
- Keep track of the original timestamps of the messages and blocks and, use the timestamp of the first message/block of a sentence as the 'started' field of the merged message.
|
|
64
|
-
- The 'rangeId' field of the merged message should include 'messageId'-s and 'rangeId'-s of all messages that have been used to construct this message.
|
|
65
|
-
|
|
66
|
-
# Punctuation and Capitalization:
|
|
67
|
-
- Do not rely on the original punctuation and capitalization—they may be incorrect.
|
|
68
|
-
- Use logical reasoning to apply appropriate punctuation (e.g., period, comma, question mark, exclamation mark) and capitalization.
|
|
69
|
-
|
|
70
|
-
# Restrictions:
|
|
71
|
-
- Do not alter the order of the original messages; maintain the natural flow of speech.
|
|
72
|
-
- Do not interpret or infer meaning beyond reconstructing sentences.
|
|
73
|
-
- Do not add or remove any words or phrases.
|
|
74
|
-
`;
|
|
75
|
-
|
|
76
46
|
export const sentenceNormalization = defineFunction<NormalizationInput, NormalizationOutput>({
|
|
77
47
|
description: 'Post process of transcription for sentence normalization',
|
|
78
48
|
inputSchema: NormalizationInput,
|
|
@@ -80,21 +50,27 @@ export const sentenceNormalization = defineFunction<NormalizationInput, Normaliz
|
|
|
80
50
|
handler: async ({ data: { messages }, context }) => {
|
|
81
51
|
log.info('input', { messages });
|
|
82
52
|
const ai = context.getService(AiService);
|
|
83
|
-
const session = new
|
|
53
|
+
const session = new AiSession({ operationModel: 'configured' });
|
|
84
54
|
|
|
85
|
-
|
|
86
|
-
|
|
55
|
+
// TODO(dmaretskyi): This got broken after effect-ai transition.
|
|
56
|
+
const response = session.runStructured(NormalizationOutput, {
|
|
57
|
+
generationOptions: {
|
|
58
|
+
model: DEFAULT_EDGE_MODEL,
|
|
59
|
+
},
|
|
87
60
|
client: ai.client,
|
|
88
61
|
tools: [],
|
|
89
62
|
artifacts: [],
|
|
90
63
|
history: [
|
|
91
|
-
Obj.make(Message, {
|
|
92
|
-
|
|
93
|
-
|
|
64
|
+
Obj.make(DataType.Message, {
|
|
65
|
+
created: new Date().toISOString(),
|
|
66
|
+
sender: {
|
|
67
|
+
role: 'user',
|
|
68
|
+
},
|
|
69
|
+
blocks: messages.map((message) => ({ _tag: 'text', text: JSON.stringify(message) }) as const),
|
|
94
70
|
}),
|
|
95
71
|
],
|
|
96
72
|
prompt,
|
|
97
|
-
});
|
|
73
|
+
}) as any;
|
|
98
74
|
|
|
99
75
|
response.sentences.forEach((sentence) => {
|
|
100
76
|
sentence.id = ObjectId.random();
|
|
@@ -103,3 +79,37 @@ export const sentenceNormalization = defineFunction<NormalizationInput, Normaliz
|
|
|
103
79
|
return response;
|
|
104
80
|
},
|
|
105
81
|
});
|
|
82
|
+
|
|
83
|
+
const prompt = trim`
|
|
84
|
+
You are observing a real-time transcript of a single person speaking.
|
|
85
|
+
The transcription is delivered in chunks of 10 seconds or less. As a result, individual sentences may be split across multiple messages, or multiple sentences may appear within a single message. Additionally, because this is real-time transcription, punctuation and capitalization may be incorrect or missing.
|
|
86
|
+
|
|
87
|
+
# Task Description:
|
|
88
|
+
- Your task is to detect and reconstruct broken or incomplete sentences by merging segments into coherent, grammatically correct sentences where appropriate.
|
|
89
|
+
|
|
90
|
+
# Input Format:
|
|
91
|
+
- The input is an array of messages that contain transcription blocks.
|
|
92
|
+
- Each message is a DataType.Message, as described in the output format.
|
|
93
|
+
|
|
94
|
+
# Output Format:
|
|
95
|
+
- You have been provided with the tool that defines the output format; make sure to query it.
|
|
96
|
+
- Do not output anything other than the expected format.
|
|
97
|
+
|
|
98
|
+
# Segment Handling Rules:
|
|
99
|
+
- Sort messages by timestamp.
|
|
100
|
+
- Leave ID of the first message in a sentence.
|
|
101
|
+
- Each divided sentence should be added to the output as a separate message or block within existing messages, preserving the original order.
|
|
102
|
+
- If one message contains multiple sentences, you should not split them.
|
|
103
|
+
- The last sentence may be incomplete; it should still be output as a separate block or message.
|
|
104
|
+
- Keep track of the original timestamps of the messages and blocks and, use the timestamp of the first message/block of a sentence as the 'started' field of the merged message.
|
|
105
|
+
- The 'rangeId' field of the merged message should include 'messageId'-s and 'rangeId'-s of all messages that have been used to construct this message.
|
|
106
|
+
|
|
107
|
+
# Punctuation and Capitalization:
|
|
108
|
+
- Do not rely on the original punctuation and capitalization—they may be incorrect.
|
|
109
|
+
- Use logical reasoning to apply appropriate punctuation (e.g., period, comma, question mark, exclamation mark) and capitalization.
|
|
110
|
+
|
|
111
|
+
# Restrictions:
|
|
112
|
+
- Do not alter the order of the original messages; maintain the natural flow of speech.
|
|
113
|
+
- Do not interpret or infer meaning beyond reconstructing sentences.
|
|
114
|
+
- Do not add or remove any words or phrases.
|
|
115
|
+
`;
|
package/src/testing/testing.ts
CHANGED
|
@@ -5,9 +5,7 @@
|
|
|
5
5
|
import { Schema } from 'effect';
|
|
6
6
|
import { useEffect, useMemo, useState } from 'react';
|
|
7
7
|
|
|
8
|
-
import {
|
|
9
|
-
import { AI_SERVICE_ENDPOINT } from '@dxos/ai/testing';
|
|
10
|
-
import { extractionAnthropicFn, processTranscriptMessage } from '@dxos/assistant';
|
|
8
|
+
import { extractionAnthropicFn, processTranscriptMessage } from '@dxos/assistant/extraction';
|
|
11
9
|
import { scheduleTaskInterval } from '@dxos/async';
|
|
12
10
|
import { Filter, type Queue } from '@dxos/client/echo';
|
|
13
11
|
import { Context } from '@dxos/context';
|
|
@@ -17,7 +15,7 @@ import { FunctionExecutor, ServiceContainer } from '@dxos/functions';
|
|
|
17
15
|
import { IdentityDid } from '@dxos/keys';
|
|
18
16
|
import { log } from '@dxos/log';
|
|
19
17
|
import { faker } from '@dxos/random';
|
|
20
|
-
import {
|
|
18
|
+
import { type Space, useQueue } from '@dxos/react-client/echo';
|
|
21
19
|
import { DataType } from '@dxos/schema';
|
|
22
20
|
import { Testing, seedTestData } from '@dxos/schema/testing';
|
|
23
21
|
|
|
@@ -69,7 +67,7 @@ export class MessageBuilder extends AbstractMessageBuilder {
|
|
|
69
67
|
});
|
|
70
68
|
}
|
|
71
69
|
|
|
72
|
-
createBlock(): DataType.MessageBlock.
|
|
70
|
+
createBlock(): DataType.MessageBlock.Transcript {
|
|
73
71
|
let text = faker.lorem.paragraph();
|
|
74
72
|
if (this._space) {
|
|
75
73
|
const label = faker.commerce.productName();
|
|
@@ -81,7 +79,7 @@ export class MessageBuilder extends AbstractMessageBuilder {
|
|
|
81
79
|
}
|
|
82
80
|
|
|
83
81
|
return {
|
|
84
|
-
|
|
82
|
+
_tag: 'transcript',
|
|
85
83
|
started: this.next().toISOString(),
|
|
86
84
|
text,
|
|
87
85
|
};
|
|
@@ -95,11 +93,13 @@ export class MessageBuilder extends AbstractMessageBuilder {
|
|
|
95
93
|
|
|
96
94
|
// TODO(burdon): Reconcile with BlockBuilder.
|
|
97
95
|
class EntityExtractionMessageBuilder extends AbstractMessageBuilder {
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
96
|
+
executor = new FunctionExecutor(
|
|
97
|
+
new ServiceContainer().setServices({
|
|
98
|
+
// ai: {
|
|
99
|
+
// client: this.AiService,
|
|
100
|
+
// },
|
|
101
|
+
}),
|
|
102
|
+
);
|
|
103
103
|
|
|
104
104
|
space: Space | undefined;
|
|
105
105
|
currentMessage: number = 0;
|
package/src/transcriber/index.ts
CHANGED
|
@@ -10,7 +10,7 @@ import { invariant } from '@dxos/invariant';
|
|
|
10
10
|
import { log } from '@dxos/log';
|
|
11
11
|
import { trace } from '@dxos/tracing';
|
|
12
12
|
|
|
13
|
-
import { type
|
|
13
|
+
import { type AudioChunk, type AudioRecorder, type WavConfig } from './audio-recorder';
|
|
14
14
|
|
|
15
15
|
let initializingPromise: Promise<void> | undefined;
|
|
16
16
|
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
import fs from 'node:fs';
|
|
6
6
|
import path from 'node:path';
|
|
7
|
+
|
|
7
8
|
import { describe, expect, test } from 'vitest';
|
|
8
9
|
import { WaveFile } from 'wavefile';
|
|
9
10
|
|
|
@@ -11,7 +12,7 @@ import { Trigger } from '@dxos/async';
|
|
|
11
12
|
import { log } from '@dxos/log';
|
|
12
13
|
import { type DataType } from '@dxos/schema';
|
|
13
14
|
import { openAndClose } from '@dxos/test-utils';
|
|
14
|
-
import {
|
|
15
|
+
import { TRACE_PROCESSOR, trace } from '@dxos/tracing';
|
|
15
16
|
|
|
16
17
|
import { type AudioChunk, type AudioRecorder, Transcriber } from '../transcriber';
|
|
17
18
|
import { mergeFloat64Arrays } from '../util';
|
|
@@ -93,7 +94,7 @@ describe.skip('Transcriber', () => {
|
|
|
93
94
|
});
|
|
94
95
|
|
|
95
96
|
test.skip('transcription of audio recording', { timeout: 10_000 }, async () => {
|
|
96
|
-
const trigger = new Trigger<DataType.MessageBlock.
|
|
97
|
+
const trigger = new Trigger<DataType.MessageBlock.Transcript[]>({ autoReset: true });
|
|
97
98
|
const recorder = new MockAudioRecorder({
|
|
98
99
|
buffer: await readFile('test.wav'),
|
|
99
100
|
chunkDuration: 3_000,
|
|
@@ -127,7 +128,7 @@ describe.skip('Transcriber', () => {
|
|
|
127
128
|
});
|
|
128
129
|
|
|
129
130
|
test.skip('transcription of audio recording with overlapping chunks', { timeout: 20_000 }, async () => {
|
|
130
|
-
const trigger = new Trigger<DataType.MessageBlock.
|
|
131
|
+
const trigger = new Trigger<DataType.MessageBlock.Transcript[]>({ autoReset: true });
|
|
131
132
|
const recorder = new MockAudioRecorder({
|
|
132
133
|
buffer: await readFile('test.wav'),
|
|
133
134
|
chunkDuration: 3_000,
|
|
@@ -10,10 +10,11 @@ import { log } from '@dxos/log';
|
|
|
10
10
|
import { type DataType } from '@dxos/schema';
|
|
11
11
|
import { trace } from '@dxos/tracing';
|
|
12
12
|
|
|
13
|
-
import { type AudioRecorder, type AudioChunk } from './audio-recorder';
|
|
14
13
|
import { TRANSCRIPTION_URL } from '../types';
|
|
15
14
|
import { mergeFloat64Arrays } from '../util';
|
|
16
15
|
|
|
16
|
+
import { type AudioChunk, type AudioRecorder } from './audio-recorder';
|
|
17
|
+
|
|
17
18
|
type WhisperWord = {
|
|
18
19
|
word: string;
|
|
19
20
|
|
|
@@ -69,7 +70,7 @@ export type TranscriberParams = {
|
|
|
69
70
|
* Callback to handle the transcribed segments, after all segment transformers are applied.
|
|
70
71
|
* @param segments - The transcribed segments.
|
|
71
72
|
*/
|
|
72
|
-
onSegments: (segments: DataType.MessageBlock.
|
|
73
|
+
onSegments: (segments: DataType.MessageBlock.Transcript[]) => Promise<void>;
|
|
73
74
|
};
|
|
74
75
|
|
|
75
76
|
/**
|
|
@@ -220,10 +221,7 @@ export class Transcriber extends Resource {
|
|
|
220
221
|
return segments;
|
|
221
222
|
}
|
|
222
223
|
|
|
223
|
-
private _alignSegments(
|
|
224
|
-
segments: WhisperSegment[],
|
|
225
|
-
originalChunks: AudioChunk[],
|
|
226
|
-
): DataType.MessageBlock.Transcription[] {
|
|
224
|
+
private _alignSegments(segments: WhisperSegment[], originalChunks: AudioChunk[]): DataType.MessageBlock.Transcript[] {
|
|
227
225
|
// Absolute zero for all relative timestamps in the segments.
|
|
228
226
|
const zeroTimestamp = originalChunks.at(0)!.timestamp;
|
|
229
227
|
|
|
@@ -252,7 +250,7 @@ export class Transcriber extends Resource {
|
|
|
252
250
|
|
|
253
251
|
// Add absolute timestamp to each segment.
|
|
254
252
|
return filteredSegments.map((segment) => ({
|
|
255
|
-
|
|
253
|
+
_tag: 'transcript',
|
|
256
254
|
started: new Date(zeroTimestamp + segment.start * 1_000).toISOString(),
|
|
257
255
|
text: segment.text.trim(),
|
|
258
256
|
}));
|
|
@@ -133,7 +133,7 @@ export class TranscriptionManager extends Resource {
|
|
|
133
133
|
// TODO(burdon): Started and stopped blocks appear twice.
|
|
134
134
|
const block = Obj.make(DataType.Message, {
|
|
135
135
|
created: new Date().toISOString(),
|
|
136
|
-
blocks: [{
|
|
136
|
+
blocks: [{ _tag: 'transcript', text: 'Started', started: new Date().toISOString() }],
|
|
137
137
|
sender: { role: 'assistant' },
|
|
138
138
|
});
|
|
139
139
|
await this._queue?.append([block]);
|
|
@@ -141,7 +141,7 @@ export class TranscriptionManager extends Resource {
|
|
|
141
141
|
await this._transcriber?.close();
|
|
142
142
|
const block = Obj.make(DataType.Message, {
|
|
143
143
|
created: new Date().toISOString(),
|
|
144
|
-
blocks: [{
|
|
144
|
+
blocks: [{ _tag: 'transcript', text: 'Stopped', started: new Date().toISOString() }],
|
|
145
145
|
sender: { role: 'assistant' },
|
|
146
146
|
});
|
|
147
147
|
await this._queue?.append([block]);
|
|
@@ -175,7 +175,7 @@ export class TranscriptionManager extends Resource {
|
|
|
175
175
|
}
|
|
176
176
|
}
|
|
177
177
|
|
|
178
|
-
private async _onSegments(segments: DataType.MessageBlock.
|
|
178
|
+
private async _onSegments(segments: DataType.MessageBlock.Transcript[]): Promise<void> {
|
|
179
179
|
if (!this.isOpen || !this._queue) {
|
|
180
180
|
return;
|
|
181
181
|
}
|
package/src/translations.ts
CHANGED
|
@@ -2,12 +2,14 @@
|
|
|
2
2
|
// Copyright 2023 DXOS.org
|
|
3
3
|
//
|
|
4
4
|
|
|
5
|
-
import {
|
|
5
|
+
import { type Resource } from '@dxos/react-ui';
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
import { meta } from './meta';
|
|
8
|
+
|
|
9
|
+
export const translations = [
|
|
8
10
|
{
|
|
9
11
|
'en-US': {
|
|
10
|
-
[
|
|
12
|
+
[meta.id]: {
|
|
11
13
|
'plugin name': 'Transcription',
|
|
12
14
|
'transcript companion label': 'Transcript',
|
|
13
15
|
|
|
@@ -19,4 +21,4 @@ export default [
|
|
|
19
21
|
},
|
|
20
22
|
},
|
|
21
23
|
},
|
|
22
|
-
];
|
|
24
|
+
] as const satisfies Resource[];
|
package/src/types/types.ts
CHANGED
|
@@ -6,9 +6,10 @@ import { Schema } from 'effect';
|
|
|
6
6
|
|
|
7
7
|
import { SpaceId } from '@dxos/react-client/echo';
|
|
8
8
|
|
|
9
|
-
import { TranscriptType } from './schema';
|
|
10
9
|
import { TRANSCRIPTION_PLUGIN } from '../meta';
|
|
11
10
|
|
|
11
|
+
import { TranscriptType } from './schema';
|
|
12
|
+
|
|
12
13
|
// TODO(burdon): Move to separate proto.
|
|
13
14
|
|
|
14
15
|
/**
|
|
@@ -1,11 +0,0 @@
|
|
|
1
|
-
import {
|
|
2
|
-
TranscriptContainer_default,
|
|
3
|
-
TranscriptionContainer
|
|
4
|
-
} from "./chunk-53AXUEIQ.mjs";
|
|
5
|
-
import "./chunk-W2KNYJD6.mjs";
|
|
6
|
-
import "./chunk-GOF7MJNV.mjs";
|
|
7
|
-
export {
|
|
8
|
-
TranscriptionContainer,
|
|
9
|
-
TranscriptContainer_default as default
|
|
10
|
-
};
|
|
11
|
-
//# sourceMappingURL=TranscriptContainer-H3AZQ4CQ.mjs.map
|