@dxos/plugin-transcription 0.8.3 → 0.8.4-main.1da679c

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (198) hide show
  1. package/dist/lib/browser/TranscriptContainer-W6CHJPIB.mjs +11 -0
  2. package/dist/lib/browser/blueprint-definition-C62MH2FL.mjs +142 -0
  3. package/dist/lib/browser/blueprint-definition-C62MH2FL.mjs.map +7 -0
  4. package/dist/lib/browser/{chunk-GOF7MJNV.mjs → chunk-3X7HP44M.mjs} +9 -2
  5. package/dist/lib/browser/{chunk-GOF7MJNV.mjs.map → chunk-3X7HP44M.mjs.map} +1 -1
  6. package/dist/lib/browser/{chunk-IQ5ZRMHZ.mjs → chunk-BA64VLLR.mjs} +93 -102
  7. package/dist/lib/browser/chunk-BA64VLLR.mjs.map +7 -0
  8. package/dist/lib/browser/{chunk-53AXUEIQ.mjs → chunk-DHMLTHPB.mjs} +424 -442
  9. package/dist/lib/browser/chunk-DHMLTHPB.mjs.map +7 -0
  10. package/dist/lib/browser/chunk-HXVGV7T6.mjs +63 -0
  11. package/dist/lib/browser/chunk-HXVGV7T6.mjs.map +7 -0
  12. package/dist/lib/browser/{chunk-W2KNYJD6.mjs → chunk-TNEQFHIU.mjs} +3 -3
  13. package/dist/lib/browser/{chunk-W2KNYJD6.mjs.map → chunk-TNEQFHIU.mjs.map} +1 -1
  14. package/dist/lib/browser/{chunk-UDCCT32F.mjs → chunk-ZNCGBPXR.mjs} +3 -3
  15. package/dist/lib/browser/index.mjs +27 -21
  16. package/dist/lib/browser/index.mjs.map +3 -3
  17. package/dist/lib/browser/intent-resolver-LKRGWUX2.mjs +25 -0
  18. package/dist/lib/browser/intent-resolver-LKRGWUX2.mjs.map +7 -0
  19. package/dist/lib/browser/meta.json +1 -1
  20. package/dist/lib/browser/{react-surface-HMJS5FSS.mjs → react-surface-2AGVHCYU.mjs} +11 -11
  21. package/dist/lib/browser/react-surface-2AGVHCYU.mjs.map +7 -0
  22. package/dist/lib/browser/{transcriber-OYPLZFG5.mjs → transcriber-L3NQP2OM.mjs} +8 -8
  23. package/dist/lib/browser/transcriber-L3NQP2OM.mjs.map +7 -0
  24. package/dist/lib/browser/types/index.mjs +6 -8
  25. package/dist/lib/node-esm/{TranscriptContainer-G75LDEO5.mjs → TranscriptContainer-LJLH47VF.mjs} +4 -4
  26. package/dist/lib/node-esm/blueprint-definition-PH3DDDHR.mjs +143 -0
  27. package/dist/lib/node-esm/blueprint-definition-PH3DDDHR.mjs.map +7 -0
  28. package/dist/lib/node-esm/{chunk-MFJPEIVB.mjs → chunk-3QAIZGMW.mjs} +3 -3
  29. package/dist/lib/node-esm/{chunk-MFJPEIVB.mjs.map → chunk-3QAIZGMW.mjs.map} +1 -1
  30. package/dist/lib/node-esm/{chunk-PRL64SRU.mjs → chunk-J5GQ3KGX.mjs} +93 -102
  31. package/dist/lib/node-esm/chunk-J5GQ3KGX.mjs.map +7 -0
  32. package/dist/lib/node-esm/{chunk-56QHOX4A.mjs → chunk-JQARMZNW.mjs} +424 -442
  33. package/dist/lib/node-esm/chunk-JQARMZNW.mjs.map +7 -0
  34. package/dist/lib/node-esm/{chunk-MFJO7EKD.mjs → chunk-JQBXD6KX.mjs} +3 -3
  35. package/dist/lib/node-esm/{chunk-TXMDLLBR.mjs → chunk-QARUDLHN.mjs} +8 -2
  36. package/dist/lib/node-esm/{chunk-TXMDLLBR.mjs.map → chunk-QARUDLHN.mjs.map} +1 -1
  37. package/dist/lib/node-esm/chunk-W72HVQMD.mjs +64 -0
  38. package/dist/lib/node-esm/chunk-W72HVQMD.mjs.map +7 -0
  39. package/dist/lib/node-esm/index.mjs +27 -21
  40. package/dist/lib/node-esm/index.mjs.map +3 -3
  41. package/dist/lib/node-esm/intent-resolver-AGLWBOCO.mjs +26 -0
  42. package/dist/lib/node-esm/intent-resolver-AGLWBOCO.mjs.map +7 -0
  43. package/dist/lib/node-esm/meta.json +1 -1
  44. package/dist/lib/node-esm/{react-surface-JGQSYGRR.mjs → react-surface-HPQQ3WIF.mjs} +11 -11
  45. package/dist/lib/node-esm/react-surface-HPQQ3WIF.mjs.map +7 -0
  46. package/dist/lib/node-esm/{transcriber-TVG2MBBN.mjs → transcriber-KJX4U74D.mjs} +8 -8
  47. package/dist/lib/node-esm/transcriber-KJX4U74D.mjs.map +7 -0
  48. package/dist/lib/node-esm/types/index.mjs +6 -8
  49. package/dist/types/src/TranscriptionPlugin.d.ts.map +1 -1
  50. package/dist/types/src/capabilities/blueprint-definition.d.ts +5 -0
  51. package/dist/types/src/capabilities/blueprint-definition.d.ts.map +1 -0
  52. package/dist/types/src/capabilities/capabilities.d.ts +3 -3
  53. package/dist/types/src/capabilities/capabilities.d.ts.map +1 -1
  54. package/dist/types/src/capabilities/index.d.ts +3 -2
  55. package/dist/types/src/capabilities/index.d.ts.map +1 -1
  56. package/dist/types/src/capabilities/intent-resolver.d.ts +1 -1
  57. package/dist/types/src/capabilities/intent-resolver.d.ts.map +1 -1
  58. package/dist/types/src/capabilities/react-surface.d.ts +1 -1
  59. package/dist/types/src/capabilities/transcriber.d.ts +1 -1
  60. package/dist/types/src/capabilities/transcriber.d.ts.map +1 -1
  61. package/dist/types/src/components/Transcript/Transcript.d.ts +5 -5
  62. package/dist/types/src/components/Transcript/Transcript.d.ts.map +1 -1
  63. package/dist/types/src/components/Transcript/Transcript.stories.d.ts +30 -15
  64. package/dist/types/src/components/Transcript/Transcript.stories.d.ts.map +1 -1
  65. package/dist/types/src/components/Transcript/deprecated/OldTranscript.d.ts +2 -2
  66. package/dist/types/src/components/Transcript/deprecated/OldTranscript.d.ts.map +1 -1
  67. package/dist/types/src/components/Transcript/transcript-extension.d.ts +1 -5
  68. package/dist/types/src/components/Transcript/transcript-extension.d.ts.map +1 -1
  69. package/dist/types/src/components/TranscriptContainer.d.ts +6 -5
  70. package/dist/types/src/components/TranscriptContainer.d.ts.map +1 -1
  71. package/dist/types/src/components/index.d.ts +1 -4
  72. package/dist/types/src/components/index.d.ts.map +1 -1
  73. package/dist/types/src/components/stories/FileTranscription.stories.d.ts +5 -3
  74. package/dist/types/src/components/stories/FileTranscription.stories.d.ts.map +1 -1
  75. package/dist/types/src/components/stories/KeyWordRecognition.stories.d.ts +10 -5
  76. package/dist/types/src/components/stories/KeyWordRecognition.stories.d.ts.map +1 -1
  77. package/dist/types/src/components/stories/MicrophoneTranscription.stories.d.ts +7 -5
  78. package/dist/types/src/components/stories/MicrophoneTranscription.stories.d.ts.map +1 -1
  79. package/dist/types/src/components/stories/TranscriptionStory.d.ts +1 -0
  80. package/dist/types/src/components/stories/TranscriptionStory.d.ts.map +1 -1
  81. package/dist/types/src/functions/index.d.ts +3 -0
  82. package/dist/types/src/functions/index.d.ts.map +1 -0
  83. package/dist/types/src/functions/open.d.ts +7 -0
  84. package/dist/types/src/functions/open.d.ts.map +1 -0
  85. package/dist/types/src/functions/summarize.d.ts +11 -0
  86. package/dist/types/src/functions/summarize.d.ts.map +1 -0
  87. package/dist/types/src/hooks/useAudioFile.d.ts.map +1 -1
  88. package/dist/types/src/hooks/useTranscriber.d.ts +1 -1
  89. package/dist/types/src/hooks/useVoiceInput.d.ts.map +1 -1
  90. package/dist/types/src/segments-normalization/normalization.d.ts +375 -243
  91. package/dist/types/src/segments-normalization/normalization.d.ts.map +1 -1
  92. package/dist/types/src/testing/testing.d.ts +1 -1
  93. package/dist/types/src/testing/testing.d.ts.map +1 -1
  94. package/dist/types/src/transcriber/index.d.ts +1 -1
  95. package/dist/types/src/transcriber/index.d.ts.map +1 -1
  96. package/dist/types/src/transcriber/media-stream-recorder.d.ts +1 -1
  97. package/dist/types/src/transcriber/media-stream-recorder.d.ts.map +1 -1
  98. package/dist/types/src/transcriber/transcriber.d.ts +1 -1
  99. package/dist/types/src/transcriber/transcriber.d.ts.map +1 -1
  100. package/dist/types/src/transcriber/transcription-manager.d.ts +3 -7
  101. package/dist/types/src/transcriber/transcription-manager.d.ts.map +1 -1
  102. package/dist/types/src/translations.d.ts +11 -11
  103. package/dist/types/src/translations.d.ts.map +1 -1
  104. package/dist/types/src/types/Transcript.d.ts +30 -0
  105. package/dist/types/src/types/Transcript.d.ts.map +1 -0
  106. package/dist/types/src/types/TranscriptAction.d.ts +23 -0
  107. package/dist/types/src/types/TranscriptAction.d.ts.map +1 -0
  108. package/dist/types/src/types/index.d.ts +2 -1
  109. package/dist/types/src/types/index.d.ts.map +1 -1
  110. package/dist/types/src/types/types.d.ts +0 -26
  111. package/dist/types/src/types/types.d.ts.map +1 -1
  112. package/dist/types/tsconfig.tsbuildinfo +1 -1
  113. package/package.json +61 -59
  114. package/src/TranscriptionPlugin.tsx +13 -8
  115. package/src/capabilities/blueprint-definition.ts +31 -0
  116. package/src/capabilities/capabilities.ts +5 -5
  117. package/src/capabilities/index.ts +1 -0
  118. package/src/capabilities/intent-resolver.ts +5 -11
  119. package/src/capabilities/react-surface.tsx +2 -2
  120. package/src/capabilities/transcriber.ts +3 -2
  121. package/src/components/Transcript/Transcript.stories.tsx +26 -24
  122. package/src/components/Transcript/Transcript.tsx +20 -27
  123. package/src/components/Transcript/deprecated/OldTranscript.tsx +4 -4
  124. package/src/components/Transcript/transcript-extension.ts +4 -41
  125. package/src/components/TranscriptContainer.tsx +13 -7
  126. package/src/components/stories/FileTranscription.stories.tsx +53 -39
  127. package/src/components/stories/KeyWordRecognition.stories.tsx +12 -6
  128. package/src/components/stories/MicrophoneTranscription.stories.tsx +41 -36
  129. package/src/components/stories/TranscriptionStory.tsx +9 -5
  130. package/src/functions/index.ts +6 -0
  131. package/src/functions/open.ts +37 -0
  132. package/src/functions/summarize.ts +97 -0
  133. package/src/hooks/useAudioFile.ts +1 -10
  134. package/src/hooks/useTranscriber.ts +5 -5
  135. package/src/hooks/useVoiceInput.ts +3 -2
  136. package/src/model/model.test.ts +9 -9
  137. package/src/segments-normalization/message-normalizer.ts +1 -1
  138. package/src/segments-normalization/normalization.test.ts +19 -21
  139. package/src/segments-normalization/normalization.ts +54 -44
  140. package/src/testing/testing.ts +13 -13
  141. package/src/transcriber/index.ts +1 -1
  142. package/src/transcriber/media-stream-recorder.ts +1 -1
  143. package/src/transcriber/transcriber.browser.test.ts +4 -3
  144. package/src/transcriber/transcriber.ts +5 -7
  145. package/src/transcriber/transcription-manager.ts +18 -39
  146. package/src/translations.ts +6 -4
  147. package/src/types/{schema.ts → Transcript.ts} +11 -8
  148. package/src/types/TranscriptAction.ts +24 -0
  149. package/src/types/index.ts +2 -1
  150. package/src/types/types.ts +0 -23
  151. package/dist/lib/browser/TranscriptContainer-H3AZQ4CQ.mjs +0 -11
  152. package/dist/lib/browser/chunk-53AXUEIQ.mjs.map +0 -7
  153. package/dist/lib/browser/chunk-IQ5ZRMHZ.mjs.map +0 -7
  154. package/dist/lib/browser/chunk-OSISJM2F.mjs +0 -55
  155. package/dist/lib/browser/chunk-OSISJM2F.mjs.map +0 -7
  156. package/dist/lib/browser/intent-resolver-BGADCHVM.mjs +0 -30
  157. package/dist/lib/browser/intent-resolver-BGADCHVM.mjs.map +0 -7
  158. package/dist/lib/browser/react-surface-HMJS5FSS.mjs.map +0 -7
  159. package/dist/lib/browser/transcriber-OYPLZFG5.mjs.map +0 -7
  160. package/dist/lib/node/TranscriptContainer-KIG3GMAB.cjs +0 -32
  161. package/dist/lib/node/TranscriptContainer-KIG3GMAB.cjs.map +0 -7
  162. package/dist/lib/node/chunk-3PYT3BDM.cjs +0 -40
  163. package/dist/lib/node/chunk-3PYT3BDM.cjs.map +0 -7
  164. package/dist/lib/node/chunk-6LW5RA3K.cjs +0 -75
  165. package/dist/lib/node/chunk-6LW5RA3K.cjs.map +0 -7
  166. package/dist/lib/node/chunk-AFUZI63A.cjs +0 -42
  167. package/dist/lib/node/chunk-AFUZI63A.cjs.map +0 -7
  168. package/dist/lib/node/chunk-CVUN3UO7.cjs +0 -778
  169. package/dist/lib/node/chunk-CVUN3UO7.cjs.map +0 -7
  170. package/dist/lib/node/chunk-FKDJNJ57.cjs +0 -35
  171. package/dist/lib/node/chunk-FKDJNJ57.cjs.map +0 -7
  172. package/dist/lib/node/chunk-ZFAUMZXE.cjs +0 -518
  173. package/dist/lib/node/chunk-ZFAUMZXE.cjs.map +0 -7
  174. package/dist/lib/node/index.cjs +0 -133
  175. package/dist/lib/node/index.cjs.map +0 -7
  176. package/dist/lib/node/intent-resolver-LL3TTI4H.cjs +0 -45
  177. package/dist/lib/node/intent-resolver-LL3TTI4H.cjs.map +0 -7
  178. package/dist/lib/node/meta.json +0 -1
  179. package/dist/lib/node/react-surface-N25TBBG4.cjs +0 -56
  180. package/dist/lib/node/react-surface-N25TBBG4.cjs.map +0 -7
  181. package/dist/lib/node/transcriber-6J4FVFSZ.cjs +0 -68
  182. package/dist/lib/node/transcriber-6J4FVFSZ.cjs.map +0 -7
  183. package/dist/lib/node/types/index.cjs +0 -36
  184. package/dist/lib/node/types/index.cjs.map +0 -7
  185. package/dist/lib/node-esm/chunk-56QHOX4A.mjs.map +0 -7
  186. package/dist/lib/node-esm/chunk-FH75X4JC.mjs +0 -56
  187. package/dist/lib/node-esm/chunk-FH75X4JC.mjs.map +0 -7
  188. package/dist/lib/node-esm/chunk-PRL64SRU.mjs.map +0 -7
  189. package/dist/lib/node-esm/intent-resolver-5LVCYSJS.mjs +0 -31
  190. package/dist/lib/node-esm/intent-resolver-5LVCYSJS.mjs.map +0 -7
  191. package/dist/lib/node-esm/react-surface-JGQSYGRR.mjs.map +0 -7
  192. package/dist/lib/node-esm/transcriber-TVG2MBBN.mjs.map +0 -7
  193. package/dist/types/src/types/schema.d.ts +0 -41
  194. package/dist/types/src/types/schema.d.ts.map +0 -1
  195. /package/dist/lib/browser/{TranscriptContainer-H3AZQ4CQ.mjs.map → TranscriptContainer-W6CHJPIB.mjs.map} +0 -0
  196. /package/dist/lib/browser/{chunk-UDCCT32F.mjs.map → chunk-ZNCGBPXR.mjs.map} +0 -0
  197. /package/dist/lib/node-esm/{TranscriptContainer-G75LDEO5.mjs.map → TranscriptContainer-LJLH47VF.mjs.map} +0 -0
  198. /package/dist/lib/node-esm/{chunk-MFJO7EKD.mjs.map → chunk-JQBXD6KX.mjs.map} +0 -0
@@ -0,0 +1,97 @@
1
+ //
2
+ // Copyright 2025 DXOS.org
3
+ //
4
+
5
+ import { Array, Effect, Layer, Option, Schema, pipe } from 'effect';
6
+
7
+ import { AiService, ConsolePrinter, ToolExecutionService, ToolResolverService } from '@dxos/ai';
8
+ import { AiSession, GenerationObserver } from '@dxos/assistant';
9
+ import { TracingService, defineFunction } from '@dxos/functions';
10
+ import { trim } from '@dxos/util';
11
+
12
+ /**
13
+ * Summarize a transcript of a meeting.
14
+ */
15
+ export default defineFunction({
16
+ name: 'dxos.org/function/transcription/summarize',
17
+ description: 'Summarize a transcript of a meeting.',
18
+ inputSchema: Schema.Struct({
19
+ transcript: Schema.String.annotations({
20
+ description: 'The transcript of the meeting.',
21
+ }),
22
+ notes: Schema.optional(Schema.String).annotations({
23
+ description: 'Additional notes from the participants.',
24
+ }),
25
+ }),
26
+ outputSchema: Schema.Struct({
27
+ summary: Schema.String.annotations({
28
+ description: 'The summary of the transcript.',
29
+ }),
30
+ }),
31
+ handler: Effect.fnUntraced(
32
+ function* ({ data: { transcript, notes } }) {
33
+ const result = yield* new AiSession().run({
34
+ prompt: `Transcript: ${transcript}\n\nNotes: ${notes}`,
35
+ history: [],
36
+ system: systemPrompt,
37
+ observer: GenerationObserver.fromPrinter(new ConsolePrinter({ tag: 'summarize' })),
38
+ });
39
+
40
+ const summary = pipe(
41
+ result,
42
+ Array.findLast((msg) => msg.sender.role === 'assistant' && msg.blocks.some((block) => block._tag === 'text')),
43
+ Option.flatMap((msg) =>
44
+ pipe(
45
+ msg.blocks,
46
+ Array.findLast((block) => block._tag === 'text'),
47
+ Option.map((block) => block.text),
48
+ ),
49
+ ),
50
+ Option.getOrThrowWith(() => new Error('No summary found')),
51
+ );
52
+
53
+ return { summary };
54
+ },
55
+ Effect.provide(
56
+ Layer.mergeAll(
57
+ AiService.model('@anthropic/claude-sonnet-4-0'),
58
+ ToolResolverService.layerEmpty,
59
+ ToolExecutionService.layerEmpty,
60
+ TracingService.layerNoop,
61
+ ),
62
+ ),
63
+ ),
64
+ });
65
+
66
+ const systemPrompt = trim`
67
+ You are a helpful assistant that summarizes transcripts of meetings.
68
+
69
+ # Goal
70
+ Create a markdown summary of the meeting transcript with text notes provided.
71
+ Notes are very important so make sure to include them in the summary if they contain meaningful information.
72
+
73
+ # Formatting
74
+ - Format the summary as a markdown document without extra comments like "Here is the summary of the transcript:".
75
+ - Use markdown formatting for headings and bullet points.
76
+ - Format the summary as a list of key points and takeaways.
77
+ - All names of people should be in bold.
78
+
79
+ # Note Taking
80
+ - Correlate items in the summary with the person of origin to build a coherent narrative.
81
+ - Include short quotes verbatim where appropriate. Especially when concerned with design decisions and problem descriptions.
82
+
83
+ # Tasks
84
+ At the end of the summary include tasks.
85
+ Extract only the tasks that are:
86
+ - Directly actionable.
87
+ - Clearly assigned to a person or team (or can easily be inferred).
88
+ - Strongly implied by the conversation and/or user note (no speculative tasks).
89
+ - Specific enough that someone reading them would know exactly what to do next.
90
+
91
+ Format all tasks as markdown checkboxes using the syntax:
92
+ - [ ] Task description.
93
+
94
+ Additional information can be included (indented).
95
+
96
+ If no actionable tasks are found, omit this tasks section.
97
+ `;
@@ -42,16 +42,7 @@ export const useAudioFile = (audioUrl: string, constraints?: MediaTrackConstrain
42
42
  audio.addEventListener(
43
43
  'canplay',
44
44
  async () => {
45
- log.info('starting...');
46
- try {
47
- // Try to play the audio
48
- await audio.play();
49
- resolve();
50
- } catch (playError) {
51
- log.error('Play failed', { playError });
52
- // Still resolve as the audio is ready, even if autoplay failed.
53
- resolve();
54
- }
45
+ resolve();
55
46
  },
56
47
  { once: true },
57
48
  );
@@ -14,20 +14,20 @@ import { type Transcriber } from '../transcriber';
14
14
  */
15
15
  export const useTranscriber = ({
16
16
  audioStreamTrack,
17
- onSegments,
18
- transcriberConfig,
19
17
  recorderConfig,
18
+ transcriberConfig,
19
+ onSegments,
20
20
  }: Partial<TranscriptionCapabilities.GetTranscriberProps>) => {
21
21
  const [getTranscriber] = useCapabilities(TranscriptionCapabilities.Transcriber);
22
22
 
23
23
  // Initialize audio transcription.
24
24
  const transcriber = useMemo<Transcriber | undefined>(() => {
25
- if (!onSegments || !audioStreamTrack || !getTranscriber) {
25
+ if (!getTranscriber || !audioStreamTrack || !onSegments) {
26
26
  return undefined;
27
27
  }
28
28
 
29
- return getTranscriber({ audioStreamTrack, onSegments, transcriberConfig, recorderConfig });
30
- }, [audioStreamTrack, onSegments, getTranscriber, transcriberConfig, recorderConfig]);
29
+ return getTranscriber({ audioStreamTrack, recorderConfig, transcriberConfig, onSegments });
30
+ }, [getTranscriber, audioStreamTrack, recorderConfig, transcriberConfig, onSegments]);
31
31
 
32
32
  useEffect(() => {
33
33
  return () => {
@@ -2,16 +2,17 @@
2
2
  // Copyright 2025 DXOS.org
3
3
  //
4
4
 
5
- import { useState, useEffect, useCallback } from 'react';
5
+ import { useCallback, useEffect, useState } from 'react';
6
6
 
7
7
  import { scheduleMicroTask } from '@dxos/async';
8
8
  import { Context } from '@dxos/context';
9
9
  import { log } from '@dxos/log';
10
10
  import { useSoundEffect } from '@dxos/react-ui-sfx';
11
11
 
12
+ import { type TranscriberParams } from '../transcriber';
13
+
12
14
  import { useAudioTrack } from './useAudioTrack';
13
15
  import { useTranscriber } from './useTranscriber';
14
- import { type TranscriberParams } from '../transcriber';
15
16
 
16
17
  export type UseVoiceInputProps = {
17
18
  active?: boolean;
@@ -9,7 +9,7 @@ import { Obj } from '@dxos/echo';
9
9
  import type { ObjectId } from '@dxos/keys';
10
10
  import { DataType } from '@dxos/schema';
11
11
 
12
- import { SerializationModel, DocumentAdapter, type ChunkRenderer } from './model';
12
+ import { type ChunkRenderer, DocumentAdapter, SerializationModel } from './model';
13
13
 
14
14
  const blockToMarkdown: ChunkRenderer<DataType.Message> = (
15
15
  message: DataType.Message,
@@ -18,7 +18,7 @@ const blockToMarkdown: ChunkRenderer<DataType.Message> = (
18
18
  ): string[] => {
19
19
  return [
20
20
  `###### ${message.sender.name}`,
21
- ...message.blocks.filter((block) => block.type === 'transcription').map((block) => block.text),
21
+ ...message.blocks.filter((block) => block._tag === 'transcript').map((block) => block.text),
22
22
  '',
23
23
  ];
24
24
  };
@@ -37,7 +37,7 @@ describe('SerializationModel', () => {
37
37
  sender: { name: 'Alice' },
38
38
  blocks: [
39
39
  {
40
- type: 'transcription',
40
+ _tag: 'transcript',
41
41
  started: createDate(),
42
42
  text: 'Hello world!',
43
43
  },
@@ -51,7 +51,7 @@ describe('SerializationModel', () => {
51
51
  }
52
52
 
53
53
  // Update message.
54
- message.blocks.push({ type: 'transcription', started: createDate(), text: 'Hello again!' });
54
+ message.blocks.push({ _tag: 'transcript', started: createDate(), text: 'Hello again!' });
55
55
  model.updateChunk(message);
56
56
  {
57
57
  const text = model.doc.toString();
@@ -74,7 +74,7 @@ describe('SerializationModel', () => {
74
74
  sender: { name: 'Alice' },
75
75
  blocks: [
76
76
  {
77
- type: 'transcription',
77
+ _tag: 'transcript',
78
78
  started: createDate(),
79
79
  text: 'Hello world!',
80
80
  },
@@ -92,7 +92,7 @@ describe('SerializationModel', () => {
92
92
  sender: { name: 'Bob' },
93
93
  blocks: [
94
94
  {
95
- type: 'transcription',
95
+ _tag: 'transcript',
96
96
  started: createDate(),
97
97
  text: 'Hello world!',
98
98
  },
@@ -119,7 +119,7 @@ describe('SerializationModel', () => {
119
119
  sender: { name: 'Alice' },
120
120
  blocks: [
121
121
  {
122
- type: 'transcription',
122
+ _tag: 'transcript',
123
123
  started: createDate(),
124
124
  text: 'Hello world!',
125
125
  },
@@ -131,7 +131,7 @@ describe('SerializationModel', () => {
131
131
  expect(view.state.doc.toString()).to.eq('###### Alice\nHello world!\n\n');
132
132
 
133
133
  // Update message (add block).
134
- message.blocks.push({ type: 'transcription', started: createDate(), text: 'Hello again!' });
134
+ message.blocks.push({ _tag: 'transcript', started: createDate(), text: 'Hello again!' });
135
135
  model.updateChunk(message);
136
136
  model.sync(adapter);
137
137
  expect(view.state.doc.toString()).to.eq('###### Alice\nHello world!\nHello again!\n\n');
@@ -150,7 +150,7 @@ describe('SerializationModel', () => {
150
150
  const message = Obj.make(DataType.Message, {
151
151
  created: createDate(),
152
152
  sender: { name: 'Bob' },
153
- blocks: [{ type: 'transcription', started: createDate(), text: 'Hello again!' }],
153
+ blocks: [{ _tag: 'transcript', started: createDate(), text: 'Hello again!' }],
154
154
  });
155
155
  model.appendChunk(message);
156
156
  model.sync(adapter);
@@ -8,7 +8,7 @@
8
8
 
9
9
  import { effect } from '@preact/signals-core';
10
10
 
11
- import { asyncTimeout, DeferredTask } from '@dxos/async';
11
+ import { DeferredTask, asyncTimeout } from '@dxos/async';
12
12
  import { LifecycleState, Resource } from '@dxos/context';
13
13
  import { type Queue } from '@dxos/echo-db';
14
14
  import { type FunctionExecutor } from '@dxos/functions';
@@ -5,8 +5,6 @@
5
5
  import { effect } from '@preact/signals-core';
6
6
  import { describe, test } from 'vitest';
7
7
 
8
- import { EdgeAiServiceClient, OllamaAiServiceClient } from '@dxos/ai';
9
- import { AI_SERVICE_ENDPOINT } from '@dxos/ai/testing';
10
8
  import { scheduleTaskInterval } from '@dxos/async';
11
9
  import { Context } from '@dxos/context';
12
10
  import { Obj } from '@dxos/echo';
@@ -51,33 +49,32 @@ const messages: MessageWithRangeId[] = [
51
49
  'in classical physics objects have well-defined properties such as position speed and momentum',
52
50
  ].map((string, index) =>
53
51
  Obj.make(DataType.Message, {
54
- created: new Date(Date.now() + 1000 * index).toISOString(),
52
+ created: new Date(Date.now() + 1_000 * index).toISOString(),
55
53
  sender,
56
- blocks: [{ type: 'transcription', started: new Date(Date.now() + 1000 * index).toISOString(), text: string }],
57
- rangeId: [],
58
- } as any),
54
+ blocks: [{ _tag: 'transcript', started: new Date(Date.now() + 1_000 * index).toISOString(), text: string }],
55
+ }),
59
56
  );
60
57
 
61
- const REMOTE_AI = true;
58
+ // const REMOTE_AI = true;
62
59
 
63
60
  describe.skip('SentenceNormalization', () => {
64
61
  const getExecutor = () => {
65
62
  return new FunctionExecutor(
66
63
  new ServiceContainer().setServices({
67
- ai: {
68
- client: REMOTE_AI
69
- ? new EdgeAiServiceClient({
70
- endpoint: AI_SERVICE_ENDPOINT.REMOTE,
71
- defaultGenerationOptions: {
72
- model: '@anthropic/claude-3-5-sonnet-20241022',
73
- },
74
- })
75
- : new OllamaAiServiceClient({
76
- overrides: {
77
- model: 'llama3.1:8b',
78
- },
79
- }),
80
- },
64
+ // ai: {
65
+ // client: REMOTE_AI
66
+ // ? new Edge AiServiceClient({
67
+ // endpoint: AI_SERVICE_ENDPOINT.REMOTE,
68
+ // defaultGenerationOptions: {
69
+ // model: '@anthropic/claude-3-5-sonnet-20241022',
70
+ // },
71
+ // })
72
+ // : new Ollama AiServiceClient({
73
+ // overrides: {
74
+ // model: 'llama3.1:8b',
75
+ // },
76
+ // }),
77
+ // },
81
78
  }),
82
79
  );
83
80
  };
@@ -100,6 +97,7 @@ describe.skip('SentenceNormalization', () => {
100
97
  buffer = sentences.slice(activeSentenceIndex);
101
98
  }
102
99
  }
100
+
103
101
  log.info('sentences', {
104
102
  originalMessages: JSON.stringify(messages, null, 2),
105
103
  sentences: JSON.stringify(sentences, null, 2),
@@ -2,15 +2,19 @@
2
2
  // Copyright 2025 DXOS.org
3
3
  //
4
4
 
5
+ // ISSUE(burdon): defineFunction
6
+ // @ts-nocheck
7
+
5
8
  import { Schema } from 'effect';
6
9
 
7
- import { DEFAULT_EDGE_MODEL, Message } from '@dxos/ai';
8
- import { AISession } from '@dxos/assistant';
10
+ import { AiService, DEFAULT_EDGE_MODEL } from '@dxos/ai';
11
+ import { AiSession } from '@dxos/assistant';
9
12
  import { Obj } from '@dxos/echo';
10
- import { AiService, defineFunction } from '@dxos/functions';
13
+ import { defineFunction } from '@dxos/functions';
11
14
  import { ObjectId } from '@dxos/keys';
12
15
  import { log } from '@dxos/log';
13
16
  import { DataType } from '@dxos/schema';
17
+ import { trim } from '@dxos/util';
14
18
 
15
19
  const MessageWithRangeId = Schema.extend(
16
20
  DataType.Message,
@@ -39,40 +43,6 @@ export const NormalizationOutput = Schema.Struct({
39
43
  });
40
44
  export interface NormalizationOutput extends Schema.Schema.Type<typeof NormalizationOutput> {}
41
45
 
42
- const prompt = `
43
- You are observing a real-time transcript of a single person speaking.
44
- The transcription is delivered in chunks of 10 seconds or less. As a result, individual sentences may be split across multiple messages, or multiple sentences may appear within a single message. Additionally, because this is real-time transcription, punctuation and capitalization may be incorrect or missing.
45
-
46
- # Task Description:
47
- - Your task is to detect and reconstruct broken or incomplete sentences by merging segments into coherent, grammatically correct sentences where appropriate.
48
-
49
- # Input Format:
50
- - The input is an array of messages that contain transcription blocks.
51
- - Each message is a DataType.Message, as described in the output format.
52
-
53
- # Output Format:
54
- - You have been provided with the tool that defines the output format; make sure to query it.
55
- - Do not output anything other than the expected format.
56
-
57
- # Segment Handling Rules:
58
- - Sort messages by timestamp.
59
- - Leave ID of the first message in a sentence.
60
- - Each divided sentence should be added to the output as a separate message or block within existing messages, preserving the original order.
61
- - If one message contains multiple sentences, you should not split them.
62
- - The last sentence may be incomplete; it should still be output as a separate block or message.
63
- - Keep track of the original timestamps of the messages and blocks and, use the timestamp of the first message/block of a sentence as the 'started' field of the merged message.
64
- - The 'rangeId' field of the merged message should include 'messageId'-s and 'rangeId'-s of all messages that have been used to construct this message.
65
-
66
- # Punctuation and Capitalization:
67
- - Do not rely on the original punctuation and capitalization—they may be incorrect.
68
- - Use logical reasoning to apply appropriate punctuation (e.g., period, comma, question mark, exclamation mark) and capitalization.
69
-
70
- # Restrictions:
71
- - Do not alter the order of the original messages; maintain the natural flow of speech.
72
- - Do not interpret or infer meaning beyond reconstructing sentences.
73
- - Do not add or remove any words or phrases.
74
- `;
75
-
76
46
  export const sentenceNormalization = defineFunction<NormalizationInput, NormalizationOutput>({
77
47
  description: 'Post process of transcription for sentence normalization',
78
48
  inputSchema: NormalizationInput,
@@ -80,21 +50,27 @@ export const sentenceNormalization = defineFunction<NormalizationInput, Normaliz
80
50
  handler: async ({ data: { messages }, context }) => {
81
51
  log.info('input', { messages });
82
52
  const ai = context.getService(AiService);
83
- const session = new AISession({ operationModel: 'configured' });
53
+ const session = new AiSession({ operationModel: 'configured' });
84
54
 
85
- const response = await session.runStructured(NormalizationOutput, {
86
- generationOptions: { model: DEFAULT_EDGE_MODEL },
55
+ // TODO(dmaretskyi): This got broken after effect-ai transition.
56
+ const response = session.runStructured(NormalizationOutput, {
57
+ generationOptions: {
58
+ model: DEFAULT_EDGE_MODEL,
59
+ },
87
60
  client: ai.client,
88
61
  tools: [],
89
62
  artifacts: [],
90
63
  history: [
91
- Obj.make(Message, {
92
- role: 'user',
93
- content: messages.map((message) => ({ type: 'text', text: JSON.stringify(message) }) as const),
64
+ Obj.make(DataType.Message, {
65
+ created: new Date().toISOString(),
66
+ sender: {
67
+ role: 'user',
68
+ },
69
+ blocks: messages.map((message) => ({ _tag: 'text', text: JSON.stringify(message) }) as const),
94
70
  }),
95
71
  ],
96
72
  prompt,
97
- });
73
+ }) as any;
98
74
 
99
75
  response.sentences.forEach((sentence) => {
100
76
  sentence.id = ObjectId.random();
@@ -103,3 +79,37 @@ export const sentenceNormalization = defineFunction<NormalizationInput, Normaliz
103
79
  return response;
104
80
  },
105
81
  });
82
+
83
+ const prompt = trim`
84
+ You are observing a real-time transcript of a single person speaking.
85
+ The transcription is delivered in chunks of 10 seconds or less. As a result, individual sentences may be split across multiple messages, or multiple sentences may appear within a single message. Additionally, because this is real-time transcription, punctuation and capitalization may be incorrect or missing.
86
+
87
+ # Task Description:
88
+ - Your task is to detect and reconstruct broken or incomplete sentences by merging segments into coherent, grammatically correct sentences where appropriate.
89
+
90
+ # Input Format:
91
+ - The input is an array of messages that contain transcription blocks.
92
+ - Each message is a DataType.Message, as described in the output format.
93
+
94
+ # Output Format:
95
+ - You have been provided with the tool that defines the output format; make sure to query it.
96
+ - Do not output anything other than the expected format.
97
+
98
+ # Segment Handling Rules:
99
+ - Sort messages by timestamp.
100
+ - Leave ID of the first message in a sentence.
101
+ - Each divided sentence should be added to the output as a separate message or block within existing messages, preserving the original order.
102
+ - If one message contains multiple sentences, you should not split them.
103
+ - The last sentence may be incomplete; it should still be output as a separate block or message.
104
+ - Keep track of the original timestamps of the messages and blocks and, use the timestamp of the first message/block of a sentence as the 'started' field of the merged message.
105
+ - The 'rangeId' field of the merged message should include 'messageId'-s and 'rangeId'-s of all messages that have been used to construct this message.
106
+
107
+ # Punctuation and Capitalization:
108
+ - Do not rely on the original punctuation and capitalization—they may be incorrect.
109
+ - Use logical reasoning to apply appropriate punctuation (e.g., period, comma, question mark, exclamation mark) and capitalization.
110
+
111
+ # Restrictions:
112
+ - Do not alter the order of the original messages; maintain the natural flow of speech.
113
+ - Do not interpret or infer meaning beyond reconstructing sentences.
114
+ - Do not add or remove any words or phrases.
115
+ `;
@@ -5,9 +5,7 @@
5
5
  import { Schema } from 'effect';
6
6
  import { useEffect, useMemo, useState } from 'react';
7
7
 
8
- import { EdgeAiServiceClient } from '@dxos/ai';
9
- import { AI_SERVICE_ENDPOINT } from '@dxos/ai/testing';
10
- import { extractionAnthropicFn, processTranscriptMessage } from '@dxos/assistant';
8
+ import { extractionAnthropicFunction, processTranscriptMessage } from '@dxos/assistant/extraction';
11
9
  import { scheduleTaskInterval } from '@dxos/async';
12
10
  import { Filter, type Queue } from '@dxos/client/echo';
13
11
  import { Context } from '@dxos/context';
@@ -17,7 +15,7 @@ import { FunctionExecutor, ServiceContainer } from '@dxos/functions';
17
15
  import { IdentityDid } from '@dxos/keys';
18
16
  import { log } from '@dxos/log';
19
17
  import { faker } from '@dxos/random';
20
- import { useQueue, type Space } from '@dxos/react-client/echo';
18
+ import { type Space, useQueue } from '@dxos/react-client/echo';
21
19
  import { DataType } from '@dxos/schema';
22
20
  import { Testing, seedTestData } from '@dxos/schema/testing';
23
21
 
@@ -69,19 +67,19 @@ export class MessageBuilder extends AbstractMessageBuilder {
69
67
  });
70
68
  }
71
69
 
72
- createBlock(): DataType.MessageBlock.Transcription {
70
+ createBlock(): DataType.MessageBlock.Transcript {
73
71
  let text = faker.lorem.paragraph();
74
72
  if (this._space) {
75
73
  const label = faker.commerce.productName();
76
74
  const obj = this._space.db.add(Obj.make(TestItem, { title: label, description: faker.lorem.paragraph() }));
77
75
  const dxn = Ref.make(obj).dxn.toString();
78
76
  const words = text.split(' ');
79
- words.splice(Math.floor(Math.random() * words.length), 0, `[${label}][${dxn}]`);
77
+ words.splice(Math.floor(Math.random() * words.length), 0, `[${label}](${dxn})`);
80
78
  text = words.join(' ');
81
79
  }
82
80
 
83
81
  return {
84
- type: 'transcription',
82
+ _tag: 'transcript',
85
83
  started: this.next().toISOString(),
86
84
  text,
87
85
  };
@@ -95,11 +93,13 @@ export class MessageBuilder extends AbstractMessageBuilder {
95
93
 
96
94
  // TODO(burdon): Reconcile with BlockBuilder.
97
95
  class EntityExtractionMessageBuilder extends AbstractMessageBuilder {
98
- AiService = new EdgeAiServiceClient({
99
- endpoint: AI_SERVICE_ENDPOINT.REMOTE,
100
- });
101
-
102
- executor = new FunctionExecutor(new ServiceContainer().setServices({ ai: { client: this.AiService } }));
96
+ executor = new FunctionExecutor(
97
+ new ServiceContainer().setServices({
98
+ // ai: {
99
+ // client: this.AiService,
100
+ // },
101
+ }),
102
+ );
103
103
 
104
104
  space: Space | undefined;
105
105
  currentMessage: number = 0;
@@ -130,7 +130,7 @@ class EntityExtractionMessageBuilder extends AbstractMessageBuilder {
130
130
  const { message: enhancedMessage } = await processTranscriptMessage({
131
131
  input: { message },
132
132
  executor: this.executor,
133
- function: extractionAnthropicFn,
133
+ function: extractionAnthropicFunction,
134
134
  });
135
135
 
136
136
  return enhancedMessage;
@@ -2,7 +2,7 @@
2
2
  // Copyright 2025 DXOS.org
3
3
  //
4
4
 
5
- export * from './audio-recorder';
5
+ export type * from './audio-recorder';
6
6
  export * from './media-stream-recorder';
7
7
  export * from './transcriber';
8
8
  export * from './transcription-manager';
@@ -10,7 +10,7 @@ import { invariant } from '@dxos/invariant';
10
10
  import { log } from '@dxos/log';
11
11
  import { trace } from '@dxos/tracing';
12
12
 
13
- import { type WavConfig, type AudioChunk, type AudioRecorder } from './audio-recorder';
13
+ import { type AudioChunk, type AudioRecorder, type WavConfig } from './audio-recorder';
14
14
 
15
15
  let initializingPromise: Promise<void> | undefined;
16
16
 
@@ -4,6 +4,7 @@
4
4
 
5
5
  import fs from 'node:fs';
6
6
  import path from 'node:path';
7
+
7
8
  import { describe, expect, test } from 'vitest';
8
9
  import { WaveFile } from 'wavefile';
9
10
 
@@ -11,7 +12,7 @@ import { Trigger } from '@dxos/async';
11
12
  import { log } from '@dxos/log';
12
13
  import { type DataType } from '@dxos/schema';
13
14
  import { openAndClose } from '@dxos/test-utils';
14
- import { trace, TRACE_PROCESSOR } from '@dxos/tracing';
15
+ import { TRACE_PROCESSOR, trace } from '@dxos/tracing';
15
16
 
16
17
  import { type AudioChunk, type AudioRecorder, Transcriber } from '../transcriber';
17
18
  import { mergeFloat64Arrays } from '../util';
@@ -93,7 +94,7 @@ describe.skip('Transcriber', () => {
93
94
  });
94
95
 
95
96
  test.skip('transcription of audio recording', { timeout: 10_000 }, async () => {
96
- const trigger = new Trigger<DataType.MessageBlock.Transcription[]>({ autoReset: true });
97
+ const trigger = new Trigger<DataType.MessageBlock.Transcript[]>({ autoReset: true });
97
98
  const recorder = new MockAudioRecorder({
98
99
  buffer: await readFile('test.wav'),
99
100
  chunkDuration: 3_000,
@@ -127,7 +128,7 @@ describe.skip('Transcriber', () => {
127
128
  });
128
129
 
129
130
  test.skip('transcription of audio recording with overlapping chunks', { timeout: 20_000 }, async () => {
130
- const trigger = new Trigger<DataType.MessageBlock.Transcription[]>({ autoReset: true });
131
+ const trigger = new Trigger<DataType.MessageBlock.Transcript[]>({ autoReset: true });
131
132
  const recorder = new MockAudioRecorder({
132
133
  buffer: await readFile('test.wav'),
133
134
  chunkDuration: 3_000,
@@ -10,10 +10,11 @@ import { log } from '@dxos/log';
10
10
  import { type DataType } from '@dxos/schema';
11
11
  import { trace } from '@dxos/tracing';
12
12
 
13
- import { type AudioRecorder, type AudioChunk } from './audio-recorder';
14
13
  import { TRANSCRIPTION_URL } from '../types';
15
14
  import { mergeFloat64Arrays } from '../util';
16
15
 
16
+ import { type AudioChunk, type AudioRecorder } from './audio-recorder';
17
+
17
18
  type WhisperWord = {
18
19
  word: string;
19
20
 
@@ -69,7 +70,7 @@ export type TranscriberParams = {
69
70
  * Callback to handle the transcribed segments, after all segment transformers are applied.
70
71
  * @param segments - The transcribed segments.
71
72
  */
72
- onSegments: (segments: DataType.MessageBlock.Transcription[]) => Promise<void>;
73
+ onSegments: (segments: DataType.MessageBlock.Transcript[]) => Promise<void>;
73
74
  };
74
75
 
75
76
  /**
@@ -220,10 +221,7 @@ export class Transcriber extends Resource {
220
221
  return segments;
221
222
  }
222
223
 
223
- private _alignSegments(
224
- segments: WhisperSegment[],
225
- originalChunks: AudioChunk[],
226
- ): DataType.MessageBlock.Transcription[] {
224
+ private _alignSegments(segments: WhisperSegment[], originalChunks: AudioChunk[]): DataType.MessageBlock.Transcript[] {
227
225
  // Absolute zero for all relative timestamps in the segments.
228
226
  const zeroTimestamp = originalChunks.at(0)!.timestamp;
229
227
 
@@ -252,7 +250,7 @@ export class Transcriber extends Resource {
252
250
 
253
251
  // Add absolute timestamp to each segment.
254
252
  return filteredSegments.map((segment) => ({
255
- type: 'transcription',
253
+ _tag: 'transcript',
256
254
  started: new Date(zeroTimestamp + segment.start * 1_000).toISOString(),
257
255
  text: segment.text.trim(),
258
256
  }));