arcane-os 0.5.14 → 0.5.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -88,6 +88,7 @@ own asynchronous work, cancellation, and backpressure.
88
88
  | [`Mail.js`](#mailjs) | esm | Builds complete reports and prefers the native mail capability with an explicit HTTP transport fallback. | Browser/native hybrid + cloud | Mail inputs/results normalized; transport failures mixed. |
89
89
  | [`MailOutbox.mjs`](#mailoutboxmjs) | esm | Persists complete mail reports before delivery and normalizes idempotent enqueue, retry, reconciliation, and invalid-record maintenance. | Browser/native WebView or compatible injected host | Complete records, full work, cancellation, and lifecycle states normalized; storage, lock, and delivery failures coded. |
90
90
  | [`MailTransport.mjs`](#mailtransportmjs) | esm | Sends one complete mail report to a normalized HTTP(S) endpoint. | Browser/server with fetch + cloud | Normalized endpoint and transport errors; remote detail preserved. |
91
+ | [`MarkdownSpeech.js`](#markdownspeechjs) | esm | Removes repeated Markdown formatting marks from streamed narration. | Cross-host | Speech-only filtering; single marks and ordinary punctuation remain literal. |
91
92
  | [`Marked.min.js`](#markedminjs) | esm | Vendored Marked 18.0.5 Markdown lexer, parser, renderer, extension, and walk-token API. | Cross-host vendor module | Vendor-native Marked contract. |
92
93
  | [`MD.js`](#mdjs) | esm | Renders complete Markdown with Marked and exposes the complete rendered markup. | Browser / native WebView | Complete raw and rendered Marked values; parse errors vendor-native. |
93
94
  | [`MemoryRecords.js`](#memoryrecordsjs) | esm | Normalizes memory content and detects meaningful stored memory. | Cross-host | Fully normalized string/boolean results. |
@@ -274,7 +275,7 @@ rejected. It propagates the caller-owned signal and returns a playable `Blob`;
274
275
  it does not independently choose a provider, cloud fallback, model, runtime, or
275
276
  voice policy for the application. `streamTTS(text='',end=false,options={})` and
276
277
  `finishTTS()` use this same request boundary. The third-argument options below
277
- are available in SDK `0.5.12`:
278
+ are available in SDK `0.5.12`, with `textFormat` added in SDK `0.5.16`:
278
279
 
279
280
  | Field | Default | Meaning |
280
281
  | --- | --- | --- |
@@ -282,12 +283,24 @@ are available in SDK `0.5.12`:
282
283
  | `speed` | Current `ai.voiceSpeed` | A supplied positive speed is captured for those segments and forwarded to `fetchTTS()`. It does not change `ai.voiceSpeed`. |
283
284
  | `pauseAfterMs` | `0` | Finite, nonnegative milliseconds placed after the final extracted segment on the existing audio clock. Invalid values throw `RangeError`; no pause is inserted between this call's other segments. |
284
285
  | `waitForPlayback` | `false` | Omission retains the preparation promise. With `true`, the promise resolves after every extracted segment reaches a terminal playback state: `true` when all naturally end, or `false` after terminal cancellation or failure. |
286
+ | `textFormat` | `plain` for a new stream | `markdown` removes repeated same formatting marks (`*`, `#`, `_`, backtick, `~`) before segmentation. Single marks and ordinary punctuation remain literal. The selected format lasts through this producer's terminal flush. |
285
287
 
286
288
  The voice and speed use the existing `fetchTTS()` validation and error path.
287
- No option rewrites the submitted text. Overrides belong to the segments
289
+ Plain mode preserves the submitted text. Markdown mode changes narration only;
290
+ it leaves displayed, stored, and model content untouched. It is a narrow
291
+ formatting-mark filter, not a full Markdown parser: links, code contents, list
292
+ text, and other characters remain literal. A trailing candidate mark waits for
293
+ the next character so a repeated run split across chunks is still omitted.
294
+ Ordinary prose streams immediately. `end:true`, `finishTTS()`, muted terminal
295
+ calls, and `stopAudio()` clear the formatting state. An explicit format change
296
+ flushes pending formatting marks under the preceding mode before accepting
297
+ the next text. `fetchTTS()` retains its exact-text contract.
298
+
299
+ Voice, speed, pause, and playback overrides belong to the segments
288
300
  extracted in that invocation, including any text buffered by an earlier call.
289
- Options are not retained with an unfinished `end:false` remainder; a later
290
- call supplies its own options, and `finishTTS()` uses defaults. Use `end:true`
301
+ Those overrides are not retained with an unfinished `end:false` remainder; a
302
+ later call supplies its own options, and `finishTTS()` uses their defaults
303
+ while retaining the active text format. Use `end:true`
291
304
  for a complete passage. A call extracting no segments resolves
292
305
  `true` without waiting for earlier jobs; `finishTTS()` remains a preparation
293
306
  flush, not a queue-wide playback barrier. A muted call resolves `false`.
@@ -2244,6 +2257,41 @@ import * as module from '/arcane/modules/MailTransport.mjs';
2244
2257
  console.log(Object.keys(module));
2245
2258
  ```
2246
2259
 
2260
+ ## MarkdownSpeech.js
2261
+
2262
+ ### Overview
2263
+
2264
+ Filters repeated same Markdown formatting marks from narration before speech
2265
+ segmentation. The filter recognizes `*`, `#`, `_`, backtick, and `~` runs of two
2266
+ or more, including runs arriving across separate chunks. Single marks,
2267
+ ellipses, quoted sentence endings, whitespace, and all other text remain
2268
+ literal. It neither interprets links nor changes language or voice.
2269
+
2270
+ ### Public surface
2271
+
2272
+ `MarkdownSpeech`; `append(text='',end=false)`, `reset()`.
2273
+
2274
+ Exact exports: `MarkdownSpeech`.
2275
+
2276
+ ### Availability and normalization
2277
+
2278
+ **Cross-host.** Plain JavaScript with no dependencies. `append()` returns only
2279
+ the newly available narration. Only a trailing candidate marker and whether
2280
+ it repeats are retained; ordinary text is emitted immediately. Terminal
2281
+ `append(text,true)` flushes a single pending mark and resets state. `reset()`
2282
+ clears pending formatting state when its narration is cancelled. Non-string
2283
+ input throws `TypeError`.
2284
+
2285
+ ### Example
2286
+
2287
+ ```javascript
2288
+ import {MarkdownSpeech} from '/arcane/modules/MarkdownSpeech.js';
2289
+
2290
+ const speech = new MarkdownSpeech();
2291
+ const first = speech.append('**Hello'); // Hello
2292
+ const last = speech.append('**...', true); // ...
2293
+ ```
2294
+
2247
2295
  ## Marked.min.js
2248
2296
 
2249
2297
  ### Overview
@@ -2849,8 +2897,9 @@ console.log(Object.keys(module));
2849
2897
 
2850
2898
  ### Overview
2851
2899
 
2852
- Preserves exact nonblank text as one segment, queues latest-request speech
2853
- synthesis, and controls lookahead HTML audio playback.
2900
+ Preserves exact nonblank text, admits complete speech segments according to the
2901
+ selected client's advertised capacity, and plays indexed HTML audio in exact
2902
+ input order.
2854
2903
 
2855
2904
  ### Public surface
2856
2905
 
@@ -2888,6 +2937,26 @@ freezing it. `prepare()` likewise preserves each nonblank part's exact `input`
2888
2937
  string while normalizing its other playback fields into a new mutable record.
2889
2938
  The class applies no part-count, character-count, pause, or input upper cap.
2890
2939
 
2940
+ ### Admission and playback order
2941
+
2942
+ When `speech` exposes both `fetchTTS(payload, signal)` and
2943
+ `providerRuntime.status('tts', {execution:true}).execution.maxConcurrentRequests`
2944
+ as a positive safe integer, `prepare()` submits every complete `parts` entry
2945
+ immediately. `SpeechPlayback` does not create a second limiter: the provider
2946
+ runtime owns bounded FIFO admission. Browser Kokoro defaults that capacity to
2947
+ four, so up to four segments can synthesize while later submissions wait in the
2948
+ provider queue. A later segment may finish first, but its Blob URL stays at its
2949
+ original index and is never played ahead of an earlier segment.
2950
+
2951
+ If that capacity is absent, invalid, or unavailable, `SpeechPlayback` retains
2952
+ the compatible serialized path and prepares only one lookahead segment. This is
2953
+ the default for `Arcane.speech.synthesize` and custom clients, so native hosts
2954
+ with one synthesis lock are not driven concurrently. Pause and Resume control
2955
+ the same supplied `audio` element. Stop aborts every owned synthesis signal and
2956
+ releases prepared URLs. Replay keeps completed URLs and still-pending provider
2957
+ work, then re-submits only failed missing provider segments before starting
2958
+ again from index zero.
2959
+
2891
2960
  Every preparation owns an operation ID and one AbortController for each active
2892
2961
  synthesis segment or playback delay. Replacement,
2893
2962
  `stop()`, `cancel()`, and `destroy()` abort their owned signals, suppress stale
@@ -2895,9 +2964,10 @@ settlement, release Blob URLs, and publish synchronous
2895
2964
  `speech-playback-state` occurrences through `globalThis.arcaneEvents`.
2896
2965
  Subscribers receive mutable public state detail. The detail contains
2897
2966
  `state`, `message`, `key`, `index`, `total`, `producing`, `buffered`, `hasAudio`,
2898
- `operationId`, `code`, and `reason`; provider rejection remains preserved to the
2899
- `prepare()` caller. `destroy()` also removes every audio listener and disposes
2900
- its per-instance canonical source handle; repeated destroy returns
2967
+ `operationId`, `code`, and `reason`; a first-segment provider rejection remains
2968
+ preserved to the `prepare()` caller. Later failures surface when ordered
2969
+ playback reaches that segment. `destroy()` also removes every audio listener
2970
+ and disposes its per-instance canonical source handle; repeated destroy returns
2901
2971
  `false`. Signal abortion proves delivery suppression; whether provider work
2902
2972
  actually stops remains the selected provider's cancellation boundary.
2903
2973
 
@@ -2934,23 +3004,38 @@ const audio = document.body.appendChild(document.createElement('audio'));
2934
3004
  audio.controls = true;
2935
3005
  const speech = new SpeechPlayback({
2936
3006
  audio,
2937
- speech: globalThis.ai,
2938
- model: 'caller-selected-model',
2939
- voice: 'caller-selected-voice',
2940
- responseFormat: 'wav'
3007
+ speech: globalThis.ai
2941
3008
  });
2942
- const speakButton = document.body.appendChild(document.createElement('button'));
2943
- speakButton.type = 'button';
2944
- speakButton.textContent = 'Speak';
2945
- speakButton.addEventListener('click', async () => {
3009
+ const button = document.body.appendChild(document.createElement('button'));
3010
+ button.textContent = 'Speak';
3011
+ button.addEventListener('click', async function speakCompleteSegments() {
3012
+ await globalThis.ai.setSpeechMuted(false);
2946
3013
  await speech.prepare({
2947
- key: 'ready',
2948
- parts: ['Arcane is ready.'],
3014
+ parts: [
3015
+ 'First complete segment.',
3016
+ 'Second complete segment.'
3017
+ ],
2949
3018
  autoplay: true
2950
3019
  });
2951
3020
  });
2952
3021
  ```
2953
3022
 
3023
+ With the default browser speech configuration, both parts enter its capacity-4
3024
+ queue immediately and still play first, then second. Inspect the selected
3025
+ execution device without guessing from console warnings:
3026
+
3027
+ ```javascript
3028
+ const execution = globalThis.ai.providerRuntime.status(
3029
+ 'tts',
3030
+ { execution: true }
3031
+ ).execution;
3032
+ console.log(execution.selectedDevice, execution.maxConcurrentRequests);
3033
+ ```
3034
+
3035
+ `requestedDevice:'auto'` attempts the complete ONNX Worker/session pool on
3036
+ WebGPU and recreates it on WASM if WebGPU loading fails. The reported selected
3037
+ device proves provider selection, not physical GPU kernel overlap.
3038
+
2954
3039
  ## StaticDocumentCatalog.js
2955
3040
 
2956
3041
  ### Overview
@@ -18,7 +18,7 @@ This table is the Node `package.json#exports` map: it defines package
18
18
  entrypoints for SDK/tooling code. It is distinct from the generated browser
19
19
  import map that resolves application-facing `arcane/*` modules and the focused
20
20
  EventManager entry. See [browser runtime delivery](protocols.md#browser-runtime-delivery)
21
- for the installed-inventory-derived physical-runtime contract in SDK `0.5.14`.
21
+ for the installed-inventory-derived physical-runtime contract in SDK `0.5.16`.
22
22
 
23
23
  | Specifier | Purpose |
24
24
  | --- | --- |
@@ -751,7 +751,7 @@ deterministic map. The package root also contains the public
751
751
  {
752
752
  schemaVersion: 1,
753
753
  kind: 'arcane-app-runtime-projection',
754
- sdkVersion: '0.5.14',
754
+ sdkVersion: '0.5.16',
755
755
  pathPrefix: 'arcane/',
756
756
  files: [{path}]
757
757
  }
@@ -1500,6 +1500,14 @@ specifier `arcane-os/speech-playback` works in Node package consumers and maps
1500
1500
  to the same runtime module in managed browsers. `arcane/SpeechPlayback` is the
1501
1501
  direct browser runtime name generated into the managed import map.
1502
1502
 
1503
+ When the supplied `speech` client has `fetchTTS` and reports a positive
1504
+ `providerRuntime.status('tts', {execution:true}).execution.maxConcurrentRequests`,
1505
+ `prepare()` submits every complete part immediately. The provider owns its
1506
+ bounded FIFO queue (browser Kokoro defaults to four), while Blob URLs and
1507
+ playback remain in exact input order. A native or custom client without that
1508
+ advertised capacity stays serialized with one lookahead. Replay keeps completed
1509
+ URLs and pending requests while retrying failed missing provider segments.
1510
+
1503
1511
  ### Signature and result
1504
1512
 
1505
1513
  ```text
@@ -1516,7 +1524,18 @@ runtime module; provider and media availability remain host-owned.
1516
1524
 
1517
1525
  ```javascript
1518
1526
  import SpeechPlayback from 'arcane-os/speech-playback';
1519
- const playback=new SpeechPlayback({audio,speech});
1527
+
1528
+ const audio=document.body.appendChild(document.createElement('audio'));
1529
+ const playback=new SpeechPlayback({audio,speech:globalThis.ai});
1530
+ const button=document.body.appendChild(document.createElement('button'));
1531
+ button.textContent='Speak';
1532
+ button.addEventListener('click',async function speakCompleteSegments(){
1533
+ await globalThis.ai.setSpeechMuted(false);
1534
+ await playback.prepare({
1535
+ parts:['First complete segment.','Second complete segment.'],
1536
+ autoplay:true
1537
+ });
1538
+ });
1520
1539
  ```
1521
1540
 
1522
1541
  ## SPEECH_PLAYBACK_STATE_EVENT
@@ -1546,7 +1565,9 @@ console.log(SPEECH_PLAYBACK_STATE_EVENT);
1546
1565
 
1547
1566
  ### Overview
1548
1567
 
1549
- Named binding for the same canonical class exposed as the speech-playback default. `prepare()` preserves each nonblank part's exact input string without trimming, splitting, or freezing that content.
1568
+ Named binding for the same capability-aware, exact-order canonical class exposed
1569
+ as the speech-playback default. `prepare()` preserves each nonblank part's exact
1570
+ input string without trimming, splitting, or freezing that content.
1550
1571
 
1551
1572
  ### Signature and result
1552
1573
 
@@ -1557,7 +1578,9 @@ new SpeechPlayback(options={})
1557
1578
  ### Availability and normalization
1558
1579
 
1559
1580
  **Node with injected media adapters, or browser/native WebView media.** Binding
1560
- identity equals the default export; provider and media availability remain host-owned.
1581
+ identity equals the default export. Provider-advertised capacity enables eager
1582
+ submission while that provider owns its bound; native and custom speech stays
1583
+ serialized, and media availability remains host-owned.
1561
1584
 
1562
1585
  ### Example
1563
1586
 
@@ -3440,7 +3463,7 @@ workspace it additionally returns the exact installed package authority:
3440
3463
  packageSource,
3441
3464
  canonicalPackageRoot,
3442
3465
  packageName: 'arcane-os',
3443
- packageVersion: '0.5.14',
3466
+ packageVersion: '0.5.16',
3444
3467
  runtimeRoot,
3445
3468
  browserRuntimeRoot
3446
3469
  }
@@ -3448,9 +3471,9 @@ workspace it additionally returns the exact installed package authority:
3448
3471
  ```
3449
3472
 
3450
3473
  The dependency can be named `arcane-os` or be one exact npm alias for
3451
- `npm:arcane-os@0.5.14`. The selected installation must still be one direct,
3474
+ `npm:arcane-os@0.5.16`. The selected installation must still be one direct,
3452
3475
  physical, non-link package directory whose manifest identifies exactly as
3453
- `arcane-os@0.5.14`; duplicate canonical/alias declarations reject.
3476
+ `arcane-os@0.5.16`; duplicate canonical/alias declarations reject.
3454
3477
  `allowMissingManagedImportMap` is an internal packaging/development seam. An
3455
3478
  ordinary caller should leave it `false`.
3456
3479
 
@@ -8,6 +8,8 @@ For the smallest first request, start with the [browser speech quick start](../.
8
8
 
9
9
  The demo deliberately omits `tts.execution` in `speechConfiguration(dbopfs)` so it consumes the SDK default `{device:'auto',maxConcurrentRequests:4}`. Change that application's TTS record to `execution:{device:'wasm',maxConcurrentRequests:1}` to use less memory, or choose `webgpu` explicitly. Allowed capacities are 1 through 4; Whisper and the LLM retain capacity one.
10
10
 
11
+ The maintained Kokoro selection uses `fp32` because [Kokoro.js recommends `fp32` when using WebGPU](https://github.com/hexgrad/kokoro/tree/main/kokoro.js#usage). Automatic fallback keeps the same dtype on WASM. `selectedDevice` identifies the loaded route; it does not validate speech correctness or audio quality, so evaluate actual output for each browser and device combination your application supports.
12
+
11
13
  Capacity 4 means up to four segments synthesize at once. Segment 5 and later wait in the SDK's FIFO queue; they are not dropped. Synthesis may finish out of order, but playback waits for earlier segments and plays exact input order. Each slot owns a Worker/model session, so raising capacity trades memory for latency.
12
14
 
13
15
  Use the shared Speech component to load/unmute speech, then select **Inspect speech execution** below Chat. The button reads `ai.providerRuntime.status('tts',{execution:true}).execution` and displays requested device, selected device, capacity, active requests, and automatic WASM fallback. It is a snapshot at the moment clicked. `selectedDevice:null` means no pool is selected; `webgpu` reports the upstream execution-provider selection and does not prove physical GPU kernel overlap or audio quality. Auto may fall back to WASM with the same selected model/dtype; explicit WebGPU reports an error when it cannot load.
@@ -39,7 +41,7 @@ Model weights are deliberately excluded from Git. Place the maintained GGUF shar
39
41
 
40
42
  The maintained model descriptors include each shard's known byte length as progress metadata. During a cold install, the SDK reports aggregate loaded and remaining bytes, transfer speed, estimated time, and active file or Range transfers while Chat presents the existing activation progress bar. Completed shards and completed Range parts within a shard remain available after interruption, so retry downloads only missing work.
41
43
 
42
- The app gives the SDK browser-speech owner the existing Whisper tiny.en FP32 and Kokoro 82M Q8 descriptors. The app does not import those runtimes or create a speech worker itself.
44
+ The app gives the SDK browser-speech owner the existing Whisper tiny.en FP32 and Kokoro 82M FP32 descriptors. The app does not import those runtimes or create a speech worker itself.
43
45
 
44
46
  ## Local retrieval
45
47
 
@@ -435,12 +435,13 @@ function speechConfiguration(dbopfs) {
435
435
  },
436
436
  tts: {
437
437
  // Omit execution to use the SDK's auto device and four synthesis slots.
438
+ // Kokoro.js recommends fp32 for the WebGPU route attempted by auto.
438
439
  providerId: "wasm-ai-demo-browser-kokoro",
439
440
  model: {
440
441
  id: "onnx-community/Kokoro-82M-v1.0-ONNX",
441
442
  repository: "onnx-community/Kokoro-82M-v1.0-ONNX",
442
443
  revision: "1939ad2a8e416c0acfeecc08a694d14ef25f2231",
443
- dtype: "q8",
444
+ dtype: "fp32",
444
445
  defaultVoice: "af_heart",
445
446
  },
446
447
  runtime: {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "arcane-os",
3
- "version": "0.5.14",
3
+ "version": "0.5.16",
4
4
  "description": "Arcane OS JavaScript SDK, project-local CLI, browser runtime, and repository-portable application packager.",
5
5
  "type": "module",
6
6
  "main": "./src/index.mjs",
@@ -64,9 +64,9 @@
64
64
  "test": "npm run test:unit && npm run test:functional && npm run test:integration && npm run test:regression",
65
65
  "test:release": "node ./bin/arcane-test.mjs test/npm-release.test.mjs",
66
66
  "test:unit": "node ./bin/arcane-test.mjs test/app-descriptor.test.mjs test/app-schema.test.mjs test/contracts.test.mjs test/doctor.test.mjs test/mail-credentials.test.mjs test/mail-outbox.test.mjs test/mail-public-api.test.mjs test/mail-send.test.mjs test/mail-transport.test.mjs test/targets.test.mjs test/workspace-operation-lock.test.mjs",
67
- "test:functional": "node ./bin/arcane-test.mjs test/browser-speech-providers.test.mjs test/browser-wasm-gpu-notice.test.mjs test/browser-wasm-download-resume.test.mjs test/cli.test.mjs test/dbopfs-document-library.test.mjs test/dev-server.test.mjs test/dom-event-instrumentation.test.mjs test/event-manager.test.mjs test/events.test.mjs test/import-map.test.mjs test/mail-cli.test.mjs test/mail-runtime.test.mjs test/mail-server.test.mjs test/packaging.test.mjs test/persistent-ai-chat-session.test.mjs test/reference-completeness.test.mjs test/runtime-api-behavior.test.mjs test/runtime.test.mjs test/scaffold.test.mjs test/site.test.mjs test/update-check.test.mjs",
67
+ "test:functional": "node ./bin/arcane-test.mjs test/browser-speech-providers.test.mjs test/browser-wasm-gpu-notice.test.mjs test/browser-wasm-download-resume.test.mjs test/cli.test.mjs test/dbopfs-document-library.test.mjs test/dev-server.test.mjs test/dom-event-instrumentation.test.mjs test/event-manager.test.mjs test/events.test.mjs test/import-map.test.mjs test/mail-cli.test.mjs test/mail-runtime.test.mjs test/mail-server.test.mjs test/packaging.test.mjs test/persistent-ai-chat-session.test.mjs test/reference-completeness.test.mjs test/runtime-api-behavior.test.mjs test/runtime.test.mjs test/scaffold.test.mjs test/speech-playback.test.mjs test/site.test.mjs test/update-check.test.mjs",
68
68
  "test:integration": "node ./bin/arcane-test.mjs test/integrated-shared.test.mjs test/integrated-workspace.test.mjs test/mail-browser.test.mjs test/native-plan.test.mjs test/native-provider-loader.test.mjs test/npm-release.test.mjs test/release-bundle.test.mjs test/release-capability-smoke.test.mjs test/shared-payload-batch.test.mjs test/tarball.test.mjs test/wllama-webgpu-runtime.test.mjs",
69
- "test:regression": "node ./bin/arcane-test.mjs test/channel-workflows.test.mjs test/logging-regression.test.mjs test/native-provider-generation.test.mjs test/speech-queue-regression.test.mjs test/testing.test.mjs test/test-sets.test.mjs",
69
+ "test:regression": "node ./bin/arcane-test.mjs test/channel-workflows.test.mjs test/logging-regression.test.mjs test/markdown-speech.test.mjs test/native-provider-generation.test.mjs test/speech-queue-regression.test.mjs test/testing.test.mjs test/test-sets.test.mjs",
70
70
  "check": "node tools/check-source.mjs && npm test",
71
71
  "check:release": "node tools/check-source.mjs --package-only && npm run test:release",
72
72
  "pack:local": "npm pack --ignore-scripts",
@@ -2817,7 +2817,9 @@
2817
2817
  ){
2818
2818
  const aiRuntime=boundChatAI??globalThis.ai;
2819
2819
  if(typeof aiRuntime?.streamTTS==='function'){
2820
- void Promise.resolve(aiRuntime.streamTTS(text)).catch(
2820
+ void Promise.resolve(aiRuntime.streamTTS(
2821
+ text,false,{textFormat:'markdown'}
2822
+ )).catch(
2821
2823
  reportTTSError
2822
2824
  );
2823
2825
  }else{
@@ -13,6 +13,7 @@ import {
13
13
  } from './AIProviderRuntime.js';
14
14
  import {normalizeOllamaModelIdentifier} from './OllamaModelIdentifier.js';
15
15
  import {arcaneLogging} from 'arcane-os/logging';
16
+ import {MarkdownSpeech} from './MarkdownSpeech.js';
16
17
 
17
18
  const completeValue=(value)=>value;
18
19
 
@@ -2029,6 +2030,8 @@ class AI {
2029
2030
  }
2030
2031
 
2031
2032
  audioMessageChunks='';
2033
+ #speechTextFormat='plain';
2034
+ #markdownSpeech=new MarkdownSpeech();
2032
2035
  sourceNodes=[];
2033
2036
  isSpeaking=false;
2034
2037
  audioContext=null;
@@ -5731,6 +5734,8 @@ class AI {
5731
5734
  if(this.muted){
5732
5735
  if(end){
5733
5736
  this.audioMessageChunks='';
5737
+ this.#speechTextFormat='plain';
5738
+ this.#markdownSpeech.reset();
5734
5739
  }
5735
5740
  this.#traceSpeech('streamTTS.result',{callId,result:false,reason:'muted'});
5736
5741
  return Promise.resolve(false);
@@ -5742,11 +5747,24 @@ class AI {
5742
5747
  throw new RangeError('Speech pauses must be a nonnegative number of milliseconds.');
5743
5748
  }
5744
5749
 
5745
- this.audioMessageChunks+=String(text||'');
5750
+ const textFormat=options.textFormat??this.#speechTextFormat;
5751
+ if(textFormat!=='plain'&&textFormat!=='markdown'){
5752
+ throw new TypeError('Speech textFormat must be plain or markdown.');
5753
+ }
5754
+ let speechText='';
5755
+ if(textFormat!==this.#speechTextFormat){
5756
+ speechText=this.#markdownSpeech.append('',true);
5757
+ }
5758
+ const incomingText=String(text||'');
5759
+ speechText+=textFormat==='markdown'
5760
+ ?this.#markdownSpeech.append(incomingText,end)
5761
+ :incomingText;
5762
+ this.#speechTextFormat=end?'plain':textFormat;
5763
+ this.audioMessageChunks+=speechText;
5746
5764
  const outputs=this.#extractSpeechSegments(end);
5747
5765
  this.#traceSpeech('streamTTS.segments',{
5748
5766
  callId,segments:outputs,remainder:this.audioMessageChunks,
5749
- segmentation:this.ttsSegmentation
5767
+ segmentation:this.ttsSegmentation,textFormat,speechText
5750
5768
  });
5751
5769
 
5752
5770
  if(!outputs.length){
@@ -6414,6 +6432,8 @@ class AI {
6414
6432
  this.speechScheduleContext=null;
6415
6433
  this.speechScheduleTime=0;
6416
6434
  this.audioMessageChunks='';
6435
+ this.#speechTextFormat='plain';
6436
+ this.#markdownSpeech.reset();
6417
6437
  this.#clearSpeechUnlock();
6418
6438
 
6419
6439
  for(const job of this.speechJobs){
@@ -0,0 +1,50 @@
1
+ const FORMATTING_MARKERS=new Set(['*','#','_','`','~']);
2
+
3
+ // Speech-only filtering; single markers and all other text remain literal.
4
+ class MarkdownSpeech {
5
+ #pendingMarker='';
6
+ #repeated=false;
7
+
8
+ append(text='',end=false){
9
+ if(typeof text!=='string'){
10
+ throw new TypeError('Markdown speech input must be text.');
11
+ }
12
+
13
+ let narration='';
14
+
15
+ for(const character of text){
16
+ if(character===this.#pendingMarker){
17
+ this.#repeated=true;
18
+ continue;
19
+ }
20
+
21
+ if(this.#pendingMarker&&!this.#repeated){
22
+ narration+=this.#pendingMarker;
23
+ }
24
+ this.reset();
25
+
26
+ if(FORMATTING_MARKERS.has(character)){
27
+ // Wait for the next character to distinguish one mark from a run.
28
+ this.#pendingMarker=character;
29
+ }else{
30
+ narration+=character;
31
+ }
32
+ }
33
+
34
+ if(end){
35
+ if(this.#pendingMarker&&!this.#repeated){
36
+ narration+=this.#pendingMarker;
37
+ }
38
+ this.reset();
39
+ }
40
+
41
+ return narration;
42
+ }
43
+
44
+ reset(){
45
+ this.#pendingMarker='';
46
+ this.#repeated=false;
47
+ }
48
+ }
49
+
50
+ export {MarkdownSpeech};