arcane-os 0.5.16 → 0.5.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,12 +5,12 @@
5
5
  "repository": "https://github.com/TheWizardNexus/arcane-os-sdk.git",
6
6
  "branch": "main",
7
7
  "path": "runtime/arcane",
8
- "sdkVersion": "0.5.16",
8
+ "sdkVersion": "0.5.17",
9
9
  "protocol": "arcane/1"
10
10
  },
11
11
  "artifactCount": 85,
12
12
  "javascriptArtifactCount": 83,
13
- "esmExportCount": 354,
13
+ "esmExportCount": 356,
14
14
  "artifacts": [
15
15
  {
16
16
  "file": "runtime/arcane/modules/AI.js",
@@ -29,8 +29,8 @@
29
29
  "summary": "Owns provider-selectable chat and the one-time caller-authority browser STT/TTS configuration, lifecycle, synthesis, transcription, and playback boundary.",
30
30
  "availability": "Browser + native bridge + TWiN Cloud",
31
31
  "protocol": "arcane-ai-browser-speech-configuration/1, AIProviderRuntime arcane-ai-provider/2 routes, globalThis.arcaneEvents, TWiN Cloud HTTPS, Arcane.ollama, Arcane.speech, Android WebView bridge",
32
- "normalization": "Complete mutable caller-owned browser speech authority, SDK-owned provider registration/replacement/disposal, explicit STT/TTS activation, normalized Kokoro TTS execution with default capacity 4, explicit provider-neutral execution snapshots through status(role,{execution:true}), sticky readiness, canonical window.user readiness without compatibility-event object-identity admission or polling, exact ordered structural calls with required arguments.message, atomic nonblank all-ID tool-result sequencing, complete all-choice streaming/data validation, native Ollama structural adaptation, complete-response then all validated tool callbacks then completion ordering, observed async callbacks, Blob/File transcription, immediate exact-segment synthesis, per-call voice/speed capture, optional final-segment pause on the existing audio clock, optional per-call terminal playback completion, ordered audio-clock scheduling, playable audio, active-generation TTS operation-failure routing, mute, and cancellation are normalized; provider/model/runtime/voice policy remains caller-owned.",
33
- "surface": "Browser-speech protocol/event/error/reason constants; default `AI`; read-only `providerRuntime`, `browserSpeechConfiguration`, and `browserSpeechDescriptor`; explicit `providerRuntime.status(role,{execution:true})` snapshots; `streamTTS(text,end,options={})` with optional voice, speed, pauseAfterMs, waitForPlayback, and speech-only textFormat plain/markdown; configure/dispose speech, route lifecycle, declaration-validated chat/stream requests with exact ordered calls, synthesis/transcription, and playback controls; initializes from current canonical user readiness, installs `window.ai`, and projects `ai-ready` plus active-generation `ai-tts-failure`."
32
+ "normalization": "Complete mutable caller-owned browser speech authority, SDK-owned provider registration/replacement/disposal, explicit STT/TTS activation, normalized Kokoro TTS execution with default capacity 4, explicit provider-neutral execution snapshots through status(role,{execution:true}), sticky readiness, canonical window.user readiness without compatibility-event object-identity admission or polling, exact ordered structural calls with required arguments.message, atomic nonblank all-ID tool-result sequencing, complete all-choice streaming/data validation, native Ollama structural adaptation, complete-response then all validated tool callbacks then completion ordering, observed async callbacks, Blob/File transcription, automatic repeated-formatting-mark removal from cloned outbound TTS input while caller and visible content stays exact, immediate exact-segment synthesis, per-call voice/speed capture, optional final-segment pause on the existing audio clock, optional per-call terminal playback completion, ordered audio-clock scheduling, playable audio, active-generation TTS operation-failure routing, mute, and cancellation are normalized; provider/model/runtime/voice policy remains caller-owned.",
33
+ "surface": "Browser-speech protocol/event/error/reason constants; default `AI`; read-only `providerRuntime`, `browserSpeechConfiguration`, and `browserSpeechDescriptor`; explicit `providerRuntime.status(role,{execution:true})` snapshots; `streamTTS(text,end,options={})` with optional voice, speed, pauseAfterMs, and waitForPlayback; automatic speech-input cleanup with SDK-internal preparation metadata; configure/dispose speech, route lifecycle, declaration-validated chat/stream requests with exact ordered calls, synthesis/transcription, and playback controls; initializes from current canonical user readiness, installs `window.ai`, and projects `ai-ready` plus active-generation `ai-tts-failure`."
34
34
  },
35
35
  {
36
36
  "file": "runtime/arcane/modules/AIPreferenceRuntime.js",
@@ -76,8 +76,8 @@
76
76
  "summary": "Provider-neutral selection, lifecycle, routing, startup, request, streaming, cancellation, and independent LLM/STT/TTS state.",
77
77
  "availability": "Cross-host in-process runtime; registered providers remain browser, native, or cloud specific",
78
78
  "protocol": "arcane-ai-runtime/2, arcane-ai-provider/2, arcane-ai-model-authority/1",
79
- "normalization": "Normalizes mutable complete per-role routes, lifecycle/status, cancellation, streaming cleanup, capacity-1 FIFO LLM/STT lanes, bounded provider-declared parallel TTS requests with FIFO overflow, local-only selection, direct LLM history/declaration/terminal structural contracts, complete all-choice content/reasoning FIFO iteration, private automatic provider draining with result-first buffering, terminal-only structural calls, exact per-choice observed/terminal correlation, and a separate complete validated terminal result without creating a fallback.",
80
- "surface": "Protocol constants; singleton-only `AIProviderRuntime`; `aiProviderRuntime`; `getAIProviderRuntime()`; provider registration/identity/selection, closed three-role and STT/TTS-only validation/configuration/replacement, status/catalog/inspection, startup, per-role load/unload/dispose/cancel, dispose-all, request/stream/speech aliases, and mute controls."
79
+ "normalization": "Normalizes mutable complete per-role routes, lifecycle/status, cancellation, streaming cleanup, capacity-1 FIFO LLM/STT lanes, bounded provider-declared parallel TTS requests with FIFO overflow, automatic repeated-formatting-mark removal from cloned direct TTS payloads, local-only selection, direct LLM history/declaration/terminal structural contracts, complete all-choice content/reasoning FIFO iteration, private automatic provider draining with result-first buffering, terminal-only structural calls, exact per-choice observed/terminal correlation, and a separate complete validated terminal result without creating a fallback.",
80
+ "surface": "Protocol constants; singleton-only `AIProviderRuntime`; `aiProviderRuntime`; `getAIProviderRuntime()`; provider registration/identity/selection, closed three-role and STT/TTS-only validation/configuration/replacement, status/catalog/inspection, startup, per-role load/unload/dispose/cancel, dispose-all, request/stream/speech aliases, SDK-internal speech-input preparation metadata, and mute controls."
81
81
  },
82
82
  {
83
83
  "file": "runtime/arcane/modules/AIResponseLength.js",
@@ -831,10 +831,10 @@
831
831
  "exports": [
832
832
  "MarkdownSpeech"
833
833
  ],
834
- "summary": "Removes repeated same Markdown formatting marks from streamed narration before speech segmentation.",
834
+ "summary": "Re-exports the shared MarkdownSpeech class that removes repeated same formatting marks from streamed narration before speech segmentation.",
835
835
  "availability": "Cross-host",
836
- "protocol": "In-process only",
837
- "normalization": "Speech-only filtering of repeated *, #, _, backtick, and ~ runs across chunks; single marks and all other characters remain literal; terminal flush and reset clear pending formatting state.",
836
+ "protocol": "Shared arcane-os/speech-text package entrypoint",
837
+ "normalization": "The runtime projection and public package subpath share dependency-free speech-only filtering of repeated *, #, _, backtick, and ~ runs across chunks; single marks and all other characters remain literal; terminal flush and reset clear pending formatting state.",
838
838
  "surface": "`MarkdownSpeech`; `append(text='',end=false)`, `reset()`."
839
839
  },
840
840
  {
@@ -1162,15 +1162,17 @@
1162
1162
  "kind": "esm",
1163
1163
  "exports": [
1164
1164
  "SPEECH_PLAYBACK_STATE_EVENT",
1165
+ "SPEECH_VOICE_ALIASES",
1166
+ "SPEECH_VOICE_OPTIONS",
1165
1167
  "SpeechPlayback",
1166
1168
  "default",
1167
1169
  "splitSpeechText"
1168
1170
  ],
1169
- "summary": "Preserves exact nonblank segment text without trimming, splitting, or freezing it; eagerly submits complete parts to a capacity-advertising fetchTTS provider while keeping playback indexed, and retains serialized one-segment lookahead for other clients.",
1171
+ "summary": "Preserves exact nonblank stored segment text; removes repeated formatting marks from only each outbound synthesis copy; eagerly submits complete parts to a capacity-advertising fetchTTS provider while keeping playback indexed; and retains serialized one-segment lookahead for other clients.",
1170
1172
  "availability": "Browser + compatible AI/native bridge",
1171
1173
  "protocol": "AI.fetchTTS plus providerRuntime TTS execution capacity, or compatible serialized Arcane.speech.synthesize; globalThis.arcaneEvents, Blob URLs, audio element",
1172
- "normalization": "Exact nonblank split/part input strings are preserved in mutable records. A capable provider receives complete segments immediately and owns bounded FIFO admission; URLs and playback stay in exact input order. Native/custom clients remain serialized. Cancellation, Replay recovery, Blob results, and lifecycle state are normalized; provider failures remain external.",
1173
- "surface": "SpeechPlayback class/default, `SPEECH_PLAYBACK_STATE_EVENT`, `splitSpeechText()`, canonical playback lifecycle events, cancellation, and destroy APIs."
1174
+ "normalization": "Exact nonblank split/part input strings are preserved in mutable records. Each synthesis request clones that part, removes repeated formatting marks from only the outbound input, and carries SDK-internal preparation metadata to prevent a second cleanup pass. A capable provider receives complete segments immediately and owns bounded FIFO admission; URLs and playback stay in exact input order. Native/custom clients remain serialized. Canonical state dispatch precedes the optional synchronous onState callback. Cancellation, Replay recovery, Blob results, and lifecycle state are normalized; provider failures remain external.",
1175
+ "surface": "SpeechPlayback class/default, mutable `SPEECH_VOICE_OPTIONS` records, mutable `SPEECH_VOICE_ALIASES` membership, `SPEECH_PLAYBACK_STATE_EVENT`, optional constructor `onState(detail)`, `splitSpeechText()`, canonical playback lifecycle events, cancellation, and destroy APIs."
1174
1176
  },
1175
1177
  {
1176
1178
  "file": "runtime/arcane/modules/StaticDocumentCatalog.js",
@@ -61,11 +61,13 @@ explicit identity fails, and the selected page then must pass exact path-relativ
61
61
  base validation.
62
62
  Included HTML with neither signal is a component fragment and remains a
63
63
  complete package file.
64
- `arcane-os/preference-store` and
65
- `arcane-os/speech-playback` are the two portable runtime subpaths: Node package
66
- exports and managed browser keys both resolve directly to the canonical runtime
67
- module namespaces, while `arcane/PreferenceStore` and
68
- `arcane/SpeechPlayback` are the established browser import-map names. There is
64
+ `arcane-os/preference-store` and `arcane-os/speech-playback` are the two
65
+ portable subpaths that resolve directly to canonical runtime-module namespaces
66
+ from both Node package exports and managed browser keys;
67
+ `arcane/PreferenceStore` and `arcane/SpeechPlayback` remain their established
68
+ browser import-map names. The additional portable `arcane-os/speech-text`
69
+ subpath owns shared dependency-free speech-input cleanup and has the same
70
+ managed browser key. There is
69
71
  no exported `importMapApplication()` function, `generateImportMap()` function, or
70
72
  `arcane-os/import-map` package subpath.
71
73
 
@@ -159,7 +161,7 @@ imports such as:
159
161
  import ollama from 'arcane/Ollama';
160
162
  ```
161
163
 
162
- The physical-v1 tree lives entirely beneath `arcane/`. SDK `0.5.16` projects the
164
+ The physical-v1 tree lives entirely beneath `arcane/`. SDK `0.5.17` projects the
163
165
  complete canonical runtime and browser runtime selected by the installed SDK
164
166
  package. Runtime dependencies stay under
165
167
  `arcane/dependencies/`; the SDK event and browser-AI closure stays under
@@ -174,13 +176,13 @@ integrated physical route uses the same ordered include list in its one route.
174
176
  Omitting only that final optional entry is compatible. Removing, reordering, or
175
177
  renaming any preceding entry changes the physical contract.
176
178
 
177
- The `0.5.16` map derives its complete entries from the selected runtime graph;
179
+ The `0.5.17` map derives its complete entries from the selected runtime graph;
178
180
  application source imports do not select a fixed entry count. The operation
179
181
  result reports `imports`, `entryCount`, and `excludedModules`; reached-file
180
182
  traversal remains internal. The
181
183
  managed graph exposes `arcane-os/event-manager`, `arcane-os/ai/browser-wasm`,
182
184
  `arcane-os/ai/browser-speech`, `arcane-os/preference-store`, and
183
- `arcane-os/speech-playback`; dependency compatibility mappings are added when
185
+ `arcane-os/speech-playback` plus `arcane-os/speech-text`; dependency compatibility mappings are added when
184
186
  runtime or SDK root traversal observes them.
185
187
 
186
188
  The focused physical targets remain stable when their bindings are reached:
@@ -190,6 +192,7 @@ The focused physical targets remain stable when their bindings are reached:
190
192
  | `arcane-os/event-manager` | `./arcane/sdk/event-manager.mjs` |
191
193
  | `arcane-os/ai/browser-wasm` | `./arcane/sdk/ai/browser-wasm.mjs` |
192
194
  | `arcane-os/ai/browser-speech` | `./arcane/sdk/ai/browser-speech.mjs` |
195
+ | `arcane-os/speech-text` | `./arcane/sdk/speech-text.mjs` |
193
196
  | `arcane-os/preference-store` | `./arcane/modules/PreferenceStore.js` |
194
197
  | `arcane-os/speech-playback` | `./arcane/modules/SpeechPlayback.js` |
195
198
  | `event-pubsub` | `./arcane/sdk/dependencies/event-pubsub/index.js` |
@@ -202,8 +205,9 @@ invented package bindings. Development serves the selected app plus the
202
205
  complete selected tree.
203
206
  Packaging copies the same map, app entry, and physical content into `dist/<id>`;
204
207
  targets never resolve through the consumer workspace's root `node_modules/`.
205
- The two lowercase static runtime package specifiers above are also exact npm
206
- package exports, so Node and managed-browser source can share those specifiers
208
+ The two lowercase static runtime package specifiers above and the shared
209
+ speech-text helper are exact npm package exports, so Node and managed-browser
210
+ source can share those specifiers
207
211
  without data URLs, copied modules, or a consumer-owned loader.
208
212
 
209
213
  `generateImportMap()` is an internal toolchain operation, not a package export.
@@ -239,8 +243,8 @@ exactly `dependencyName`, `packageSource`,
239
243
  `canonicalPackageRoot`, `packageName`, `packageVersion`, `runtimeRoot`,
240
244
  and `browserRuntimeRoot`. A
241
245
  workspace may use the canonical dependency name or one exact npm alias such as
242
- `npm:arcane-os@0.5.16`; the physical package manifest must still identify
243
- exactly as `arcane-os@0.5.16`. Canonical-plus-alias duplicates, multiple aliases,
246
+ `npm:arcane-os@0.5.17`; the physical package manifest must still identify
247
+ exactly as `arcane-os@0.5.17`. Canonical-plus-alias duplicates, multiple aliases,
244
248
  links/junctions, indirect package roots, or version drift are reported.
245
249
 
246
250
  For external workspaces, `arcane dev` serves the projected `arcane/` root,
@@ -351,7 +355,7 @@ does not silently delete the app-owned cache. A complete whole member supersedes
351
355
  its current resumable fragments. Cleanup failure is warned without hiding the
352
356
  usable model.
353
357
 
354
- SDK `0.5.16` requires WebGPU. Load requests full offload and waits for the runtime
358
+ SDK `0.5.17` requires WebGPU. Load requests full offload and waits for the runtime
355
359
  to report a loaded model. `navigator.gpu` presence alone is
356
360
  not readiness. There is no CPU fallback, partial-offload success mode, or
357
361
  silent switch to native/Core/cloud inference.
@@ -366,14 +366,15 @@ terminal result of the segments submitted by one invocation; Chat continues
366
366
  to feed chunks without waiting for playback. Runtime mute, cancellation,
367
367
  permission waiting, and stale generations remain non-errors.
368
368
 
369
- Each visible model chunk is forwarded to
370
- `AI.streamTTS(text,false,{textFormat:'markdown'})` in arrival order. The AI
371
- owner removes repeated Markdown formatting marks from narration before the
369
+ Each visible model chunk is forwarded to `AI.streamTTS(text)` in arrival
370
+ order. The AI owner automatically removes repeated formatting marks from an
371
+ outbound speech-input copy before the
372
372
  configured segmentation, admits completed segments to bounded synthesis
373
373
  immediately, and schedules the contiguous ready audio prefix on the audio
374
374
  clock. Chat retains the original Markdown for display and storage and owns no
375
- second speech queue. Single punctuation, ordinary repeated punctuation,
376
- language, and voice retain their existing behavior.
375
+ second speech queue. No app option selects or disables this cleanup. Single
376
+ punctuation, ordinary repeated punctuation, language, and voice retain their
377
+ existing behavior.
377
378
 
378
379
  Events: `chat-ready`, `chat-session-bound`, `chat-session-message`,
379
380
  `chat-session-error`, `chat-send-message`, `chat-send-error`,
@@ -109,7 +109,7 @@ own asynchronous work, cancellation, and backpressure.
109
109
  | [`ScamRiskPolicy.js`](#scamriskpolicyjs) | esm | Combines deterministic scam signals with optional Arcane blocked-domain evidence and safety guidance. | Cross-host | Complete mutable results; blocked-domain policy requires `secure:true`. |
110
110
  | [`ScopedOPFSCache.js`](#scopedopfscachejs) | esm | Provides a narrow exact-key JSON cache inside one app-owned OPFS namespace. | Browser / native WebView | Filename-safe keys, complete JSON values, and malformed-cache cleanup normalized; storage errors mixed. |
111
111
  | [`ScreenCapture.js`](#screencapturejs) | esm | Captures a display surface as image, video, or GIF with explicit lifecycle events. | Browser / native WebView | State/events normalized; permission and codec errors mixed. |
112
- | [`SpeechPlayback.js`](#speechplaybackjs) | esm | Preserves exact nonblank text as one speech segment, queues latest-request synthesis, and controls lookahead HTML audio playback. | Browser + native bridge | Exact input text and state normalized; provider/media failures mixed. |
112
+ | [`SpeechPlayback.js`](#speechplaybackjs) | esm | Admits complete speech segments to a capacity-advertising provider immediately, retains serialized native/custom lookahead, and plays every result in exact indexed order. | Browser + native bridge | Stored part text stays exact; the outbound speech-input copy receives automatic formatting-mark cleanup; provider/media failures remain mixed. |
113
113
  | [`StaticDocumentCatalog.js`](#staticdocumentcatalogjs) | esm | Loads a positive static document inventory with cache, search, and complete context. | Browser / native WebView / server with fetch | Mutable complete catalog/content normalization; malformed data and transport failures remain visible. |
114
114
  | [`SystemAppearance.js`](#systemappearancejs) | esm | Reads or applies native appearance, returning an explicit unsupported browser state when no bridge exists. | Browser/native hybrid | Absent bridge normalized; native result/error preserved. |
115
115
  | [`SystemPlatformPresentation.js`](#systemplatformpresentationjs) | classic-script | Maps kernel names to presentation labels/classes without granting platform authority. | Browser / native WebView classic script | Fully normalized presentation only. |
@@ -263,7 +263,7 @@ reports `requestedDevice`, `selectedDevice`, `maxConcurrentRequests`, and
263
263
  WASM fallback after a successful load. Calling `status()` without options keeps
264
264
  the existing sticky lifecycle snapshot and does not inspect provider execution.
265
265
  Provider inspection failures are surfaced to the caller.
266
- `fetchTTS({model,voice,input,responseFormat,speed},signal)` accepts the public
266
+ `fetchTTS({model,voice,input,responseFormat,speed},signal,preparation={})` accepts the public
267
267
  provider-neutral synthesis shape, requires any explicit model to match the
268
268
  selected route, and fills an omitted voice only from the selected model
269
269
  catalog's `defaultVoice`. An omitted response format preserves the instance's
@@ -273,9 +273,14 @@ the setting is the instance's `opus` default and the model rejects it, the catal
273
273
  `speech.defaultResponseFormat` is used, while any other unsupported setting is
274
274
  rejected. It propagates the caller-owned signal and returns a playable `Blob`;
275
275
  it does not independently choose a provider, cloud fallback, model, runtime, or
276
- voice policy for the application. `streamTTS(text='',end=false,options={})` and
276
+ voice policy for the application. Every call removes repeated same formatting
277
+ marks from a cloned outbound input before delegation; the caller's payload stays
278
+ unchanged. The third `preparation` argument is reserved for SDK-internal
279
+ delegation, where `{speechInputPrepared:true}` prevents a second cleanup pass;
280
+ applications omit it. `streamTTS(text='',end=false,options={})` and
277
281
  `finishTTS()` use this same request boundary. The third-argument options below
278
- are available in SDK `0.5.12`, with `textFormat` added in SDK `0.5.16`:
282
+ are available in SDK `0.5.12`. The `textFormat` compatibility extra added in
283
+ SDK `0.5.16` is ignored beginning in `0.5.17`:
279
284
 
280
285
  | Field | Default | Meaning |
281
286
  | --- | --- | --- |
@@ -283,24 +288,23 @@ are available in SDK `0.5.12`, with `textFormat` added in SDK `0.5.16`:
283
288
  | `speed` | Current `ai.voiceSpeed` | A supplied positive speed is captured for those segments and forwarded to `fetchTTS()`. It does not change `ai.voiceSpeed`. |
284
289
  | `pauseAfterMs` | `0` | Finite, nonnegative milliseconds placed after the final extracted segment on the existing audio clock. Invalid values throw `RangeError`; no pause is inserted between this call's other segments. |
285
290
  | `waitForPlayback` | `false` | Omission retains the preparation promise. With `true`, the promise resolves after every extracted segment reaches a terminal playback state: `true` when all naturally end, or `false` after terminal cancellation or failure. |
286
- | `textFormat` | `plain` for a new stream | `markdown` removes repeated same formatting marks (`*`, `#`, `_`, backtick, `~`) before segmentation. Single marks and ordinary punctuation remain literal. The selected format lasts through this producer's terminal flush. |
291
+ | `textFormat` | Ignored compatibility extra | Repeated same formatting marks are removed automatically from every TTS call. This value no longer selects or disables cleanup. |
287
292
 
288
293
  The voice and speed use the existing `fetchTTS()` validation and error path.
289
- Plain mode preserves the submitted text. Markdown mode changes narration only;
290
- it leaves displayed, stored, and model content untouched. It is a narrow
294
+ The automatic cleanup changes only the outbound speech-input copy; displayed,
295
+ stored, model, and caller-owned content remains exact. It is a narrow
291
296
  formatting-mark filter, not a full Markdown parser: links, code contents, list
292
- text, and other characters remain literal. A trailing candidate mark waits for
293
- the next character so a repeated run split across chunks is still omitted.
294
- Ordinary prose streams immediately. `end:true`, `finishTTS()`, muted terminal
295
- calls, and `stopAudio()` clear the formatting state. An explicit format change
296
- flushes pending formatting marks under the preceding mode before accepting
297
- the next text. `fetchTTS()` retains its exact-text contract.
297
+ text, single marks, ordinary punctuation, and other characters remain literal.
298
+ A trailing candidate mark waits for the next character so a repeated run split
299
+ across chunks is still omitted. Ordinary prose streams immediately.
300
+ `end:true`, `finishTTS()`, muted terminal calls, and `stopAudio()` clear pending
301
+ formatting state. `textFormat` is not an opt-out.
298
302
 
299
303
  Voice, speed, pause, and playback overrides belong to the segments
300
304
  extracted in that invocation, including any text buffered by an earlier call.
301
305
  Those overrides are not retained with an unfinished `end:false` remainder; a
302
306
  later call supplies its own options, and `finishTTS()` uses their defaults
303
- while retaining the active text format. Use `end:true`
307
+ while flushing any pending formatting mark. Use `end:true`
304
308
  for a complete passage. A call extracting no segments resolves
305
309
  `true` without waiting for earlier jobs; `finishTTS()` remains a preparation
306
310
  flush, not a queue-wide playback barrier. A muted call resolves `false`.
@@ -324,8 +328,9 @@ commas, and hyphens remain inside a segment when they join Unicode letters or
324
328
  numbers. A potentially joining mark at the current end of an incremental stream
325
329
  waits for the next character or terminal flush before the boundary is decided;
326
330
  `wordCadence` completes one after that many whole words. The earliest available
327
- boundary wins. Segmentation preserves every character, including punctuation
328
- and whitespace. Every completed segment enters synthesis immediately; provider
331
+ boundary wins. Segmentation preserves every character of the already prepared
332
+ speech text, including punctuation and whitespace. Every completed segment
333
+ enters synthesis immediately; provider
329
334
  capacity supplies FIFO backpressure while allowing bounded TTS work to overlap.
330
335
  A later segment may finish synthesis first, but playback schedules only the
331
336
  contiguous ready prefix in original order. Decoded buffers with known duration
@@ -662,9 +667,11 @@ The singleton exposes read-only `protocol`, `configured`, and `speechMuted`;
662
667
  `status(role=null,options={})`; `catalog(role)`;
663
668
  `inspect(role,options={})`; `start(options)`; `load(role,options={})`;
664
669
  `unload(role,options={})`; `dispose(role,options={})`;
665
- `disposeAll(options={})`; `cancel(role)`; `request(role,options={})`;
670
+ `disposeAll(options={})`; `cancel(role)`;
671
+ `request(role,options={},preparation={})`;
666
672
  `chat(payload,options={})`; `stream(payload,options={})`;
667
- `transcribe(payload,options={})`; `synthesize(payload,options={})`; and
673
+ `transcribe(payload,options={})`;
674
+ `synthesize(payload,options={},preparation={})`; and
668
675
  `setSpeechMuted(muted)`. Provider payloads must be data-only; callbacks,
669
676
  accessors, symbols, and cycles are rejected at the provider boundary.
670
677
 
@@ -680,6 +687,14 @@ records are the closed `{llm,stt,tts}`, `{stt,tts}`, or
680
687
  `{providers,routes,expectedProviders}` shapes described below.
681
688
  `configureFromTuple()` accepts exactly six provider/model preference entries.
682
689
 
690
+ Every direct TTS `request()` or `synthesize()` call removes repeated same
691
+ formatting marks from a cloned outbound payload's `input` or `text` field. The
692
+ caller's payload and request records remain unchanged. The optional
693
+ `preparation` argument is reserved for SDK-owned delegation;
694
+ `{speechInputPrepared:true}` prevents a second pass after another SDK speech
695
+ boundary has already cleaned the copy. Applications omit that argument. LLM
696
+ and STT payloads are unaffected.
697
+
683
698
  `register()` returns the provider's single unregister closure; caller-
684
699
  registered providers remain caller-owned. The high-level
685
700
  `AI.configureBrowserSpeech()` boundary is different: AI constructs, registers,
@@ -2261,11 +2276,12 @@ console.log(Object.keys(module));
2261
2276
 
2262
2277
  ### Overview
2263
2278
 
2264
- Filters repeated same Markdown formatting marks from narration before speech
2265
- segmentation. The filter recognizes `*`, `#`, `_`, backtick, and `~` runs of two
2266
- or more, including runs arriving across separate chunks. Single marks,
2267
- ellipses, quoted sentence endings, whitespace, and all other text remain
2268
- literal. It neither interprets links nor changes language or voice.
2279
+ Re-exports the streaming `MarkdownSpeech` filter from the shared
2280
+ `arcane-os/speech-text` package entrypoint. The filter removes repeated runs of
2281
+ `*`, `#`, `_`, backtick, and `~` before speech segmentation, including runs
2282
+ arriving across separate chunks. Single marks, ellipses, quoted sentence
2283
+ endings, whitespace, and all other text remain literal. It neither interprets
2284
+ links nor changes language or voice.
2269
2285
 
2270
2286
  ### Public surface
2271
2287
 
@@ -2275,7 +2291,8 @@ Exact exports: `MarkdownSpeech`.
2275
2291
 
2276
2292
  ### Availability and normalization
2277
2293
 
2278
- **Cross-host.** Plain JavaScript with no dependencies. `append()` returns only
2294
+ **Cross-host.** The runtime projection and public package entrypoint share the
2295
+ same dependency-free implementation. `append()` returns only
2279
2296
  the newly available narration. Only a trailing candidate marker and whether
2280
2297
  it repeats are retained; ordinary text is emitted immediately. Terminal
2281
2298
  `append(text,true)` flushes a single pending mark and resets state. `reset()`
@@ -2899,15 +2916,17 @@ console.log(Object.keys(module));
2899
2916
 
2900
2917
  Preserves exact nonblank text, admits complete speech segments according to the
2901
2918
  selected client's advertised capacity, and plays indexed HTML audio in exact
2902
- input order.
2919
+ input order. Stored parts remain exact; only each outbound synthesis payload
2920
+ copy receives automatic formatting-mark cleanup.
2903
2921
 
2904
2922
  ### Public surface
2905
2923
 
2906
- `SpeechPlayback` class/default, `SPEECH_PLAYBACK_STATE_EVENT`,
2907
- `splitSpeechText()`, and playback lifecycle APIs.
2924
+ `SpeechPlayback` class/default, the shared voice compatibility catalogs,
2925
+ `SPEECH_PLAYBACK_STATE_EVENT`, `splitSpeechText()`, the optional constructor
2926
+ state callback, and playback lifecycle APIs.
2908
2927
 
2909
- Exact exports: `SPEECH_PLAYBACK_STATE_EVENT`, `SpeechPlayback`, `default`, and
2910
- `splitSpeechText`.
2928
+ Exact exports: `SPEECH_PLAYBACK_STATE_EVENT`, `SPEECH_VOICE_ALIASES`,
2929
+ `SPEECH_VOICE_OPTIONS`, `SpeechPlayback`, `default`, and `splitSpeechText`.
2911
2930
 
2912
2931
  ```text
2913
2932
  new SpeechPlayback({
@@ -2917,6 +2936,7 @@ new SpeechPlayback({
2917
2936
  voice=null,
2918
2937
  responseFormat=null,
2919
2938
  speed=1,
2939
+ onState=()=>{},
2920
2940
  createObjectURL,
2921
2941
  revokeObjectURL,
2922
2942
  delay,
@@ -2925,17 +2945,28 @@ new SpeechPlayback({
2925
2945
  ```
2926
2946
 
2927
2947
  `speech` must expose either `fetchTTS(payload, signal)` or
2928
- `synthesize(payload, {signal})`. `prepare({key,parts,model,voice,responseFormat,
2948
+ `synthesize(payload, {signal})`. `SpeechPlayback` also supplies a third
2949
+ SDK-internal preparation object; existing two-argument clients may ignore it.
2950
+ `prepare({key,parts,model,voice,responseFormat,
2929
2951
  speed,autoplay=true})` uses only caller-supplied model, voice, and response-format
2930
2952
  values; those three omitted values remain omitted so the selected AI/model
2931
2953
  catalog may provide its documented defaults. Speed defaults to `1`, is normalized
2932
- as a positive number, and is always sent. There is no
2933
- hard-coded model, response format, voice, or cloud/browser fallback.
2954
+ as a positive number, and is always sent. `SPEECH_VOICE_OPTIONS` is the ordered
2955
+ mutable compatibility array `{value,label}` for `alloy`, `ash`, `ballad`,
2956
+ `coral`, `echo`, `fable`, `nova`, `onyx`, `sage`, and `shimmer`;
2957
+ `SPEECH_VOICE_ALIASES` is the mutable `Set` of those values. The class does not
2958
+ select either catalog or promise that a selected provider supports its values.
2959
+ There is no hard-coded model, response format, voice, or cloud/browser fallback.
2934
2960
  `splitSpeechText(value)` uses trimming only to detect blank input, then returns
2935
2961
  the caller's exact string in one mutable array without trimming, splitting, or
2936
2962
  freezing it. `prepare()` likewise preserves each nonblank part's exact `input`
2937
2963
  string while normalizing its other playback fields into a new mutable record.
2938
- The class applies no part-count, character-count, pause, or input upper cap.
2964
+ At synthesis time, `requestSpeech()` copies that record, removes repeated same
2965
+ formatting marks from only the outbound `input`, and delegates with the
2966
+ SDK-internal `{speechInputPrepared:true}` argument so downstream SDK boundaries
2967
+ do not apply the non-idempotent filter again. Original part objects, stored
2968
+ parts, displayed text, and all non-input payload fields remain unchanged. The
2969
+ class applies no part-count, character-count, pause, or input upper cap.
2939
2970
 
2940
2971
  ### Admission and playback order
2941
2972
 
@@ -2961,8 +2992,13 @@ Every preparation owns an operation ID and one AbortController for each active
2961
2992
  synthesis segment or playback delay. Replacement,
2962
2993
  `stop()`, `cancel()`, and `destroy()` abort their owned signals, suppress stale
2963
2994
  settlement, release Blob URLs, and publish synchronous
2964
- `speech-playback-state` occurrences through `globalThis.arcaneEvents`.
2965
- Subscribers receive mutable public state detail. The detail contains
2995
+ `speech-playback-state` occurrences through `globalThis.arcaneEvents` before
2996
+ calling the optional `onState(detail)` function synchronously. Canonical
2997
+ subscribers and the callback observe the same public field values at dispatch
2998
+ time, but object identity is not promised. Both surfaces expose mutable public
2999
+ state detail. A callback failure is reported through `globalThis.reportError`
3000
+ when available, otherwise `console.error`, and does not replace playback
3001
+ settlement. The detail contains
2966
3002
  `state`, `message`, `key`, `index`, `total`, `producing`, `buffered`, `hasAudio`,
2967
3003
  `operationId`, `code`, and `reason`; a first-segment provider rejection remains
2968
3004
  preserved to the `prepare()` caller. Later failures surface when ordered
@@ -3004,7 +3040,10 @@ const audio = document.body.appendChild(document.createElement('audio'));
3004
3040
  audio.controls = true;
3005
3041
  const speech = new SpeechPlayback({
3006
3042
  audio,
3007
- speech: globalThis.ai
3043
+ speech: globalThis.ai,
3044
+ onState(detail) {
3045
+ console.log('Speech state:', detail.state);
3046
+ }
3008
3047
  });
3009
3048
  const button = document.body.appendChild(document.createElement('button'));
3010
3049
  button.textContent = 'Speak';
@@ -3020,6 +3059,19 @@ button.addEventListener('click', async function speakCompleteSegments() {
3020
3059
  });
3021
3060
  ```
3022
3061
 
3062
+ The shared compatibility catalogs are also available directly. They do not
3063
+ select a voice for `SpeechPlayback`:
3064
+
3065
+ ```javascript
3066
+ import {
3067
+ SPEECH_VOICE_ALIASES,
3068
+ SPEECH_VOICE_OPTIONS
3069
+ } from '/arcane/modules/SpeechPlayback.js';
3070
+
3071
+ console.log(SPEECH_VOICE_OPTIONS[0]); // {value: 'alloy', label: 'Alloy'}
3072
+ console.log(SPEECH_VOICE_ALIASES.has('alloy')); // true
3073
+ ```
3074
+
3023
3075
  With the default browser speech configuration, both parts enter its capacity-4
3024
3076
  queue immediately and still play first, then second. Inspect the selected
3025
3077
  execution device without guessing from console warnings: