arcane-os 0.5.15 → 0.5.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,7 @@
5
5
  "repository": "https://github.com/TheWizardNexus/arcane-os-sdk.git",
6
6
  "branch": "main",
7
7
  "path": "runtime/arcane/components",
8
- "sdkVersion": "0.5.15"
8
+ "sdkVersion": "0.5.17"
9
9
  },
10
10
  "componentCount": 39,
11
11
  "loader": "/arcane/modules/HTMLImport.js",
@@ -171,7 +171,7 @@
171
171
  ],
172
172
  "availability": "Browser and supported native WebViews",
173
173
  "transport": "HTMLImport + DOM; injected Arcane/provider modules where listed",
174
- "normalization": "UI/runtime state, honest model progress, complete timestamped transcript including ui_hidden stored records, user-facing structural arguments.message, nonblank all-ID tool-result restoration, collapsed complete raw inspection, ordered parallel-call replay validation, per-choice streamed/terminal complete-envelope correlation, complete nonstructural stream rendering, generic visible failure outcomes with complete console diagnostics, atomic executed/declined/cancelled/not-executed result batches with one continuation, plural pending-call reload recovery, reentrant terminal-event ownership, and BFCache-preserving page lifecycle are normalized; destroy aborts observation and returns true once/false thereafter; AI/storage/media behavior remains mixed"
174
+ "normalization": "UI/runtime state, honest model progress, complete timestamped transcript including ui_hidden stored records, user-facing structural arguments.message, nonblank all-ID tool-result restoration, collapsed complete raw inspection, ordered parallel-call replay validation, per-choice streamed/terminal complete-envelope correlation, complete nonstructural stream rendering, automatic repeated-formatting-mark removal from only the outbound speech-input copy while displayed and stored Markdown stays exact, generic visible failure outcomes with complete console diagnostics, atomic executed/declined/cancelled/not-executed result batches with one continuation, plural pending-call reload recovery, reentrant terminal-event ownership, and BFCache-preserving page lifecycle are normalized; destroy aborts observation and returns true once/false thereafter; AI/storage/media behavior remains mixed"
175
175
  },
176
176
  {
177
177
  "file": "runtime/arcane/components/conversation-view.html",
@@ -5,12 +5,12 @@
5
5
  "repository": "https://github.com/TheWizardNexus/arcane-os-sdk.git",
6
6
  "branch": "main",
7
7
  "path": "runtime/arcane",
8
- "sdkVersion": "0.5.15",
8
+ "sdkVersion": "0.5.17",
9
9
  "protocol": "arcane/1"
10
10
  },
11
- "artifactCount": 84,
12
- "javascriptArtifactCount": 82,
13
- "esmExportCount": 353,
11
+ "artifactCount": 85,
12
+ "javascriptArtifactCount": 83,
13
+ "esmExportCount": 356,
14
14
  "artifacts": [
15
15
  {
16
16
  "file": "runtime/arcane/modules/AI.js",
@@ -29,8 +29,8 @@
29
29
  "summary": "Owns provider-selectable chat and the one-time caller-authority browser STT/TTS configuration, lifecycle, synthesis, transcription, and playback boundary.",
30
30
  "availability": "Browser + native bridge + TWiN Cloud",
31
31
  "protocol": "arcane-ai-browser-speech-configuration/1, AIProviderRuntime arcane-ai-provider/2 routes, globalThis.arcaneEvents, TWiN Cloud HTTPS, Arcane.ollama, Arcane.speech, Android WebView bridge",
32
- "normalization": "Complete mutable caller-owned browser speech authority, SDK-owned provider registration/replacement/disposal, explicit STT/TTS activation, normalized Kokoro TTS execution with default capacity 4, explicit provider-neutral execution snapshots through status(role,{execution:true}), sticky readiness, canonical window.user readiness without compatibility-event object-identity admission or polling, exact ordered structural calls with required arguments.message, atomic nonblank all-ID tool-result sequencing, complete all-choice streaming/data validation, native Ollama structural adaptation, complete-response then all validated tool callbacks then completion ordering, observed async callbacks, Blob/File transcription, immediate exact-segment synthesis, per-call voice/speed capture, optional final-segment pause on the existing audio clock, optional per-call terminal playback completion, ordered audio-clock scheduling, playable audio, active-generation TTS operation-failure routing, mute, and cancellation are normalized; provider/model/runtime/voice policy remains caller-owned.",
33
- "surface": "Browser-speech protocol/event/error/reason constants; default `AI`; read-only `providerRuntime`, `browserSpeechConfiguration`, and `browserSpeechDescriptor`; explicit `providerRuntime.status(role,{execution:true})` snapshots; `streamTTS(text,end,options={})` with optional voice, speed, pauseAfterMs, and waitForPlayback (available in SDK 0.5.12); configure/dispose speech, route lifecycle, declaration-validated chat/stream requests with exact ordered calls, synthesis/transcription, and playback controls; initializes from current canonical user readiness, installs `window.ai`, and projects `ai-ready` plus active-generation `ai-tts-failure`."
32
+ "normalization": "Complete mutable caller-owned browser speech authority, SDK-owned provider registration/replacement/disposal, explicit STT/TTS activation, normalized Kokoro TTS execution with default capacity 4, explicit provider-neutral execution snapshots through status(role,{execution:true}), sticky readiness, canonical window.user readiness without compatibility-event object-identity admission or polling, exact ordered structural calls with required arguments.message, atomic nonblank all-ID tool-result sequencing, complete all-choice streaming/data validation, native Ollama structural adaptation, complete-response then all validated tool callbacks then completion ordering, observed async callbacks, Blob/File transcription, automatic repeated-formatting-mark removal from cloned outbound TTS input while caller and visible content stays exact, immediate exact-segment synthesis, per-call voice/speed capture, optional final-segment pause on the existing audio clock, optional per-call terminal playback completion, ordered audio-clock scheduling, playable audio, active-generation TTS operation-failure routing, mute, and cancellation are normalized; provider/model/runtime/voice policy remains caller-owned.",
33
+ "surface": "Browser-speech protocol/event/error/reason constants; default `AI`; read-only `providerRuntime`, `browserSpeechConfiguration`, and `browserSpeechDescriptor`; explicit `providerRuntime.status(role,{execution:true})` snapshots; `streamTTS(text,end,options={})` with optional voice, speed, pauseAfterMs, and waitForPlayback; automatic speech-input cleanup with SDK-internal preparation metadata; configure/dispose speech, route lifecycle, declaration-validated chat/stream requests with exact ordered calls, synthesis/transcription, and playback controls; initializes from current canonical user readiness, installs `window.ai`, and projects `ai-ready` plus active-generation `ai-tts-failure`."
34
34
  },
35
35
  {
36
36
  "file": "runtime/arcane/modules/AIPreferenceRuntime.js",
@@ -76,8 +76,8 @@
76
76
  "summary": "Provider-neutral selection, lifecycle, routing, startup, request, streaming, cancellation, and independent LLM/STT/TTS state.",
77
77
  "availability": "Cross-host in-process runtime; registered providers remain browser, native, or cloud specific",
78
78
  "protocol": "arcane-ai-runtime/2, arcane-ai-provider/2, arcane-ai-model-authority/1",
79
- "normalization": "Normalizes mutable complete per-role routes, lifecycle/status, cancellation, streaming cleanup, capacity-1 FIFO LLM/STT lanes, bounded provider-declared parallel TTS requests with FIFO overflow, local-only selection, direct LLM history/declaration/terminal structural contracts, complete all-choice content/reasoning FIFO iteration, private automatic provider draining with result-first buffering, terminal-only structural calls, exact per-choice observed/terminal correlation, and a separate complete validated terminal result without creating a fallback.",
80
- "surface": "Protocol constants; singleton-only `AIProviderRuntime`; `aiProviderRuntime`; `getAIProviderRuntime()`; provider registration/identity/selection, closed three-role and STT/TTS-only validation/configuration/replacement, status/catalog/inspection, startup, per-role load/unload/dispose/cancel, dispose-all, request/stream/speech aliases, and mute controls."
79
+ "normalization": "Normalizes mutable complete per-role routes, lifecycle/status, cancellation, streaming cleanup, capacity-1 FIFO LLM/STT lanes, bounded provider-declared parallel TTS requests with FIFO overflow, automatic repeated-formatting-mark removal from cloned direct TTS payloads, local-only selection, direct LLM history/declaration/terminal structural contracts, complete all-choice content/reasoning FIFO iteration, private automatic provider draining with result-first buffering, terminal-only structural calls, exact per-choice observed/terminal correlation, and a separate complete validated terminal result without creating a fallback.",
80
+ "surface": "Protocol constants; singleton-only `AIProviderRuntime`; `aiProviderRuntime`; `getAIProviderRuntime()`; provider registration/identity/selection, closed three-role and STT/TTS-only validation/configuration/replacement, status/catalog/inspection, startup, per-role load/unload/dispose/cancel, dispose-all, request/stream/speech aliases, SDK-internal speech-input preparation metadata, and mute controls."
81
81
  },
82
82
  {
83
83
  "file": "runtime/arcane/modules/AIResponseLength.js",
@@ -824,6 +824,19 @@
824
824
  "normalization": "Endpoint, timeout, cancellation, network, HTTP, and response-contract errors are normalized.",
825
825
  "surface": "`MailTransportError`, `normalizeMailEndpoint()`, `serializeMailReport()`, and `sendMailReport()`."
826
826
  },
827
+ {
828
+ "file": "runtime/arcane/modules/MarkdownSpeech.js",
829
+ "name": "MarkdownSpeech.js",
830
+ "kind": "esm",
831
+ "exports": [
832
+ "MarkdownSpeech"
833
+ ],
834
+ "summary": "Re-exports the shared MarkdownSpeech class that removes repeated same formatting marks from streamed narration before speech segmentation.",
835
+ "availability": "Cross-host",
836
+ "protocol": "Shared arcane-os/speech-text package entrypoint",
837
+ "normalization": "The runtime projection and public package subpath share dependency-free speech-only filtering of repeated *, #, _, backtick, and ~ runs across chunks; single marks and all other characters remain literal; terminal flush and reset clear pending formatting state.",
838
+ "surface": "`MarkdownSpeech`; `append(text='',end=false)`, `reset()`."
839
+ },
827
840
  {
828
841
  "file": "runtime/arcane/modules/Marked.min.js",
829
842
  "name": "Marked.min.js",
@@ -1149,15 +1162,17 @@
1149
1162
  "kind": "esm",
1150
1163
  "exports": [
1151
1164
  "SPEECH_PLAYBACK_STATE_EVENT",
1165
+ "SPEECH_VOICE_ALIASES",
1166
+ "SPEECH_VOICE_OPTIONS",
1152
1167
  "SpeechPlayback",
1153
1168
  "default",
1154
1169
  "splitSpeechText"
1155
1170
  ],
1156
- "summary": "Preserves exact nonblank segment text without trimming, splitting, or freezing it; eagerly submits complete parts to a capacity-advertising fetchTTS provider while keeping playback indexed, and retains serialized one-segment lookahead for other clients.",
1171
+ "summary": "Preserves exact nonblank stored segment text; removes repeated formatting marks from only each outbound synthesis copy; eagerly submits complete parts to a capacity-advertising fetchTTS provider while keeping playback indexed; and retains serialized one-segment lookahead for other clients.",
1157
1172
  "availability": "Browser + compatible AI/native bridge",
1158
1173
  "protocol": "AI.fetchTTS plus providerRuntime TTS execution capacity, or compatible serialized Arcane.speech.synthesize; globalThis.arcaneEvents, Blob URLs, audio element",
1159
- "normalization": "Exact nonblank split/part input strings are preserved in mutable records. A capable provider receives complete segments immediately and owns bounded FIFO admission; URLs and playback stay in exact input order. Native/custom clients remain serialized. Cancellation, Replay recovery, Blob results, and lifecycle state are normalized; provider failures remain external.",
1160
- "surface": "SpeechPlayback class/default, `SPEECH_PLAYBACK_STATE_EVENT`, `splitSpeechText()`, canonical playback lifecycle events, cancellation, and destroy APIs."
1174
+ "normalization": "Exact nonblank split/part input strings are preserved in mutable records. Each synthesis request clones that part, removes repeated formatting marks from only the outbound input, and carries SDK-internal preparation metadata to prevent a second cleanup pass. A capable provider receives complete segments immediately and owns bounded FIFO admission; URLs and playback stay in exact input order. Native/custom clients remain serialized. Canonical state dispatch precedes the optional synchronous onState callback. Cancellation, Replay recovery, Blob results, and lifecycle state are normalized; provider failures remain external.",
1175
+ "surface": "SpeechPlayback class/default, mutable `SPEECH_VOICE_OPTIONS` records, mutable `SPEECH_VOICE_ALIASES` membership, `SPEECH_PLAYBACK_STATE_EVENT`, optional constructor `onState(detail)`, `splitSpeechText()`, canonical playback lifecycle events, cancellation, and destroy APIs."
1161
1176
  },
1162
1177
  {
1163
1178
  "file": "runtime/arcane/modules/StaticDocumentCatalog.js",
@@ -61,11 +61,13 @@ explicit identity fails, and the selected page then must pass exact path-relativ
61
61
  base validation.
62
62
  Included HTML with neither signal is a component fragment and remains a
63
63
  complete package file.
64
- `arcane-os/preference-store` and
65
- `arcane-os/speech-playback` are the two portable runtime subpaths: Node package
66
- exports and managed browser keys both resolve directly to the canonical runtime
67
- module namespaces, while `arcane/PreferenceStore` and
68
- `arcane/SpeechPlayback` are the established browser import-map names. There is
64
+ `arcane-os/preference-store` and `arcane-os/speech-playback` are the two
65
+ portable subpaths that resolve directly to canonical runtime-module namespaces
66
+ from both Node package exports and managed browser keys;
67
+ `arcane/PreferenceStore` and `arcane/SpeechPlayback` remain their established
68
+ browser import-map names. The additional portable `arcane-os/speech-text`
69
+ subpath owns shared dependency-free speech-input cleanup and has the same
70
+ managed browser key. There is
69
71
  no exported `importMapApplication()` function, `generateImportMap()` function, or
70
72
  `arcane-os/import-map` package subpath.
71
73
 
@@ -159,7 +161,7 @@ imports such as:
159
161
  import ollama from 'arcane/Ollama';
160
162
  ```
161
163
 
162
- The physical-v1 tree lives entirely beneath `arcane/`. SDK `0.5.15` projects the
164
+ The physical-v1 tree lives entirely beneath `arcane/`. SDK `0.5.17` projects the
163
165
  complete canonical runtime and browser runtime selected by the installed SDK
164
166
  package. Runtime dependencies stay under
165
167
  `arcane/dependencies/`; the SDK event and browser-AI closure stays under
@@ -174,13 +176,13 @@ integrated physical route uses the same ordered include list in its one route.
174
176
  Omitting only that final optional entry is compatible. Removing, reordering, or
175
177
  renaming any preceding entry changes the physical contract.
176
178
 
177
- The `0.5.15` map derives its complete entries from the selected runtime graph;
179
+ The `0.5.17` map derives its complete entries from the selected runtime graph;
178
180
  application source imports do not select a fixed entry count. The operation
179
181
  result reports `imports`, `entryCount`, and `excludedModules`; reached-file
180
182
  traversal remains internal. The
181
183
  managed graph exposes `arcane-os/event-manager`, `arcane-os/ai/browser-wasm`,
182
184
  `arcane-os/ai/browser-speech`, `arcane-os/preference-store`, and
183
- `arcane-os/speech-playback`; dependency compatibility mappings are added when
185
+ `arcane-os/speech-playback` plus `arcane-os/speech-text`; dependency compatibility mappings are added when
184
186
  runtime or SDK root traversal observes them.
185
187
 
186
188
  The focused physical targets remain stable when their bindings are reached:
@@ -190,6 +192,7 @@ The focused physical targets remain stable when their bindings are reached:
190
192
  | `arcane-os/event-manager` | `./arcane/sdk/event-manager.mjs` |
191
193
  | `arcane-os/ai/browser-wasm` | `./arcane/sdk/ai/browser-wasm.mjs` |
192
194
  | `arcane-os/ai/browser-speech` | `./arcane/sdk/ai/browser-speech.mjs` |
195
+ | `arcane-os/speech-text` | `./arcane/sdk/speech-text.mjs` |
193
196
  | `arcane-os/preference-store` | `./arcane/modules/PreferenceStore.js` |
194
197
  | `arcane-os/speech-playback` | `./arcane/modules/SpeechPlayback.js` |
195
198
  | `event-pubsub` | `./arcane/sdk/dependencies/event-pubsub/index.js` |
@@ -202,8 +205,9 @@ invented package bindings. Development serves the selected app plus the
202
205
  complete selected tree.
203
206
  Packaging copies the same map, app entry, and physical content into `dist/<id>`;
204
207
  targets never resolve through the consumer workspace's root `node_modules/`.
205
- The two lowercase static runtime package specifiers above are also exact npm
206
- package exports, so Node and managed-browser source can share those specifiers
208
+ The two lowercase static runtime package specifiers above and the shared
209
+ speech-text helper are exact npm package exports, so Node and managed-browser
210
+ source can share those specifiers
207
211
  without data URLs, copied modules, or a consumer-owned loader.
208
212
 
209
213
  `generateImportMap()` is an internal toolchain operation, not a package export.
@@ -239,8 +243,8 @@ exactly `dependencyName`, `packageSource`,
239
243
  `canonicalPackageRoot`, `packageName`, `packageVersion`, `runtimeRoot`,
240
244
  and `browserRuntimeRoot`. A
241
245
  workspace may use the canonical dependency name or one exact npm alias such as
242
- `npm:arcane-os@0.5.15`; the physical package manifest must still identify
243
- exactly as `arcane-os@0.5.15`. Canonical-plus-alias duplicates, multiple aliases,
246
+ `npm:arcane-os@0.5.17`; the physical package manifest must still identify
247
+ exactly as `arcane-os@0.5.17`. Canonical-plus-alias duplicates, multiple aliases,
244
248
  links/junctions, indirect package roots, or version drift are reported.
245
249
 
246
250
  For external workspaces, `arcane dev` serves the projected `arcane/` root,
@@ -351,7 +355,7 @@ does not silently delete the app-owned cache. A complete whole member supersedes
351
355
  its current resumable fragments. Cleanup failure is warned without hiding the
352
356
  usable model.
353
357
 
354
- SDK `0.5.15` requires WebGPU. Load requests full offload and waits for the runtime
358
+ SDK `0.5.17` requires WebGPU. Load requests full offload and waits for the runtime
355
359
  to report a loaded model. `navigator.gpu` presence alone is
356
360
  not readiness. There is no CPU fallback, partial-offload success mode, or
357
361
  silent switch to native/Core/cloud inference.
@@ -360,17 +360,21 @@ unload.
360
360
  Chat listens for the bound (or current global) AI runtime's `ai-tts-failure`
361
361
  event and forwards its complete Error and exact operation boundary to
362
362
  `speech.reportTTSError()`. Playback-start or playback-resume failures can occur
363
- after Chat's two-argument `streamTTS()` preparation promise has resolved.
363
+ after Chat's `streamTTS()` preparation promise has resolved.
364
364
  SDK `0.5.12` adds a `waitForPlayback:true` mode for the
365
365
  terminal result of the segments submitted by one invocation; Chat continues
366
366
  to feed chunks without waiting for playback. Runtime mute, cancellation,
367
367
  permission waiting, and stale generations remain non-errors.
368
368
 
369
- Each visible model chunk is still forwarded to `AI.streamTTS()` in arrival
370
- order. The AI owner preserves exact segmentation, admits completed segments to
371
- bounded synthesis immediately, and schedules the contiguous ready audio prefix
372
- on the audio clock. Chat neither creates a second queue nor reorders, combines,
373
- or rewrites speech text.
369
+ Each visible model chunk is forwarded to `AI.streamTTS(text)` in arrival
370
+ order. The AI owner automatically removes repeated formatting marks from an
371
+ outbound speech-input copy before the
372
+ configured segmentation, admits completed segments to bounded synthesis
373
+ immediately, and schedules the contiguous ready audio prefix on the audio
374
+ clock. Chat retains the original Markdown for display and storage and owns no
375
+ second speech queue. No app option selects or disables this cleanup. Single
376
+ punctuation, ordinary repeated punctuation, language, and voice retain their
377
+ existing behavior.
374
378
 
375
379
  Events: `chat-ready`, `chat-session-bound`, `chat-session-message`,
376
380
  `chat-session-error`, `chat-send-message`, `chat-send-error`,
@@ -88,6 +88,7 @@ own asynchronous work, cancellation, and backpressure.
88
88
  | [`Mail.js`](#mailjs) | esm | Builds complete reports and prefers the native mail capability with an explicit HTTP transport fallback. | Browser/native hybrid + cloud | Mail inputs/results normalized; transport failures mixed. |
89
89
  | [`MailOutbox.mjs`](#mailoutboxmjs) | esm | Persists complete mail reports before delivery and normalizes idempotent enqueue, retry, reconciliation, and invalid-record maintenance. | Browser/native WebView or compatible injected host | Complete records, full work, cancellation, and lifecycle states normalized; storage, lock, and delivery failures coded. |
90
90
  | [`MailTransport.mjs`](#mailtransportmjs) | esm | Sends one complete mail report to a normalized HTTP(S) endpoint. | Browser/server with fetch + cloud | Normalized endpoint and transport errors; remote detail preserved. |
91
+ | [`MarkdownSpeech.js`](#markdownspeechjs) | esm | Removes repeated Markdown formatting marks from streamed narration. | Cross-host | Speech-only filtering; single marks and ordinary punctuation remain literal. |
91
92
  | [`Marked.min.js`](#markedminjs) | esm | Vendored Marked 18.0.5 Markdown lexer, parser, renderer, extension, and walk-token API. | Cross-host vendor module | Vendor-native Marked contract. |
92
93
  | [`MD.js`](#mdjs) | esm | Renders complete Markdown with Marked and exposes the complete rendered markup. | Browser / native WebView | Complete raw and rendered Marked values; parse errors vendor-native. |
93
94
  | [`MemoryRecords.js`](#memoryrecordsjs) | esm | Normalizes memory content and detects meaningful stored memory. | Cross-host | Fully normalized string/boolean results. |
@@ -108,7 +109,7 @@ own asynchronous work, cancellation, and backpressure.
108
109
  | [`ScamRiskPolicy.js`](#scamriskpolicyjs) | esm | Combines deterministic scam signals with optional Arcane blocked-domain evidence and safety guidance. | Cross-host | Complete mutable results; blocked-domain policy requires `secure:true`. |
109
110
  | [`ScopedOPFSCache.js`](#scopedopfscachejs) | esm | Provides a narrow exact-key JSON cache inside one app-owned OPFS namespace. | Browser / native WebView | Filename-safe keys, complete JSON values, and malformed-cache cleanup normalized; storage errors mixed. |
110
111
  | [`ScreenCapture.js`](#screencapturejs) | esm | Captures a display surface as image, video, or GIF with explicit lifecycle events. | Browser / native WebView | State/events normalized; permission and codec errors mixed. |
111
- | [`SpeechPlayback.js`](#speechplaybackjs) | esm | Preserves exact nonblank text as one speech segment, queues latest-request synthesis, and controls lookahead HTML audio playback. | Browser + native bridge | Exact input text and state normalized; provider/media failures mixed. |
112
+ | [`SpeechPlayback.js`](#speechplaybackjs) | esm | Admits complete speech segments to a capacity-advertising provider immediately, retains serialized native/custom lookahead, and plays every result in exact indexed order. | Browser + native bridge | Stored part text stays exact; the outbound speech-input copy receives automatic formatting-mark cleanup; provider/media failures remain mixed. |
112
113
  | [`StaticDocumentCatalog.js`](#staticdocumentcatalogjs) | esm | Loads a positive static document inventory with cache, search, and complete context. | Browser / native WebView / server with fetch | Mutable complete catalog/content normalization; malformed data and transport failures remain visible. |
113
114
  | [`SystemAppearance.js`](#systemappearancejs) | esm | Reads or applies native appearance, returning an explicit unsupported browser state when no bridge exists. | Browser/native hybrid | Absent bridge normalized; native result/error preserved. |
114
115
  | [`SystemPlatformPresentation.js`](#systemplatformpresentationjs) | classic-script | Maps kernel names to presentation labels/classes without granting platform authority. | Browser / native WebView classic script | Fully normalized presentation only. |
@@ -262,7 +263,7 @@ reports `requestedDevice`, `selectedDevice`, `maxConcurrentRequests`, and
262
263
  WASM fallback after a successful load. Calling `status()` without options keeps
263
264
  the existing sticky lifecycle snapshot and does not inspect provider execution.
264
265
  Provider inspection failures are surfaced to the caller.
265
- `fetchTTS({model,voice,input,responseFormat,speed},signal)` accepts the public
266
+ `fetchTTS({model,voice,input,responseFormat,speed},signal,preparation={})` accepts the public
266
267
  provider-neutral synthesis shape, requires any explicit model to match the
267
268
  selected route, and fills an omitted voice only from the selected model
268
269
  catalog's `defaultVoice`. An omitted response format preserves the instance's
@@ -272,9 +273,14 @@ the setting is the instance's `opus` default and the model rejects it, the catal
272
273
  `speech.defaultResponseFormat` is used, while any other unsupported setting is
273
274
  rejected. It propagates the caller-owned signal and returns a playable `Blob`;
274
275
  it does not independently choose a provider, cloud fallback, model, runtime, or
275
- voice policy for the application. `streamTTS(text='',end=false,options={})` and
276
+ voice policy for the application. Every call removes repeated same formatting
277
+ marks from a cloned outbound input before delegation; the caller's payload stays
278
+ unchanged. The third `preparation` argument is reserved for SDK-internal
279
+ delegation, where `{speechInputPrepared:true}` prevents a second cleanup pass;
280
+ applications omit it. `streamTTS(text='',end=false,options={})` and
276
281
  `finishTTS()` use this same request boundary. The third-argument options below
277
- are available in SDK `0.5.12`:
282
+ are available in SDK `0.5.12`. The `textFormat` compatibility extra added in
283
+ SDK `0.5.16` is ignored beginning in `0.5.17`:
278
284
 
279
285
  | Field | Default | Meaning |
280
286
  | --- | --- | --- |
@@ -282,12 +288,23 @@ are available in SDK `0.5.12`:
282
288
  | `speed` | Current `ai.voiceSpeed` | A supplied positive speed is captured for those segments and forwarded to `fetchTTS()`. It does not change `ai.voiceSpeed`. |
283
289
  | `pauseAfterMs` | `0` | Finite, nonnegative milliseconds placed after the final extracted segment on the existing audio clock. Invalid values throw `RangeError`; no pause is inserted between this call's other segments. |
284
290
  | `waitForPlayback` | `false` | Omission retains the preparation promise. With `true`, the promise resolves after every extracted segment reaches a terminal playback state: `true` when all naturally end, or `false` after terminal cancellation or failure. |
291
+ | `textFormat` | Ignored compatibility extra | Repeated same formatting marks are removed automatically from every TTS call. This value no longer selects or disables cleanup. |
285
292
 
286
293
  The voice and speed use the existing `fetchTTS()` validation and error path.
287
- No option rewrites the submitted text. Overrides belong to the segments
294
+ The automatic cleanup changes only the outbound speech-input copy; displayed,
295
+ stored, model, and caller-owned content remains exact. It is a narrow
296
+ formatting-mark filter, not a full Markdown parser: links, code contents, list
297
+ text, single marks, ordinary punctuation, and other characters remain literal.
298
+ A trailing candidate mark waits for the next character so a repeated run split
299
+ across chunks is still omitted. Ordinary prose streams immediately.
300
+ `end:true`, `finishTTS()`, muted terminal calls, and `stopAudio()` clear pending
301
+ formatting state. `textFormat` is not an opt-out.
302
+
303
+ Voice, speed, pause, and playback overrides belong to the segments
288
304
  extracted in that invocation, including any text buffered by an earlier call.
289
- Options are not retained with an unfinished `end:false` remainder; a later
290
- call supplies its own options, and `finishTTS()` uses defaults. Use `end:true`
305
+ Those overrides are not retained with an unfinished `end:false` remainder; a
306
+ later call supplies its own options, and `finishTTS()` uses their defaults
307
+ while flushing any pending formatting mark. Use `end:true`
291
308
  for a complete passage. A call extracting no segments resolves
292
309
  `true` without waiting for earlier jobs; `finishTTS()` remains a preparation
293
310
  flush, not a queue-wide playback barrier. A muted call resolves `false`.
@@ -311,8 +328,9 @@ commas, and hyphens remain inside a segment when they join Unicode letters or
311
328
  numbers. A potentially joining mark at the current end of an incremental stream
312
329
  waits for the next character or terminal flush before the boundary is decided;
313
330
  `wordCadence` completes one after that many whole words. The earliest available
314
- boundary wins. Segmentation preserves every character, including punctuation
315
- and whitespace. Every completed segment enters synthesis immediately; provider
331
+ boundary wins. Segmentation preserves every character of the already prepared
332
+ speech text, including punctuation and whitespace. Every completed segment
333
+ enters synthesis immediately; provider
316
334
  capacity supplies FIFO backpressure while allowing bounded TTS work to overlap.
317
335
  A later segment may finish synthesis first, but playback schedules only the
318
336
  contiguous ready prefix in original order. Decoded buffers with known duration
@@ -649,9 +667,11 @@ The singleton exposes read-only `protocol`, `configured`, and `speechMuted`;
649
667
  `status(role=null,options={})`; `catalog(role)`;
650
668
  `inspect(role,options={})`; `start(options)`; `load(role,options={})`;
651
669
  `unload(role,options={})`; `dispose(role,options={})`;
652
- `disposeAll(options={})`; `cancel(role)`; `request(role,options={})`;
670
+ `disposeAll(options={})`; `cancel(role)`;
671
+ `request(role,options={},preparation={})`;
653
672
  `chat(payload,options={})`; `stream(payload,options={})`;
654
- `transcribe(payload,options={})`; `synthesize(payload,options={})`; and
673
+ `transcribe(payload,options={})`;
674
+ `synthesize(payload,options={},preparation={})`; and
655
675
  `setSpeechMuted(muted)`. Provider payloads must be data-only; callbacks,
656
676
  accessors, symbols, and cycles are rejected at the provider boundary.
657
677
 
@@ -667,6 +687,14 @@ records are the closed `{llm,stt,tts}`, `{stt,tts}`, or
667
687
  `{providers,routes,expectedProviders}` shapes described below.
668
688
  `configureFromTuple()` accepts exactly six provider/model preference entries.
669
689
 
690
+ Every direct TTS `request()` or `synthesize()` call removes repeated same
691
+ formatting marks from a cloned outbound payload's `input` or `text` field. The
692
+ caller's payload and request records remain unchanged. The optional
693
+ `preparation` argument is reserved for SDK-owned delegation;
694
+ `{speechInputPrepared:true}` prevents a second pass after another SDK speech
695
+ boundary has already cleaned the copy. Applications omit that argument. LLM
696
+ and STT payloads are unaffected.
697
+
670
698
  `register()` returns the provider's single unregister closure; caller-
671
699
  registered providers remain caller-owned. The high-level
672
700
  `AI.configureBrowserSpeech()` boundary is different: AI constructs, registers,
@@ -2244,6 +2272,43 @@ import * as module from '/arcane/modules/MailTransport.mjs';
2244
2272
  console.log(Object.keys(module));
2245
2273
  ```
2246
2274
 
2275
+ ## MarkdownSpeech.js
2276
+
2277
+ ### Overview
2278
+
2279
+ Re-exports the streaming `MarkdownSpeech` filter from the shared
2280
+ `arcane-os/speech-text` package entrypoint. The filter removes repeated runs of
2281
+ `*`, `#`, `_`, backtick, and `~` before speech segmentation, including runs
2282
+ arriving across separate chunks. Single marks, ellipses, quoted sentence
2283
+ endings, whitespace, and all other text remain literal. It neither interprets
2284
+ links nor changes language or voice.
2285
+
2286
+ ### Public surface
2287
+
2288
+ `MarkdownSpeech`; `append(text='',end=false)`, `reset()`.
2289
+
2290
+ Exact exports: `MarkdownSpeech`.
2291
+
2292
+ ### Availability and normalization
2293
+
2294
+ **Cross-host.** The runtime projection and public package entrypoint share the
2295
+ same dependency-free implementation. `append()` returns only
2296
+ the newly available narration. Only a trailing candidate marker and whether
2297
+ it repeats are retained; ordinary text is emitted immediately. Terminal
2298
+ `append(text,true)` flushes a single pending mark and resets state. `reset()`
2299
+ clears pending formatting state when its narration is cancelled. Non-string
2300
+ input throws `TypeError`.
2301
+
2302
+ ### Example
2303
+
2304
+ ```javascript
2305
+ import {MarkdownSpeech} from '/arcane/modules/MarkdownSpeech.js';
2306
+
2307
+ const speech = new MarkdownSpeech();
2308
+ const first = speech.append('**Hello'); // Hello
2309
+ const last = speech.append('**...', true); // ...
2310
+ ```
2311
+
2247
2312
  ## Marked.min.js
2248
2313
 
2249
2314
  ### Overview
@@ -2851,15 +2916,17 @@ console.log(Object.keys(module));
2851
2916
 
2852
2917
  Preserves exact nonblank text, admits complete speech segments according to the
2853
2918
  selected client's advertised capacity, and plays indexed HTML audio in exact
2854
- input order.
2919
+ input order. Stored parts remain exact; only each outbound synthesis payload
2920
+ copy receives automatic formatting-mark cleanup.
2855
2921
 
2856
2922
  ### Public surface
2857
2923
 
2858
- `SpeechPlayback` class/default, `SPEECH_PLAYBACK_STATE_EVENT`,
2859
- `splitSpeechText()`, and playback lifecycle APIs.
2924
+ `SpeechPlayback` class/default, the shared voice compatibility catalogs,
2925
+ `SPEECH_PLAYBACK_STATE_EVENT`, `splitSpeechText()`, the optional constructor
2926
+ state callback, and playback lifecycle APIs.
2860
2927
 
2861
- Exact exports: `SPEECH_PLAYBACK_STATE_EVENT`, `SpeechPlayback`, `default`, and
2862
- `splitSpeechText`.
2928
+ Exact exports: `SPEECH_PLAYBACK_STATE_EVENT`, `SPEECH_VOICE_ALIASES`,
2929
+ `SPEECH_VOICE_OPTIONS`, `SpeechPlayback`, `default`, and `splitSpeechText`.
2863
2930
 
2864
2931
  ```text
2865
2932
  new SpeechPlayback({
@@ -2869,6 +2936,7 @@ new SpeechPlayback({
2869
2936
  voice=null,
2870
2937
  responseFormat=null,
2871
2938
  speed=1,
2939
+ onState=()=>{},
2872
2940
  createObjectURL,
2873
2941
  revokeObjectURL,
2874
2942
  delay,
@@ -2877,17 +2945,28 @@ new SpeechPlayback({
2877
2945
  ```
2878
2946
 
2879
2947
  `speech` must expose either `fetchTTS(payload, signal)` or
2880
- `synthesize(payload, {signal})`. `prepare({key,parts,model,voice,responseFormat,
2948
+ `synthesize(payload, {signal})`. `SpeechPlayback` also supplies a third
2949
+ SDK-internal preparation object; existing two-argument clients may ignore it.
2950
+ `prepare({key,parts,model,voice,responseFormat,
2881
2951
  speed,autoplay=true})` uses only caller-supplied model, voice, and response-format
2882
2952
  values; those three omitted values remain omitted so the selected AI/model
2883
2953
  catalog may provide its documented defaults. Speed defaults to `1`, is normalized
2884
- as a positive number, and is always sent. There is no
2885
- hard-coded model, response format, voice, or cloud/browser fallback.
2954
+ as a positive number, and is always sent. `SPEECH_VOICE_OPTIONS` is the ordered
2955
+ mutable compatibility array `{value,label}` for `alloy`, `ash`, `ballad`,
2956
+ `coral`, `echo`, `fable`, `nova`, `onyx`, `sage`, and `shimmer`;
2957
+ `SPEECH_VOICE_ALIASES` is the mutable `Set` of those values. The class does not
2958
+ select either catalog or promise that a selected provider supports its values.
2959
+ There is no hard-coded model, response format, voice, or cloud/browser fallback.
2886
2960
  `splitSpeechText(value)` uses trimming only to detect blank input, then returns
2887
2961
  the caller's exact string in one mutable array without trimming, splitting, or
2888
2962
  freezing it. `prepare()` likewise preserves each nonblank part's exact `input`
2889
2963
  string while normalizing its other playback fields into a new mutable record.
2890
- The class applies no part-count, character-count, pause, or input upper cap.
2964
+ At synthesis time, `requestSpeech()` copies that record, removes repeated same
2965
+ formatting marks from only the outbound `input`, and delegates with the
2966
+ SDK-internal `{speechInputPrepared:true}` argument so downstream SDK boundaries
2967
+ do not apply the non-idempotent filter again. Original part objects, stored
2968
+ parts, displayed text, and all non-input payload fields remain unchanged. The
2969
+ class applies no part-count, character-count, pause, or input upper cap.
2891
2970
 
2892
2971
  ### Admission and playback order
2893
2972
 
@@ -2913,8 +2992,13 @@ Every preparation owns an operation ID and one AbortController for each active
2913
2992
  synthesis segment or playback delay. Replacement,
2914
2993
  `stop()`, `cancel()`, and `destroy()` abort their owned signals, suppress stale
2915
2994
  settlement, release Blob URLs, and publish synchronous
2916
- `speech-playback-state` occurrences through `globalThis.arcaneEvents`.
2917
- Subscribers receive mutable public state detail. The detail contains
2995
+ `speech-playback-state` occurrences through `globalThis.arcaneEvents` before
2996
+ calling the optional `onState(detail)` function synchronously. Canonical
2997
+ subscribers and the callback observe the same public field values at dispatch
2998
+ time, but object identity is not promised. Both surfaces expose mutable public
2999
+ state detail. A callback failure is reported through `globalThis.reportError`
3000
+ when available, otherwise `console.error`, and does not replace playback
3001
+ settlement. The detail contains
2918
3002
  `state`, `message`, `key`, `index`, `total`, `producing`, `buffered`, `hasAudio`,
2919
3003
  `operationId`, `code`, and `reason`; a first-segment provider rejection remains
2920
3004
  preserved to the `prepare()` caller. Later failures surface when ordered
@@ -2956,7 +3040,10 @@ const audio = document.body.appendChild(document.createElement('audio'));
2956
3040
  audio.controls = true;
2957
3041
  const speech = new SpeechPlayback({
2958
3042
  audio,
2959
- speech: globalThis.ai
3043
+ speech: globalThis.ai,
3044
+ onState(detail) {
3045
+ console.log('Speech state:', detail.state);
3046
+ }
2960
3047
  });
2961
3048
  const button = document.body.appendChild(document.createElement('button'));
2962
3049
  button.textContent = 'Speak';
@@ -2972,6 +3059,19 @@ button.addEventListener('click', async function speakCompleteSegments() {
2972
3059
  });
2973
3060
  ```
2974
3061
 
3062
+ The shared compatibility catalogs are also available directly. They do not
3063
+ select a voice for `SpeechPlayback`:
3064
+
3065
+ ```javascript
3066
+ import {
3067
+ SPEECH_VOICE_ALIASES,
3068
+ SPEECH_VOICE_OPTIONS
3069
+ } from '/arcane/modules/SpeechPlayback.js';
3070
+
3071
+ console.log(SPEECH_VOICE_OPTIONS[0]); // {value: 'alloy', label: 'Alloy'}
3072
+ console.log(SPEECH_VOICE_ALIASES.has('alloy')); // true
3073
+ ```
3074
+
2975
3075
  With the default browser speech configuration, both parts enter its capacity-4
2976
3076
  queue immediately and still play first, then second. Inspect the selected
2977
3077
  execution device without guessing from console warnings: