@appilots/web-sdk 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -391,6 +391,13 @@ interface AppilotsClientOptions {
391
391
  * id. Identifiers are dashboard-only — never sent to the LLM.
392
392
  */
393
393
  user?: AppilotsUser;
394
+ /**
395
+ * Accessibility: ask the agent to reply in plain language (short
396
+ * sentences, everyday words, numbered steps). `undefined` defers to the
397
+ * project setting in the dashboard; `true`/`false` overrides it for this
398
+ * session. Change it later with `setPlainLanguage`.
399
+ */
400
+ plainLanguage?: boolean;
394
401
  }
395
402
  interface AppilotsUser {
396
403
  /** The app's own user id (matches your backend). */
@@ -455,6 +462,8 @@ interface SendMessageContext {
455
462
  missionId?: string;
456
463
  /** Supported device language, independent of the app/map/visible UI locale. */
457
464
  deviceLocale?: SupportedAppilotsLocale;
465
+ /** Set by the client from `plainLanguage` / `setPlainLanguage`; see there. */
466
+ plainLanguage?: boolean;
458
467
  /**
459
468
  * Which client platform this observation came from — `'react-native'`,
460
469
  * `'web'`, `'android'`, `'ios'`, or any other client-defined string.
@@ -509,6 +518,7 @@ declare class AppilotsClient {
509
518
  private readonly mcpVersion;
510
519
  private readonly introspectionReporter;
511
520
  private readonly user;
521
+ private plainLanguage;
512
522
  private sessionId;
513
523
  private mission;
514
524
  private missionDispatchPaused;
@@ -549,6 +559,14 @@ declare class AppilotsClient {
549
559
  * must not take the message down with it.
550
560
  */
551
561
  private introspectionFragment;
562
+ /**
563
+ * Per-session plain-language override — e.g. from the signed-in user's
564
+ * accessibility profile. `undefined` hands the decision back to the
565
+ * project setting. Applies from the next request on.
566
+ */
567
+ setPlainLanguage(value: boolean | undefined): void;
568
+ /** Every relay-bound context carries the reply-style preference. */
569
+ private withPreferences;
552
570
  sendMessage(content: string, context?: SendMessageContext): Promise<SendMessageResponse>;
553
571
  /**
554
572
  * Streaming variant of `sendMessage` — consumes the server's SSE
@@ -733,6 +751,87 @@ declare class AppilotsClient {
733
751
  }): () => void;
734
752
  }
735
753
 
754
+ /**
755
+ * The speech seam.
756
+ *
757
+ * The SDK ships no speech engine and never will. Every usable one on
758
+ * React Native is a native module — `@react-native-voice/voice`,
759
+ * `expo-speech-recognition`, a vendor SDK — and a dependency here would
760
+ * force native linking on every integrator, including the majority who
761
+ * will never turn voice on. Worse, it would pick their engine for them:
762
+ * the right one depends on the locale they need, whether recognition
763
+ * may leave the device, and what their app store review already covers.
764
+ *
765
+ * So this is an interface and nothing else, injected the same way
766
+ * `GeometryHost` and the `Alert` in `confirmedDestructiveAlert` are.
767
+ * Pass an adapter and the mic button appears; pass nothing and
768
+ * `resolveExperience` reports `input: 'voice'` as an unmet request
769
+ * rather than rendering a button that does nothing.
770
+ *
771
+ * It lives in client-core because the web SDK takes the same contract:
772
+ * its built-in Web Speech engine is one implementation of it, and a
773
+ * host that needs recognition to stay on-device (or a vendor engine)
774
+ * passes its own.
775
+ *
776
+ * Everything here is transcription only. Playing audio back is not
777
+ * modelled, because nothing in the loop produces audio.
778
+ */
779
+ /** What the SDK needs to run a mic button (hold-to-talk or toggle). */
780
+ interface SpeechHost {
781
+ /**
782
+ * Begin listening. Resolve once capture is actually running, so the
783
+ * UI can wait before telling the user to speak.
784
+ *
785
+ * Implementations own permissions: if the OS prompt has not been
786
+ * accepted, request it here and reject when refused.
787
+ */
788
+ start(options: SpeechStartOptions): Promise<void>;
789
+ /**
790
+ * Stop listening and resolve with the final transcript.
791
+ *
792
+ * Resolving with an empty string is the normal outcome for "the user
793
+ * pressed and said nothing" — it must not reject for that.
794
+ */
795
+ stop(): Promise<string>;
796
+ /** Abandon the capture. No transcript, no error. */
797
+ cancel(): Promise<void>;
798
+ /**
799
+ * Partial results while the user is still talking, if the engine has
800
+ * them. Return a disposer. Optional — a host without interim results
801
+ * simply shows the placeholder until `stop` resolves.
802
+ */
803
+ onPartial?(listener: (text: string) => void): () => void;
804
+ /**
805
+ * The engine ended the capture on its own — silence timeout, lost
806
+ * permission, network — without `stop` or `cancel` being called.
807
+ * `error` is set when it ended by failure. Return a disposer.
808
+ *
809
+ * Optional, but without it a UI cannot leave `listening` until the
810
+ * user presses again: the button would claim to record while the
811
+ * engine is not.
812
+ */
813
+ onEnd?(listener: (error?: unknown) => void): () => void;
814
+ }
815
+ interface SpeechStartOptions {
816
+ /**
817
+ * BCP-47 tag resolved from the chat's locale, so a pt-BR chat does
818
+ * not transcribe against a US English model. Hosts that only support
819
+ * one locale may ignore it.
820
+ */
821
+ locale: string;
822
+ }
823
+ /** Where a mic button is in its cycle. */
824
+ type SpeechState = 'idle' | 'starting' | 'listening' | 'stopping' | 'error';
825
+ /**
826
+ * Is this object usable as a host?
827
+ *
828
+ * Checked at the boundary rather than trusted from the type, because
829
+ * the value crosses from the host app's JavaScript — where it may have
830
+ * come from a config file, a lazy import that failed, or a version of
831
+ * an adapter predating a method.
832
+ */
833
+ declare function isSpeechHost(value: unknown): value is SpeechHost;
834
+
736
835
  type ActionFailureCategory = 'component-not-found' | 'disabled' | 'validation' | 'network-5xx' | 'network-4xx' | 'screen-timeout' | 'ambiguous-target' | 'unknown';
737
836
  interface ActionDiagnose {
738
837
  category: ActionFailureCategory;
@@ -1312,7 +1411,17 @@ interface AppilotsWebRuntimeOptions {
1312
1411
  apiKey?: string;
1313
1412
  apiBaseUrl?: string;
1314
1413
  debug?: boolean;
1414
+ /**
1415
+ * How long (ms) the client waits for the assistant before giving up on
1416
+ * a request, and how long a reply stream may stay silent. Default
1417
+ * 60000; agent continuations always get at least 150000. Raise it for
1418
+ * users or networks that need more time — the user's message stays in
1419
+ * the transcript either way.
1420
+ */
1421
+ timeout?: number;
1315
1422
  appVersion?: string;
1423
+ /** Plain-language replies for this session; omit to follow the dashboard. */
1424
+ plainLanguage?: boolean;
1316
1425
  mcpVersion?: string;
1317
1426
  user?: {
1318
1427
  id: string;
@@ -1386,4 +1495,21 @@ declare class AppilotsWebRuntime {
1386
1495
  private screenActionsFor;
1387
1496
  }
1388
1497
 
1389
- export { type AgentAction as A, type ControlEvidence as C, type EscalationState as E, RateLimitedError as R, type ScreenActionMetadataLike as S, WebNavigationAdapter as W, type AppilotsLocale as a, type AgentPermissions as b, type AppilotsEvent as c, type ActionExecutionResult as d, type ActionDiagnose as e, ActionQueueMachine as f, type AgentActionType as g, AppilotsClient as h, type AppilotsEventHandler as i, AppilotsWebRuntime as j, type AppilotsWebRuntimeOptions as k, type ChatMessage as l, ChatSessionMachine as m, type ChatSessionOptions as n, type ChatSessionState as o, type EscalationStrings as p, type WebNavigationAdapterOptions as q, type WebNavigationState as r, type WebRoute as s, type WebScreenMetadata as t, buildPath as u, matchRoute as v };
1498
+ /**
1499
+ * A `SpeechHost` over the browser's Web Speech API.
1500
+ *
1501
+ * Opt-in and dependency-free: `createWebSpeechHost()` returns null when
1502
+ * the browser has no `SpeechRecognition` (Firefox, most WebViews), and
1503
+ * the chat hides the mic instead of rendering a button that fails.
1504
+ *
1505
+ * Privacy: in Chrome and Edge the audio is streamed to the vendor's
1506
+ * speech service — it leaves the device. Safari may recognise on-device
1507
+ * or on Apple's servers. Nothing here can make that local; a host that
1508
+ * needs it passes its own `SpeechHost`. See guides/voice-input in the docs.
1509
+ */
1510
+
1511
+ declare function isWebSpeechSupported(): boolean;
1512
+ /** Null when the browser has no speech recognition. */
1513
+ declare function createWebSpeechHost(): SpeechHost | null;
1514
+
1515
+ export { type AgentAction as A, isWebSpeechSupported as B, type ControlEvidence as C, matchRoute as D, type EscalationState as E, type MissionView as M, RateLimitedError as R, type ScreenActionMetadataLike as S, WebNavigationAdapter as W, type AppilotsLocale as a, type AgentPermissions as b, type AppilotsEvent as c, type ActionExecutionResult as d, type ActionDiagnose as e, ActionQueueMachine as f, type AgentActionType as g, AppilotsClient as h, type AppilotsEventHandler as i, AppilotsWebRuntime as j, type AppilotsWebRuntimeOptions as k, type ChatMessage as l, ChatSessionMachine as m, type ChatSessionOptions as n, type ChatSessionState as o, type EscalationStrings as p, type SpeechHost as q, type SpeechStartOptions as r, type SpeechState as s, type WebNavigationAdapterOptions as t, type WebNavigationState as u, type WebRoute as v, type WebScreenMetadata as w, buildPath as x, createWebSpeechHost as y, isSpeechHost as z };
@@ -391,6 +391,13 @@ interface AppilotsClientOptions {
391
391
  * id. Identifiers are dashboard-only — never sent to the LLM.
392
392
  */
393
393
  user?: AppilotsUser;
394
+ /**
395
+ * Accessibility: ask the agent to reply in plain language (short
396
+ * sentences, everyday words, numbered steps). `undefined` defers to the
397
+ * project setting in the dashboard; `true`/`false` overrides it for this
398
+ * session. Change it later with `setPlainLanguage`.
399
+ */
400
+ plainLanguage?: boolean;
394
401
  }
395
402
  interface AppilotsUser {
396
403
  /** The app's own user id (matches your backend). */
@@ -455,6 +462,8 @@ interface SendMessageContext {
455
462
  missionId?: string;
456
463
  /** Supported device language, independent of the app/map/visible UI locale. */
457
464
  deviceLocale?: SupportedAppilotsLocale;
465
+ /** Set by the client from `plainLanguage` / `setPlainLanguage`; see there. */
466
+ plainLanguage?: boolean;
458
467
  /**
459
468
  * Which client platform this observation came from — `'react-native'`,
460
469
  * `'web'`, `'android'`, `'ios'`, or any other client-defined string.
@@ -509,6 +518,7 @@ declare class AppilotsClient {
509
518
  private readonly mcpVersion;
510
519
  private readonly introspectionReporter;
511
520
  private readonly user;
521
+ private plainLanguage;
512
522
  private sessionId;
513
523
  private mission;
514
524
  private missionDispatchPaused;
@@ -549,6 +559,14 @@ declare class AppilotsClient {
549
559
  * must not take the message down with it.
550
560
  */
551
561
  private introspectionFragment;
562
+ /**
563
+ * Per-session plain-language override — e.g. from the signed-in user's
564
+ * accessibility profile. `undefined` hands the decision back to the
565
+ * project setting. Applies from the next request on.
566
+ */
567
+ setPlainLanguage(value: boolean | undefined): void;
568
+ /** Every relay-bound context carries the reply-style preference. */
569
+ private withPreferences;
552
570
  sendMessage(content: string, context?: SendMessageContext): Promise<SendMessageResponse>;
553
571
  /**
554
572
  * Streaming variant of `sendMessage` — consumes the server's SSE
@@ -733,6 +751,87 @@ declare class AppilotsClient {
733
751
  }): () => void;
734
752
  }
735
753
 
754
+ /**
755
+ * The speech seam.
756
+ *
757
+ * The SDK ships no speech engine and never will. Every usable one on
758
+ * React Native is a native module — `@react-native-voice/voice`,
759
+ * `expo-speech-recognition`, a vendor SDK — and a dependency here would
760
+ * force native linking on every integrator, including the majority who
761
+ * will never turn voice on. Worse, it would pick their engine for them:
762
+ * the right one depends on the locale they need, whether recognition
763
+ * may leave the device, and what their app store review already covers.
764
+ *
765
+ * So this is an interface and nothing else, injected the same way
766
+ * `GeometryHost` and the `Alert` in `confirmedDestructiveAlert` are.
767
+ * Pass an adapter and the mic button appears; pass nothing and
768
+ * `resolveExperience` reports `input: 'voice'` as an unmet request
769
+ * rather than rendering a button that does nothing.
770
+ *
771
+ * It lives in client-core because the web SDK takes the same contract:
772
+ * its built-in Web Speech engine is one implementation of it, and a
773
+ * host that needs recognition to stay on-device (or a vendor engine)
774
+ * passes its own.
775
+ *
776
+ * Everything here is transcription only. Playing audio back is not
777
+ * modelled, because nothing in the loop produces audio.
778
+ */
779
+ /** What the SDK needs to run a mic button (hold-to-talk or toggle). */
780
+ interface SpeechHost {
781
+ /**
782
+ * Begin listening. Resolve once capture is actually running, so the
783
+ * UI can wait before telling the user to speak.
784
+ *
785
+ * Implementations own permissions: if the OS prompt has not been
786
+ * accepted, request it here and reject when refused.
787
+ */
788
+ start(options: SpeechStartOptions): Promise<void>;
789
+ /**
790
+ * Stop listening and resolve with the final transcript.
791
+ *
792
+ * Resolving with an empty string is the normal outcome for "the user
793
+ * pressed and said nothing" — it must not reject for that.
794
+ */
795
+ stop(): Promise<string>;
796
+ /** Abandon the capture. No transcript, no error. */
797
+ cancel(): Promise<void>;
798
+ /**
799
+ * Partial results while the user is still talking, if the engine has
800
+ * them. Return a disposer. Optional — a host without interim results
801
+ * simply shows the placeholder until `stop` resolves.
802
+ */
803
+ onPartial?(listener: (text: string) => void): () => void;
804
+ /**
805
+ * The engine ended the capture on its own — silence timeout, lost
806
+ * permission, network — without `stop` or `cancel` being called.
807
+ * `error` is set when it ended by failure. Return a disposer.
808
+ *
809
+ * Optional, but without it a UI cannot leave `listening` until the
810
+ * user presses again: the button would claim to record while the
811
+ * engine is not.
812
+ */
813
+ onEnd?(listener: (error?: unknown) => void): () => void;
814
+ }
815
+ interface SpeechStartOptions {
816
+ /**
817
+ * BCP-47 tag resolved from the chat's locale, so a pt-BR chat does
818
+ * not transcribe against a US English model. Hosts that only support
819
+ * one locale may ignore it.
820
+ */
821
+ locale: string;
822
+ }
823
+ /** Where a mic button is in its cycle. */
824
+ type SpeechState = 'idle' | 'starting' | 'listening' | 'stopping' | 'error';
825
+ /**
826
+ * Is this object usable as a host?
827
+ *
828
+ * Checked at the boundary rather than trusted from the type, because
829
+ * the value crosses from the host app's JavaScript — where it may have
830
+ * come from a config file, a lazy import that failed, or a version of
831
+ * an adapter predating a method.
832
+ */
833
+ declare function isSpeechHost(value: unknown): value is SpeechHost;
834
+
736
835
  type ActionFailureCategory = 'component-not-found' | 'disabled' | 'validation' | 'network-5xx' | 'network-4xx' | 'screen-timeout' | 'ambiguous-target' | 'unknown';
737
836
  interface ActionDiagnose {
738
837
  category: ActionFailureCategory;
@@ -1312,7 +1411,17 @@ interface AppilotsWebRuntimeOptions {
1312
1411
  apiKey?: string;
1313
1412
  apiBaseUrl?: string;
1314
1413
  debug?: boolean;
1414
+ /**
1415
+ * How long (ms) the client waits for the assistant before giving up on
1416
+ * a request, and how long a reply stream may stay silent. Default
1417
+ * 60000; agent continuations always get at least 150000. Raise it for
1418
+ * users or networks that need more time — the user's message stays in
1419
+ * the transcript either way.
1420
+ */
1421
+ timeout?: number;
1315
1422
  appVersion?: string;
1423
+ /** Plain-language replies for this session; omit to follow the dashboard. */
1424
+ plainLanguage?: boolean;
1316
1425
  mcpVersion?: string;
1317
1426
  user?: {
1318
1427
  id: string;
@@ -1386,4 +1495,21 @@ declare class AppilotsWebRuntime {
1386
1495
  private screenActionsFor;
1387
1496
  }
1388
1497
 
1389
- export { type AgentAction as A, type ControlEvidence as C, type EscalationState as E, RateLimitedError as R, type ScreenActionMetadataLike as S, WebNavigationAdapter as W, type AppilotsLocale as a, type AgentPermissions as b, type AppilotsEvent as c, type ActionExecutionResult as d, type ActionDiagnose as e, ActionQueueMachine as f, type AgentActionType as g, AppilotsClient as h, type AppilotsEventHandler as i, AppilotsWebRuntime as j, type AppilotsWebRuntimeOptions as k, type ChatMessage as l, ChatSessionMachine as m, type ChatSessionOptions as n, type ChatSessionState as o, type EscalationStrings as p, type WebNavigationAdapterOptions as q, type WebNavigationState as r, type WebRoute as s, type WebScreenMetadata as t, buildPath as u, matchRoute as v };
1498
+ /**
1499
+ * A `SpeechHost` over the browser's Web Speech API.
1500
+ *
1501
+ * Opt-in and dependency-free: `createWebSpeechHost()` returns null when
1502
+ * the browser has no `SpeechRecognition` (Firefox, most WebViews), and
1503
+ * the chat hides the mic instead of rendering a button that fails.
1504
+ *
1505
+ * Privacy: in Chrome and Edge the audio is streamed to the vendor's
1506
+ * speech service — it leaves the device. Safari may recognise on-device
1507
+ * or on Apple's servers. Nothing here can make that local; a host that
1508
+ * needs it passes its own `SpeechHost`. See guides/voice-input in the docs.
1509
+ */
1510
+
1511
+ declare function isWebSpeechSupported(): boolean;
1512
+ /** Null when the browser has no speech recognition. */
1513
+ declare function createWebSpeechHost(): SpeechHost | null;
1514
+
1515
+ export { type AgentAction as A, isWebSpeechSupported as B, type ControlEvidence as C, matchRoute as D, type EscalationState as E, type MissionView as M, RateLimitedError as R, type ScreenActionMetadataLike as S, WebNavigationAdapter as W, type AppilotsLocale as a, type AgentPermissions as b, type AppilotsEvent as c, type ActionExecutionResult as d, type ActionDiagnose as e, ActionQueueMachine as f, type AgentActionType as g, AppilotsClient as h, type AppilotsEventHandler as i, AppilotsWebRuntime as j, type AppilotsWebRuntimeOptions as k, type ChatMessage as l, ChatSessionMachine as m, type ChatSessionOptions as n, type ChatSessionState as o, type EscalationStrings as p, type SpeechHost as q, type SpeechStartOptions as r, type SpeechState as s, type WebNavigationAdapterOptions as t, type WebNavigationState as u, type WebRoute as v, type WebScreenMetadata as w, buildPath as x, createWebSpeechHost as y, isSpeechHost as z };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@appilots/web-sdk",
3
- "version": "0.2.0",
3
+ "version": "0.3.0",
4
4
  "description": "Appilots client for web applications: DOM/ARIA introspection, DOM action execution, and a thin React binding — framework-agnostic (React, Vue, vanilla)",
5
5
  "main": "dist/index.js",
6
6
  "module": "dist/index.mjs",