ngx-transformers 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,6 @@
1
1
  import * as _angular_core from '@angular/core';
2
- import { InjectionToken, EnvironmentProviders } from '@angular/core';
2
+ import { InjectionToken, EnvironmentProviders, Injector, ResourceRef } from '@angular/core';
3
+ import { WorkerLike } from 'ngx-transformers/worker';
3
4
  import * as ngx_transformers from 'ngx-transformers';
4
5
 
5
6
  /** Lifecycle of a model pipeline. `error` is terminal until load() is retried. */
@@ -12,9 +13,27 @@ interface ModelProgress {
12
13
  progress: number;
13
14
  loadedBytes: number;
14
15
  totalBytes: number;
16
+ /** Progress over every file of the model seen so far; absent when unknown. */
17
+ overall?: OverallProgress;
18
+ }
19
+ /**
20
+ * Download progress summed over the files of a model seen so far. A model
21
+ * is several files (config, tokenizer, weights) fetched in parallel; this is
22
+ * the steady number to put on a progress bar.
23
+ */
24
+ interface OverallProgress {
25
+ /** 0-100 over the bytes of every file with a known size. */
26
+ progress: number;
27
+ loadedBytes: number;
28
+ totalBytes: number;
29
+ /** Files seen so far. */
30
+ files: number;
31
+ /** Files fully downloaded. */
32
+ filesDone: number;
15
33
  }
16
34
  type TransformersDevice = 'wasm' | 'webgpu' | 'auto';
17
- type TransformersDtype = 'fp32' | 'fp16' | 'q8' | 'q4';
35
+ /** Weight formats Transformers.js can load; a checkpoint must ship the one you ask for. */
36
+ type TransformersDtype = 'fp32' | 'fp16' | 'q8' | 'int8' | 'uint8' | 'q4' | 'bnb4' | 'q4f16';
18
37
  /** Global defaults applied to every pipeline; see provideTransformers(). */
19
38
  interface NgxTransformersConfig {
20
39
  device?: TransformersDevice;
@@ -41,6 +60,14 @@ interface PipelineRequest {
41
60
  dtype?: TransformersDtype;
42
61
  options?: Record<string, unknown>;
43
62
  }
63
+ /**
64
+ * Options every run accepts. `signal` makes a run reject with an AbortError
65
+ * before it starts when the signal has already fired; a model run cannot be
66
+ * interrupted once started, but a superseded run need not begin.
67
+ */
68
+ interface RunOptions {
69
+ signal?: AbortSignal;
70
+ }
44
71
  interface ClassificationResult {
45
72
  label: string;
46
73
  score: number;
@@ -65,7 +92,7 @@ interface Transcription {
65
92
  /** Present when transcribe() was called with returnTimestamps. */
66
93
  chunks?: TranscriptionChunk[];
67
94
  }
68
- interface TranscribeOptions {
95
+ interface TranscribeOptions extends RunOptions {
69
96
  /** true for segment timestamps, 'word' for word-level. */
70
97
  returnTimestamps?: boolean | 'word';
71
98
  /** Split audio longer than ~30 s into chunks of this many seconds. */
@@ -78,7 +105,7 @@ interface TranscribeOptions {
78
105
  task?: 'transcribe' | 'translate';
79
106
  }
80
107
  /** Options for ZeroShotClassifier.classify(). */
81
- interface ZeroShotOptions {
108
+ interface ZeroShotOptions extends RunOptions {
82
109
  /** Score every label on its own (several can be high) instead of picking one. */
83
110
  multiLabel?: boolean;
84
111
  /** NLI hypothesis with a {} placeholder for the label; default "This example is {}.". */
@@ -88,7 +115,7 @@ interface ZeroShotOptions {
88
115
  * Language pair for one translate() call. Codes follow the checkpoint:
89
116
  * ISO 639-1 for opus-mt ("en", "ru"), FLORES-200 for NLLB ("eng_Latn").
90
117
  */
91
- interface TranslateOptions {
118
+ interface TranslateOptions extends RunOptions {
92
119
  from?: string;
93
120
  to?: string;
94
121
  }
@@ -99,6 +126,24 @@ interface TranslatorOptions extends Partial<Omit<PipelineRequest, 'task'>> {
99
126
  /** Default target language for translate(). */
100
127
  to?: string;
101
128
  }
129
+ /** One turn of a chat prompt for TextGenerator. */
130
+ interface ChatMessage {
131
+ role: 'system' | 'user' | 'assistant';
132
+ content: string;
133
+ }
134
+ /** Options for TextGenerator.generate(). */
135
+ interface GenerateOptions extends RunOptions {
136
+ /** Upper bound on generated tokens; default 256. */
137
+ maxNewTokens?: number;
138
+ /** Sample instead of greedy decoding; set with temperature / topP / topK. */
139
+ doSample?: boolean;
140
+ temperature?: number;
141
+ topP?: number;
142
+ topK?: number;
143
+ repetitionPenalty?: number;
144
+ /** Called with each piece of text as it is generated. */
145
+ onToken?: (text: string) => void;
146
+ }
102
147
 
103
148
  /**
104
149
  * The callable returned by Transformers.js pipeline(), reduced to the
@@ -119,6 +164,12 @@ type PipelineFactory = (task: string, model: string | undefined, options: Record
119
164
  */
120
165
  interface TransformersModuleLike {
121
166
  pipeline(task: string, model?: string, options?: object): Promise<unknown>;
167
+ /** Streams generated text for the `onToken` run option; optional in test stubs. */
168
+ TextStreamer?: new (tokenizer: never, options: {
169
+ skip_prompt?: boolean;
170
+ skip_special_tokens?: boolean;
171
+ callback_function?: (text: string) => void;
172
+ }) => unknown;
122
173
  }
123
174
  /**
124
175
  * The factory behind PIPELINE_FACTORY. It imports @huggingface/transformers
@@ -149,6 +200,29 @@ declare const NGX_TRANSFORMERS_CONFIG: InjectionToken<NgxTransformersConfig>;
149
200
  * ```
150
201
  */
151
202
  declare function provideTransformers(config: NgxTransformersConfig): EnvironmentProviders;
203
+ /**
204
+ * Runs every pipeline in a Web Worker so inference never blocks the UI.
205
+ * The worker file imports the worker entry point:
206
+ *
207
+ * ```ts
208
+ * // transformers.worker.ts
209
+ * /// <reference lib="webworker" />
210
+ * import { runTransformersWorker } from 'ngx-transformers/worker';
211
+ * runTransformersWorker();
212
+ *
213
+ * // app.config.ts
214
+ * provideTransformersWorker(
215
+ * () => new Worker(new URL('./transformers.worker', import.meta.url), { type: 'module' }),
216
+ * )
217
+ * ```
218
+ *
219
+ * The worker is created on the first pipeline and terminated with the
220
+ * injector. Handles, signals and the wrappers work unchanged; results must
221
+ * survive structured cloning (plain objects, arrays, typed arrays and
222
+ * tensors do). A worker that fails to start rejects the calls waiting on
223
+ * it, like a failed load in-thread.
224
+ */
225
+ declare function provideTransformersWorker(createWorker: () => WorkerLike): EnvironmentProviders;
152
226
 
153
227
  /**
154
228
  * A lazily-loaded Transformers.js pipeline wrapped in signals.
@@ -163,18 +237,28 @@ declare class PipelineHandle<TIn = unknown, TOut = unknown> {
163
237
  private readonly config;
164
238
  /** idle -> loading -> ready <-> busy; error only on load failure. */
165
239
  readonly status: _angular_core.WritableSignal<PipelineStatus>;
166
- /** Download progress for the file currently transferring, else null. */
240
+ /** Download progress: the file reported last, plus the total over all files. */
167
241
  readonly progress: _angular_core.WritableSignal<ModelProgress | null>;
242
+ /** The last load error; cleared when a load starts. */
168
243
  readonly error: _angular_core.WritableSignal<unknown>;
244
+ /** The error of the most recently started run, if it failed; cleared when a run starts. */
245
+ readonly runError: _angular_core.WritableSignal<unknown>;
169
246
  /** True once the model is usable (including while a run is in flight). */
170
247
  readonly ready: _angular_core.Signal<boolean>;
171
248
  readonly busy: _angular_core.Signal<boolean>;
172
249
  private session;
173
250
  private destroyed;
251
+ /** Counts runs so only the most recently started one writes runError. */
252
+ private runSequence;
174
253
  constructor(request: PipelineRequest, factory: PipelineFactory, config: NgxTransformersConfig);
175
254
  /** Downloads and initializes the model. Idempotent; retries after error. */
176
255
  load(): Promise<void>;
177
- /** Runs the pipeline, loading the model first if needed. */
256
+ /**
257
+ * Runs the pipeline, loading the model first if needed. Besides the
258
+ * pipeline's own options, `runOptions.signal` (an AbortSignal) makes the
259
+ * run reject before it starts when the signal has already fired, which
260
+ * spares superseded runs the inference (see inferenceResource()).
261
+ */
178
262
  run(input: TIn, runOptions?: Record<string, unknown>): Promise<TOut>;
179
263
  /**
180
264
  * Runs the pipeline with extra positional arguments after the input, for
@@ -201,6 +285,12 @@ declare class PipelineHandle<TIn = unknown, TOut = unknown> {
201
285
  * unset or 'auto' device is probed for WebGPU (once, cached) at load time.
202
286
  */
203
287
  private resolveDevice;
288
+ /**
289
+ * Transformers.js reports each file on its own (initiate, download,
290
+ * progress, done). The session records every file it hears about; the
291
+ * signal is published on progress and done events only, so a file that
292
+ * has not transferred a byte never replaces the one that is moving.
293
+ */
204
294
  private onProgress;
205
295
  }
206
296
  /**
@@ -230,7 +320,7 @@ declare const DEFAULT_TEXT_CLASSIFICATION_MODEL = "Xenova/distilbert-base-uncase
230
320
  */
231
321
  declare class TextClassifier extends PipelineHandle<string, ClassificationResult[] | ClassificationResult[][]> {
232
322
  /** Classifies one text; resolves to labels sorted by score (top first). */
233
- classify(text: string, topK?: number): Promise<ClassificationResult[]>;
323
+ classify(text: string, topK?: number, options?: RunOptions): Promise<ClassificationResult[]>;
234
324
  }
235
325
  /** Creates a TextClassifier in an injection context. */
236
326
  declare function createTextClassifier(options?: Partial<Omit<PipelineRequest, 'task'>>): TextClassifier;
@@ -251,11 +341,11 @@ declare function cosineSimilarity(a: readonly number[], b: readonly number[]): n
251
341
  */
252
342
  declare class TextEmbedder extends PipelineHandle<string | string[], TensorLike> {
253
343
  /** Embeds one or many texts; always resolves to one vector per text. */
254
- embed(texts: string | string[]): Promise<number[][]>;
344
+ embed(texts: string | string[], options?: RunOptions): Promise<number[][]>;
255
345
  /** Cosine similarity of two texts in [-1, 1]. */
256
- similarity(a: string, b: string): Promise<number>;
346
+ similarity(a: string, b: string, options?: RunOptions): Promise<number>;
257
347
  /** Ranks documents against a query, most similar first. */
258
- rank(query: string, documents: string[]): Promise<RankedResult[]>;
348
+ rank(query: string, documents: string[], options?: RunOptions): Promise<RankedResult[]>;
259
349
  }
260
350
  /** Creates a TextEmbedder in an injection context. */
261
351
  declare function createTextEmbedder(options?: Partial<Omit<PipelineRequest, 'task'>>): TextEmbedder;
@@ -353,6 +443,7 @@ declare class Translator {
353
443
  readonly status: _angular_core.Signal<ngx_transformers.PipelineStatus>;
354
444
  readonly progress: _angular_core.Signal<ngx_transformers.ModelProgress | null>;
355
445
  readonly error: _angular_core.Signal<{} | null>;
446
+ readonly runError: _angular_core.Signal<{} | null>;
356
447
  readonly ready: _angular_core.Signal<boolean>;
357
448
  readonly busy: _angular_core.Signal<boolean>;
358
449
  constructor(options: TranslatorOptions, factory: PipelineFactory, config: NgxTransformersConfig);
@@ -380,6 +471,72 @@ declare class Translator {
380
471
  /** Creates a Translator in an injection context. */
381
472
  declare function createTranslator(options?: TranslatorOptions): Translator;
382
473
 
474
+ declare const DEFAULT_TEXT_GENERATION_MODEL = "HuggingFaceTB/SmolLM2-135M-Instruct";
475
+ /** Raw text-generation output: the reply as a string, or the chat with the reply appended. */
476
+ interface RawGeneration {
477
+ generated_text?: string | ChatMessage[];
478
+ }
479
+ /**
480
+ * Text generation with a small language model, streamed token by token.
481
+ * The default checkpoint is SmolLM2-135M-Instruct; swap `model` for any
482
+ * Transformers.js text-generation checkpoint (Qwen2.5, Llama, Phi...).
483
+ */
484
+ declare class TextGenerator extends PipelineHandle<string | ChatMessage[], RawGeneration | RawGeneration[]> {
485
+ /** The text generated so far by the latest generate() call; reset when a call starts. */
486
+ readonly output: _angular_core.WritableSignal<string>;
487
+ /** Only the most recently started call may write output. */
488
+ private generation;
489
+ private warnedNoStreaming;
490
+ /**
491
+ * Generates a reply to a prompt or a chat. Tokens stream into `output`
492
+ * (and `options.onToken`) while the model runs; resolves with the full,
493
+ * trimmed reply.
494
+ */
495
+ generate(prompt: string | ChatMessage[], options?: GenerateOptions): Promise<string>;
496
+ }
497
+ /**
498
+ * Creates a TextGenerator in an injection context; destroyed with the
499
+ * component. dtype defaults to 'q4', the size/quality sweet spot for small
500
+ * decoders on the WebAssembly runtime.
501
+ */
502
+ declare function createTextGenerator(options?: Partial<Omit<PipelineRequest, 'task'>>): TextGenerator;
503
+
504
+ interface InferenceResourceOptions<TIn, TOut> {
505
+ /**
506
+ * The input to run on, read reactively. Return `undefined` for "nothing
507
+ * to do": the resource stays idle and keeps no value.
508
+ */
509
+ input: () => TIn | undefined;
510
+ /**
511
+ * Runs the model for one input, typically a handle method. The abort
512
+ * signal fires when a newer input supersedes this run: pass it on as the
513
+ * method's `signal` option and a superseded run that has not started yet
514
+ * (it was waiting for the model to load) is skipped instead of queued.
515
+ */
516
+ run: (input: TIn, abortSignal: AbortSignal) => Promise<TOut>;
517
+ /** Wait this long after the last input change before running (typing). */
518
+ debounceMs?: number;
519
+ /** Required outside an injection context. */
520
+ injector?: Injector;
521
+ }
522
+ /**
523
+ * Runs inference whenever an input signal changes, as an Angular resource:
524
+ * `value()`, `isLoading()`, `error()` and `status()` are signals, and only
525
+ * the result for the latest input is kept.
526
+ *
527
+ * ```ts
528
+ * readonly text = signal('');
529
+ * readonly classifier = createTextClassifier();
530
+ * readonly sentiment = inferenceResource({
531
+ * input: () => this.text().trim() || undefined,
532
+ * run: (text, signal) => this.classifier.classify(text, 1, { signal }),
533
+ * debounceMs: 300,
534
+ * });
535
+ * // template: @if (sentiment.value(); as result) { {{ result[0].label }} }
536
+ * ```
537
+ */
538
+ declare function inferenceResource<TIn, TOut>(options: InferenceResourceOptions<TIn, TOut>): ResourceRef<TOut | undefined>;
539
+
383
540
  /** Minimal recorder surface - lets tests (and exotic runtimes) inject fakes. */
384
541
  interface RecorderLike {
385
542
  start(): void;
@@ -439,7 +596,8 @@ declare function createMicRecorder(deps?: MicRecorderDeps): MicRecorder;
439
596
 
440
597
  /**
441
598
  * Drop-in status line for a PipelineHandle: shows model download progress
442
- * while loading, then the ready/busy/error state. Themeable via CSS custom
599
+ * (over every file of the model) while loading, then the ready/busy/error
600
+ * state. Themeable via CSS custom
443
601
  * properties (--nt-accent, --nt-ink, --nt-muted, --nt-track).
444
602
  *
445
603
  * ```html
@@ -451,10 +609,12 @@ declare class ModelProgressComponent {
451
609
  progress: _angular_core.InputSignal<ModelProgress | null>;
452
610
  /** Labels per status; override to localize. */
453
611
  labels: _angular_core.InputSignal<Partial<Record<PipelineStatus, string>>>;
612
+ /** The bar follows the whole download when the handle reports it, else the current file. */
613
+ percent(progress: ModelProgress): number;
454
614
  readonly label: _angular_core.Signal<string>;
455
615
  static ɵfac: _angular_core.ɵɵFactoryDeclaration<ModelProgressComponent, never>;
456
616
  static ɵcmp: _angular_core.ɵɵComponentDeclaration<ModelProgressComponent, "ngx-model-progress", never, { "status": { "alias": "status"; "required": true; "isSignal": true; }; "progress": { "alias": "progress"; "required": false; "isSignal": true; }; "labels": { "alias": "labels"; "required": false; "isSignal": true; }; }, {}, never, never, true, never>;
457
617
  }
458
618
 
459
- export { DEFAULT_ASR_MODEL, DEFAULT_EMBEDDING_MODEL, DEFAULT_TEXT_CLASSIFICATION_MODEL, DEFAULT_ZERO_SHOT_MODEL, MicRecorder, ModelProgressComponent, NGX_TRANSFORMERS_CONFIG, PIPELINE_FACTORY, PipelineHandle, SpeechRecognizer, TextClassifier, TextEmbedder, Translator, WHISPER_SAMPLE_RATE, ZeroShotClassifier, cosineSimilarity, createDefaultPipelineFactory, createMicRecorder, createPipeline, createSpeechRecognizer, createTextClassifier, createTextEmbedder, createTranslator, createZeroShotClassifier, decodeAudio, defaultTranslationModel, detectDevice, hasWebGpu, provideTransformers, resetDeviceDetection, resolveTranslationModel };
460
- export type { ClassificationResult, MicRecorderDeps, ModelProgress, NgxTransformersConfig, PipelineFactory, PipelineLike, PipelineRequest, PipelineStatus, RankedResult, RecorderLike, TranscribeOptions, Transcription, TranscriptionChunk, TransformersDevice, TransformersDtype, TransformersModuleLike, TranslateOptions, TranslationHandle, TranslatorOptions, ZeroShotOptions };
619
+ export { DEFAULT_ASR_MODEL, DEFAULT_EMBEDDING_MODEL, DEFAULT_TEXT_CLASSIFICATION_MODEL, DEFAULT_TEXT_GENERATION_MODEL, DEFAULT_ZERO_SHOT_MODEL, MicRecorder, ModelProgressComponent, NGX_TRANSFORMERS_CONFIG, PIPELINE_FACTORY, PipelineHandle, SpeechRecognizer, TextClassifier, TextEmbedder, TextGenerator, Translator, WHISPER_SAMPLE_RATE, ZeroShotClassifier, cosineSimilarity, createDefaultPipelineFactory, createMicRecorder, createPipeline, createSpeechRecognizer, createTextClassifier, createTextEmbedder, createTextGenerator, createTranslator, createZeroShotClassifier, decodeAudio, defaultTranslationModel, detectDevice, hasWebGpu, inferenceResource, provideTransformers, provideTransformersWorker, resetDeviceDetection, resolveTranslationModel };
620
+ export type { ChatMessage, ClassificationResult, GenerateOptions, InferenceResourceOptions, MicRecorderDeps, ModelProgress, NgxTransformersConfig, OverallProgress, PipelineFactory, PipelineLike, PipelineRequest, PipelineStatus, RankedResult, RecorderLike, RunOptions, TranscribeOptions, Transcription, TranscriptionChunk, TransformersDevice, TransformersDtype, TransformersModuleLike, TranslateOptions, TranslationHandle, TranslatorOptions, ZeroShotOptions };