ngx-transformers 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,6 @@
1
1
  import * as _angular_core from '@angular/core';
2
- import { InjectionToken, EnvironmentProviders } from '@angular/core';
2
+ import { InjectionToken, EnvironmentProviders, Injector, ResourceRef } from '@angular/core';
3
+ import { WorkerLike } from 'ngx-transformers/worker';
3
4
  import * as ngx_transformers from 'ngx-transformers';
4
5
 
5
6
  /** Lifecycle of a model pipeline. `error` is terminal until load() is retried. */
@@ -12,9 +13,27 @@ interface ModelProgress {
12
13
  progress: number;
13
14
  loadedBytes: number;
14
15
  totalBytes: number;
16
+ /** Progress over every file of the model seen so far; absent when unknown. */
17
+ overall?: OverallProgress;
18
+ }
19
+ /**
20
+ * Download progress summed over the files of a model seen so far. A model
21
+ * is several files (config, tokenizer, weights) fetched in parallel; this is
22
+ * the steady number to put on a progress bar.
23
+ */
24
+ interface OverallProgress {
25
+ /** 0-100 over the bytes of every file with a known size. */
26
+ progress: number;
27
+ loadedBytes: number;
28
+ totalBytes: number;
29
+ /** Files seen so far. */
30
+ files: number;
31
+ /** Files fully downloaded. */
32
+ filesDone: number;
15
33
  }
16
34
  type TransformersDevice = 'wasm' | 'webgpu' | 'auto';
17
- type TransformersDtype = 'fp32' | 'fp16' | 'q8' | 'q4';
35
+ /** Weight formats Transformers.js can load; a checkpoint must ship the one you ask for. */
36
+ type TransformersDtype = 'fp32' | 'fp16' | 'q8' | 'int8' | 'uint8' | 'q4' | 'bnb4' | 'q4f16';
18
37
  /** Global defaults applied to every pipeline; see provideTransformers(). */
19
38
  interface NgxTransformersConfig {
20
39
  device?: TransformersDevice;
@@ -41,6 +60,14 @@ interface PipelineRequest {
41
60
  dtype?: TransformersDtype;
42
61
  options?: Record<string, unknown>;
43
62
  }
63
+ /**
64
+ * Options every run accepts. `signal` makes a run reject with an AbortError
65
+ * before it starts when the signal has already fired; a model run cannot be
66
+ * interrupted once started, but a superseded run need not begin.
67
+ */
68
+ interface RunOptions {
69
+ signal?: AbortSignal;
70
+ }
44
71
  interface ClassificationResult {
45
72
  label: string;
46
73
  score: number;
@@ -65,7 +92,7 @@ interface Transcription {
65
92
  /** Present when transcribe() was called with returnTimestamps. */
66
93
  chunks?: TranscriptionChunk[];
67
94
  }
68
- interface TranscribeOptions {
95
+ interface TranscribeOptions extends RunOptions {
69
96
  /** true for segment timestamps, 'word' for word-level. */
70
97
  returnTimestamps?: boolean | 'word';
71
98
  /** Split audio longer than ~30 s into chunks of this many seconds. */
@@ -78,7 +105,7 @@ interface TranscribeOptions {
78
105
  task?: 'transcribe' | 'translate';
79
106
  }
80
107
  /** Options for ZeroShotClassifier.classify(). */
81
- interface ZeroShotOptions {
108
+ interface ZeroShotOptions extends RunOptions {
82
109
  /** Score every label on its own (several can be high) instead of picking one. */
83
110
  multiLabel?: boolean;
84
111
  /** NLI hypothesis with a {} placeholder for the label; default "This example is {}.". */
@@ -88,7 +115,7 @@ interface ZeroShotOptions {
88
115
  * Language pair for one translate() call. Codes follow the checkpoint:
89
116
  * ISO 639-1 for opus-mt ("en", "ru"), FLORES-200 for NLLB ("eng_Latn").
90
117
  */
91
- interface TranslateOptions {
118
+ interface TranslateOptions extends RunOptions {
92
119
  from?: string;
93
120
  to?: string;
94
121
  }
@@ -99,6 +126,24 @@ interface TranslatorOptions extends Partial<Omit<PipelineRequest, 'task'>> {
99
126
  /** Default target language for translate(). */
100
127
  to?: string;
101
128
  }
129
+ /** One turn of a chat prompt for TextGenerator. */
130
+ interface ChatMessage {
131
+ role: 'system' | 'user' | 'assistant';
132
+ content: string;
133
+ }
134
+ /** Options for TextGenerator.generate(). */
135
+ interface GenerateOptions extends RunOptions {
136
+ /** Upper bound on generated tokens; default 256. */
137
+ maxNewTokens?: number;
138
+ /** Sample instead of greedy decoding; set with temperature / topP / topK. */
139
+ doSample?: boolean;
140
+ temperature?: number;
141
+ topP?: number;
142
+ topK?: number;
143
+ repetitionPenalty?: number;
144
+ /** Called with each piece of text as it is generated. */
145
+ onToken?: (text: string) => void;
146
+ }
102
147
 
103
148
  /**
104
149
  * The callable returned by Transformers.js pipeline(), reduced to the
@@ -112,6 +157,37 @@ type PipelineLike = ((input: unknown, options?: Record<string, unknown>) => Prom
112
157
  * @huggingface/transformers; tests and SSR shims can provide their own.
113
158
  */
114
159
  type PipelineFactory = (task: string, model: string | undefined, options: Record<string, unknown>) => Promise<PipelineLike>;
160
+ /**
161
+ * The slice of the @huggingface/transformers module the default factory
162
+ * uses. The real module satisfies it, so an importer can configure
163
+ * `env` and return the module as is.
164
+ */
165
+ interface TransformersModuleLike {
166
+ pipeline(task: string, model?: string, options?: object): Promise<unknown>;
167
+ /** Streams generated text for the `onToken` run option; optional in test stubs. */
168
+ TextStreamer?: new (tokenizer: never, options: {
169
+ skip_prompt?: boolean;
170
+ skip_special_tokens?: boolean;
171
+ callback_function?: (text: string) => void;
172
+ }) => unknown;
173
+ }
174
+ /**
175
+ * The factory behind PIPELINE_FACTORY. It imports @huggingface/transformers
176
+ * lazily, on the first pipeline, so the library adds nothing to the initial
177
+ * bundle. Wrap it to add options or logging while keeping the lazy import:
178
+ *
179
+ * ```ts
180
+ * const base = createDefaultPipelineFactory();
181
+ * const factory: PipelineFactory = (task, model, options) =>
182
+ * base(task, model, { ...options, revision: 'v2' });
183
+ * providers: [{ provide: PIPELINE_FACTORY, useValue: factory }]
184
+ * ```
185
+ *
186
+ * `load` is the importer. Tests pass a stub module; apps that need to
187
+ * configure Transformers.js (`env.allowRemoteModels`, `env.localModelPath`)
188
+ * do it there, before returning the module.
189
+ */
190
+ declare function createDefaultPipelineFactory(load?: () => Promise<TransformersModuleLike>): PipelineFactory;
115
191
  declare const PIPELINE_FACTORY: InjectionToken<PipelineFactory>;
116
192
  declare const NGX_TRANSFORMERS_CONFIG: InjectionToken<NgxTransformersConfig>;
117
193
  /**
@@ -124,6 +200,29 @@ declare const NGX_TRANSFORMERS_CONFIG: InjectionToken<NgxTransformersConfig>;
124
200
  * ```
125
201
  */
126
202
  declare function provideTransformers(config: NgxTransformersConfig): EnvironmentProviders;
203
+ /**
204
+ * Runs every pipeline in a Web Worker so inference never blocks the UI.
205
+ * The worker file imports the worker entry point:
206
+ *
207
+ * ```ts
208
+ * // transformers.worker.ts
209
+ * /// <reference lib="webworker" />
210
+ * import { runTransformersWorker } from 'ngx-transformers/worker';
211
+ * runTransformersWorker();
212
+ *
213
+ * // app.config.ts
214
+ * provideTransformersWorker(
215
+ * () => new Worker(new URL('./transformers.worker', import.meta.url), { type: 'module' }),
216
+ * )
217
+ * ```
218
+ *
219
+ * The worker is created on the first pipeline and terminated with the
220
+ * injector. Handles, signals and the wrappers work unchanged; results must
221
+ * survive structured cloning (plain objects, arrays, typed arrays and
222
+ * tensors do). A worker that fails to start rejects the calls waiting on
223
+ * it, like a failed load in-thread.
224
+ */
225
+ declare function provideTransformersWorker(createWorker: () => WorkerLike): EnvironmentProviders;
127
226
 
128
227
  /**
129
228
  * A lazily-loaded Transformers.js pipeline wrapped in signals.
@@ -138,18 +237,28 @@ declare class PipelineHandle<TIn = unknown, TOut = unknown> {
138
237
  private readonly config;
139
238
  /** idle -> loading -> ready <-> busy; error only on load failure. */
140
239
  readonly status: _angular_core.WritableSignal<PipelineStatus>;
141
- /** Download progress for the file currently transferring, else null. */
240
+ /** Download progress: the file reported last, plus the total over all files. */
142
241
  readonly progress: _angular_core.WritableSignal<ModelProgress | null>;
242
+ /** The last load error; cleared when a load starts. */
143
243
  readonly error: _angular_core.WritableSignal<unknown>;
244
+ /** The error of the most recently started run, if it failed; cleared when a run starts. */
245
+ readonly runError: _angular_core.WritableSignal<unknown>;
144
246
  /** True once the model is usable (including while a run is in flight). */
145
247
  readonly ready: _angular_core.Signal<boolean>;
146
248
  readonly busy: _angular_core.Signal<boolean>;
147
- private pipe;
148
- private loading;
249
+ private session;
250
+ private destroyed;
251
+ /** Counts runs so only the most recently started one writes runError. */
252
+ private runSequence;
149
253
  constructor(request: PipelineRequest, factory: PipelineFactory, config: NgxTransformersConfig);
150
254
  /** Downloads and initializes the model. Idempotent; retries after error. */
151
255
  load(): Promise<void>;
152
- /** Runs the pipeline, loading the model first if needed. */
256
+ /**
257
+ * Runs the pipeline, loading the model first if needed. Besides the
258
+ * pipeline's own options, `runOptions.signal` (an AbortSignal) makes the
259
+ * run reject before it starts when the signal has already fired, which
260
+ * spares superseded runs the inference (see inferenceResource()).
261
+ */
153
262
  run(input: TIn, runOptions?: Record<string, unknown>): Promise<TOut>;
154
263
  /**
155
264
  * Runs the pipeline with extra positional arguments after the input, for
@@ -157,19 +266,36 @@ declare class PipelineHandle<TIn = unknown, TOut = unknown> {
157
266
  * classification takes (text, candidateLabels, options).
158
267
  */
159
268
  protected runWith(input: TIn, ...extraArgs: unknown[]): Promise<TOut>;
160
- /** Frees the model. The handle can be loaded again afterwards. */
269
+ /**
270
+ * Frees the model. A download still in flight is cancelled: its model is
271
+ * released on arrival and the handle stays idle. The handle can be loaded
272
+ * again afterwards; destroy() is the terminal variant.
273
+ */
161
274
  dispose(): Promise<void>;
275
+ /**
276
+ * dispose() for good: later load() and run() calls reject instead of
277
+ * downloading a model nobody would release. create*() registers this with
278
+ * the DestroyRef of the injection context, so a handle declared in a
279
+ * component ends with the component.
280
+ */
281
+ destroy(): Promise<void>;
162
282
  private doLoad;
163
283
  /**
164
284
  * An explicit request or global device wins; with `autoDevice` on, an
165
285
  * unset or 'auto' device is probed for WebGPU (once, cached) at load time.
166
286
  */
167
287
  private resolveDevice;
288
+ /**
289
+ * Transformers.js reports each file on its own (initiate, download,
290
+ * progress, done). The session records every file it hears about; the
291
+ * signal is published on progress and done events only, so a file that
292
+ * has not transferred a byte never replaces the one that is moving.
293
+ */
168
294
  private onProgress;
169
295
  }
170
296
  /**
171
297
  * Creates a PipelineHandle in an injection context (constructor, field
172
- * initializer, or runInInjectionContext). The handle is disposed with the
298
+ * initializer, or runInInjectionContext). The handle is destroyed with the
173
299
  * surrounding component/injector.
174
300
  */
175
301
  declare function createPipeline<TIn = unknown, TOut = unknown>(request: PipelineRequest): PipelineHandle<TIn, TOut>;
@@ -194,7 +320,7 @@ declare const DEFAULT_TEXT_CLASSIFICATION_MODEL = "Xenova/distilbert-base-uncase
194
320
  */
195
321
  declare class TextClassifier extends PipelineHandle<string, ClassificationResult[] | ClassificationResult[][]> {
196
322
  /** Classifies one text; resolves to labels sorted by score (top first). */
197
- classify(text: string, topK?: number): Promise<ClassificationResult[]>;
323
+ classify(text: string, topK?: number, options?: RunOptions): Promise<ClassificationResult[]>;
198
324
  }
199
325
  /** Creates a TextClassifier in an injection context. */
200
326
  declare function createTextClassifier(options?: Partial<Omit<PipelineRequest, 'task'>>): TextClassifier;
@@ -215,11 +341,11 @@ declare function cosineSimilarity(a: readonly number[], b: readonly number[]): n
215
341
  */
216
342
  declare class TextEmbedder extends PipelineHandle<string | string[], TensorLike> {
217
343
  /** Embeds one or many texts; always resolves to one vector per text. */
218
- embed(texts: string | string[]): Promise<number[][]>;
344
+ embed(texts: string | string[], options?: RunOptions): Promise<number[][]>;
219
345
  /** Cosine similarity of two texts in [-1, 1]. */
220
- similarity(a: string, b: string): Promise<number>;
346
+ similarity(a: string, b: string, options?: RunOptions): Promise<number>;
221
347
  /** Ranks documents against a query, most similar first. */
222
- rank(query: string, documents: string[]): Promise<RankedResult[]>;
348
+ rank(query: string, documents: string[], options?: RunOptions): Promise<RankedResult[]>;
223
349
  }
224
350
  /** Creates a TextEmbedder in an injection context. */
225
351
  declare function createTextEmbedder(options?: Partial<Omit<PipelineRequest, 'task'>>): TextEmbedder;
@@ -313,9 +439,11 @@ declare class Translator {
313
439
  private readonly config;
314
440
  private readonly active;
315
441
  private readonly handles;
442
+ private destroyed;
316
443
  readonly status: _angular_core.Signal<ngx_transformers.PipelineStatus>;
317
444
  readonly progress: _angular_core.Signal<ngx_transformers.ModelProgress | null>;
318
445
  readonly error: _angular_core.Signal<{} | null>;
446
+ readonly runError: _angular_core.Signal<{} | null>;
319
447
  readonly ready: _angular_core.Signal<boolean>;
320
448
  readonly busy: _angular_core.Signal<boolean>;
321
449
  constructor(options: TranslatorOptions, factory: PipelineFactory, config: NgxTransformersConfig);
@@ -330,6 +458,11 @@ declare class Translator {
330
458
  * keep several pairs warm and want a progress line per model.
331
459
  */
332
460
  handleFor(pair?: TranslateOptions): TranslationHandle;
461
+ /**
462
+ * dispose() for good: later calls reject instead of loading a model nobody
463
+ * would release. createTranslator() registers this with the DestroyRef.
464
+ */
465
+ destroy(): Promise<void>;
333
466
  /** Frees every model. The translator can be used again afterwards. */
334
467
  dispose(): Promise<void>;
335
468
  /** src_lang / tgt_lang for multilingual checkpoints; opus-mt models get nothing. */
@@ -338,6 +471,72 @@ declare class Translator {
338
471
  /** Creates a Translator in an injection context. */
339
472
  declare function createTranslator(options?: TranslatorOptions): Translator;
340
473
 
474
+ declare const DEFAULT_TEXT_GENERATION_MODEL = "HuggingFaceTB/SmolLM2-135M-Instruct";
475
+ /** Raw text-generation output: the reply as a string, or the chat with the reply appended. */
476
+ interface RawGeneration {
477
+ generated_text?: string | ChatMessage[];
478
+ }
479
+ /**
480
+ * Text generation with a small language model, streamed token by token.
481
+ * The default checkpoint is SmolLM2-135M-Instruct; swap `model` for any
482
+ * Transformers.js text-generation checkpoint (Qwen2.5, Llama, Phi...).
483
+ */
484
+ declare class TextGenerator extends PipelineHandle<string | ChatMessage[], RawGeneration | RawGeneration[]> {
485
+ /** The text generated so far by the latest generate() call; reset when a call starts. */
486
+ readonly output: _angular_core.WritableSignal<string>;
487
+ /** Only the most recently started call may write output. */
488
+ private generation;
489
+ private warnedNoStreaming;
490
+ /**
491
+ * Generates a reply to a prompt or a chat. Tokens stream into `output`
492
+ * (and `options.onToken`) while the model runs; resolves with the full,
493
+ * trimmed reply.
494
+ */
495
+ generate(prompt: string | ChatMessage[], options?: GenerateOptions): Promise<string>;
496
+ }
497
+ /**
498
+ * Creates a TextGenerator in an injection context; destroyed with the
499
+ * component. dtype defaults to 'q4', the size/quality sweet spot for small
500
+ * decoders on the WebAssembly runtime.
501
+ */
502
+ declare function createTextGenerator(options?: Partial<Omit<PipelineRequest, 'task'>>): TextGenerator;
503
+
504
+ interface InferenceResourceOptions<TIn, TOut> {
505
+ /**
506
+ * The input to run on, read reactively. Return `undefined` for "nothing
507
+ * to do": the resource stays idle and keeps no value.
508
+ */
509
+ input: () => TIn | undefined;
510
+ /**
511
+ * Runs the model for one input, typically a handle method. The abort
512
+ * signal fires when a newer input supersedes this run: pass it on as the
513
+ * method's `signal` option and a superseded run that has not started yet
514
+ * (it was waiting for the model to load) is skipped instead of queued.
515
+ */
516
+ run: (input: TIn, abortSignal: AbortSignal) => Promise<TOut>;
517
+ /** Wait this long after the last input change before running (typing). */
518
+ debounceMs?: number;
519
+ /** Required outside an injection context. */
520
+ injector?: Injector;
521
+ }
522
+ /**
523
+ * Runs inference whenever an input signal changes, as an Angular resource:
524
+ * `value()`, `isLoading()`, `error()` and `status()` are signals, and only
525
+ * the result for the latest input is kept.
526
+ *
527
+ * ```ts
528
+ * readonly text = signal('');
529
+ * readonly classifier = createTextClassifier();
530
+ * readonly sentiment = inferenceResource({
531
+ * input: () => this.text().trim() || undefined,
532
+ * run: (text, signal) => this.classifier.classify(text, 1, { signal }),
533
+ * debounceMs: 300,
534
+ * });
535
+ * // template: @if (sentiment.value(); as result) { {{ result[0].label }} }
536
+ * ```
537
+ */
538
+ declare function inferenceResource<TIn, TOut>(options: InferenceResourceOptions<TIn, TOut>): ResourceRef<TOut | undefined>;
539
+
341
540
  /** Minimal recorder surface - lets tests (and exotic runtimes) inject fakes. */
342
541
  interface RecorderLike {
343
542
  start(): void;
@@ -397,7 +596,8 @@ declare function createMicRecorder(deps?: MicRecorderDeps): MicRecorder;
397
596
 
398
597
  /**
399
598
  * Drop-in status line for a PipelineHandle: shows model download progress
400
- * while loading, then the ready/busy/error state. Themeable via CSS custom
599
+ * (over every file of the model) while loading, then the ready/busy/error
600
+ * state. Themeable via CSS custom
401
601
  * properties (--nt-accent, --nt-ink, --nt-muted, --nt-track).
402
602
  *
403
603
  * ```html
@@ -409,10 +609,12 @@ declare class ModelProgressComponent {
409
609
  progress: _angular_core.InputSignal<ModelProgress | null>;
410
610
  /** Labels per status; override to localize. */
411
611
  labels: _angular_core.InputSignal<Partial<Record<PipelineStatus, string>>>;
612
+ /** The bar follows the whole download when the handle reports it, else the current file. */
613
+ percent(progress: ModelProgress): number;
412
614
  readonly label: _angular_core.Signal<string>;
413
615
  static ɵfac: _angular_core.ɵɵFactoryDeclaration<ModelProgressComponent, never>;
414
616
  static ɵcmp: _angular_core.ɵɵComponentDeclaration<ModelProgressComponent, "ngx-model-progress", never, { "status": { "alias": "status"; "required": true; "isSignal": true; }; "progress": { "alias": "progress"; "required": false; "isSignal": true; }; "labels": { "alias": "labels"; "required": false; "isSignal": true; }; }, {}, never, never, true, never>;
415
617
  }
416
618
 
417
- export { DEFAULT_ASR_MODEL, DEFAULT_EMBEDDING_MODEL, DEFAULT_TEXT_CLASSIFICATION_MODEL, DEFAULT_ZERO_SHOT_MODEL, MicRecorder, ModelProgressComponent, NGX_TRANSFORMERS_CONFIG, PIPELINE_FACTORY, PipelineHandle, SpeechRecognizer, TextClassifier, TextEmbedder, Translator, WHISPER_SAMPLE_RATE, ZeroShotClassifier, cosineSimilarity, createMicRecorder, createPipeline, createSpeechRecognizer, createTextClassifier, createTextEmbedder, createTranslator, createZeroShotClassifier, decodeAudio, defaultTranslationModel, detectDevice, hasWebGpu, provideTransformers, resetDeviceDetection, resolveTranslationModel };
418
- export type { ClassificationResult, MicRecorderDeps, ModelProgress, NgxTransformersConfig, PipelineFactory, PipelineLike, PipelineRequest, PipelineStatus, RankedResult, RecorderLike, TranscribeOptions, Transcription, TranscriptionChunk, TransformersDevice, TransformersDtype, TranslateOptions, TranslationHandle, TranslatorOptions, ZeroShotOptions };
619
+ export { DEFAULT_ASR_MODEL, DEFAULT_EMBEDDING_MODEL, DEFAULT_TEXT_CLASSIFICATION_MODEL, DEFAULT_TEXT_GENERATION_MODEL, DEFAULT_ZERO_SHOT_MODEL, MicRecorder, ModelProgressComponent, NGX_TRANSFORMERS_CONFIG, PIPELINE_FACTORY, PipelineHandle, SpeechRecognizer, TextClassifier, TextEmbedder, TextGenerator, Translator, WHISPER_SAMPLE_RATE, ZeroShotClassifier, cosineSimilarity, createDefaultPipelineFactory, createMicRecorder, createPipeline, createSpeechRecognizer, createTextClassifier, createTextEmbedder, createTextGenerator, createTranslator, createZeroShotClassifier, decodeAudio, defaultTranslationModel, detectDevice, hasWebGpu, inferenceResource, provideTransformers, provideTransformersWorker, resetDeviceDetection, resolveTranslationModel };
620
+ export type { ChatMessage, ClassificationResult, GenerateOptions, InferenceResourceOptions, MicRecorderDeps, ModelProgress, NgxTransformersConfig, OverallProgress, PipelineFactory, PipelineLike, PipelineRequest, PipelineStatus, RankedResult, RecorderLike, RunOptions, TranscribeOptions, Transcription, TranscriptionChunk, TransformersDevice, TransformersDtype, TransformersModuleLike, TranslateOptions, TranslationHandle, TranslatorOptions, ZeroShotOptions };