@memberjunction/ai-prompts 6.2.0-edge.0 → 6.2.0-edge.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -0
- package/dist/AIModelRunner.d.ts +9 -88
- package/dist/AIModelRunner.d.ts.map +1 -1
- package/dist/AIModelRunner.js +11 -281
- package/dist/AIModelRunner.js.map +1 -1
- package/dist/AIPromptRunner.d.ts +147 -513
- package/dist/AIPromptRunner.d.ts.map +1 -1
- package/dist/AIPromptRunner.js +460 -2126
- package/dist/AIPromptRunner.js.map +1 -1
- package/dist/BaseModelRunner.d.ts +618 -0
- package/dist/BaseModelRunner.d.ts.map +1 -0
- package/dist/BaseModelRunner.js +1933 -0
- package/dist/BaseModelRunner.js.map +1 -0
- package/dist/ExecutionPlanner.d.ts +2 -0
- package/dist/ExecutionPlanner.d.ts.map +1 -1
- package/dist/ExecutionPlanner.js +5 -1
- package/dist/ExecutionPlanner.js.map +1 -1
- package/dist/ParallelExecution.d.ts +24 -5
- package/dist/ParallelExecution.d.ts.map +1 -1
- package/dist/ParallelExecutionCoordinator.d.ts +2 -11
- package/dist/ParallelExecutionCoordinator.d.ts.map +1 -1
- package/dist/ParallelExecutionCoordinator.js +104 -91
- package/dist/ParallelExecutionCoordinator.js.map +1 -1
- package/dist/audio/AISpeechToTextRunner.d.ts +46 -0
- package/dist/audio/AISpeechToTextRunner.d.ts.map +1 -0
- package/dist/audio/AISpeechToTextRunner.js +93 -0
- package/dist/audio/AISpeechToTextRunner.js.map +1 -0
- package/dist/audio/AITextToSpeechRunner.d.ts +43 -0
- package/dist/audio/AITextToSpeechRunner.d.ts.map +1 -0
- package/dist/audio/AITextToSpeechRunner.js +80 -0
- package/dist/audio/AITextToSpeechRunner.js.map +1 -0
- package/dist/audio/audio-description.d.ts +22 -0
- package/dist/audio/audio-description.d.ts.map +1 -0
- package/dist/audio/audio-description.js +52 -0
- package/dist/audio/audio-description.js.map +1 -0
- package/dist/audio/audio-runner.types.d.ts +46 -0
- package/dist/audio/audio-runner.types.d.ts.map +1 -0
- package/dist/audio/audio-runner.types.js +7 -0
- package/dist/audio/audio-runner.types.js.map +1 -0
- package/dist/decision/AIDecisionRunner.d.ts +134 -0
- package/dist/decision/AIDecisionRunner.d.ts.map +1 -0
- package/dist/decision/AIDecisionRunner.js +498 -0
- package/dist/decision/AIDecisionRunner.js.map +1 -0
- package/dist/decision/LLMDecision.d.ts +158 -0
- package/dist/decision/LLMDecision.d.ts.map +1 -0
- package/dist/decision/LLMDecision.js +493 -0
- package/dist/decision/LLMDecision.js.map +1 -0
- package/dist/decision/decision-questions.d.ts +28 -0
- package/dist/decision/decision-questions.d.ts.map +1 -0
- package/dist/decision/decision-questions.js +123 -0
- package/dist/decision/decision-questions.js.map +1 -0
- package/dist/decision/decision-runner.types.d.ts +29 -0
- package/dist/decision/decision-runner.types.d.ts.map +1 -0
- package/dist/decision/decision-runner.types.js +14 -0
- package/dist/decision/decision-runner.types.js.map +1 -0
- package/dist/embedding/AIEmbeddingRunner.d.ts +115 -0
- package/dist/embedding/AIEmbeddingRunner.d.ts.map +1 -0
- package/dist/embedding/AIEmbeddingRunner.js +462 -0
- package/dist/embedding/AIEmbeddingRunner.js.map +1 -0
- package/dist/embedding/embedding-runner.types.d.ts +72 -0
- package/dist/embedding/embedding-runner.types.d.ts.map +1 -0
- package/dist/embedding/embedding-runner.types.js +2 -0
- package/dist/embedding/embedding-runner.types.js.map +1 -0
- package/dist/image/AIImageGenerationRunner.d.ts +101 -0
- package/dist/image/AIImageGenerationRunner.d.ts.map +1 -0
- package/dist/image/AIImageGenerationRunner.js +355 -0
- package/dist/image/AIImageGenerationRunner.js.map +1 -0
- package/dist/image/image-runner.types.d.ts +90 -0
- package/dist/image/image-runner.types.d.ts.map +1 -0
- package/dist/image/image-runner.types.js +7 -0
- package/dist/image/image-runner.types.js.map +1 -0
- package/dist/index.d.ts +17 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +16 -0
- package/dist/index.js.map +1 -1
- package/dist/linearTextScan.d.ts +20 -0
- package/dist/linearTextScan.d.ts.map +1 -0
- package/dist/linearTextScan.js +74 -0
- package/dist/linearTextScan.js.map +1 -0
- package/dist/media/BaseMediaRunner.d.ts +143 -0
- package/dist/media/BaseMediaRunner.d.ts.map +1 -0
- package/dist/media/BaseMediaRunner.js +448 -0
- package/dist/media/BaseMediaRunner.js.map +1 -0
- package/dist/media/media-runner.types.d.ts +128 -0
- package/dist/media/media-runner.types.d.ts.map +1 -0
- package/dist/media/media-runner.types.js +8 -0
- package/dist/media/media-runner.types.js.map +1 -0
- package/dist/media/media-usage.d.ts +13 -0
- package/dist/media/media-usage.d.ts.map +1 -0
- package/dist/media/media-usage.js +18 -0
- package/dist/media/media-usage.js.map +1 -0
- package/dist/video/AIVideoRunner.d.ts +46 -0
- package/dist/video/AIVideoRunner.d.ts.map +1 -0
- package/dist/video/AIVideoRunner.js +88 -0
- package/dist/video/AIVideoRunner.js.map +1 -0
- package/dist/video/video-runner.types.d.ts +29 -0
- package/dist/video/video-runner.types.d.ts.map +1 -0
- package/dist/video/video-runner.types.js +7 -0
- package/dist/video/video-runner.types.js.map +1 -0
- package/package.json +12 -12
package/README.md
CHANGED
|
@@ -110,6 +110,8 @@ Execution order:
|
|
|
110
110
|
3. Child results replace placeholders in parent template
|
|
111
111
|
4. Final composed prompt executes as a single LLM call
|
|
112
112
|
|
|
113
|
+
**Rendering children without executing.** `AIPromptRunner.RenderChildPromptTemplates(childPrompts, params)` renders the child templates and returns `{ renderedTemplates }` keyed by each child's `parentPlaceholder`, with no model call. A caller that needs the rendered text before execution — the loop agent uses it to relocate a volatile specialization into its trailing runtime-state message — passes the map back as `AIPromptParams.PreRenderedChildTemplates`, and `ExecutePrompt` embeds those strings instead of rendering the children a second time. Rendering is deterministic for the same inputs, so both paths produce the same text.
|
|
114
|
+
|
|
113
115
|
### Model Selection Strategies
|
|
114
116
|
|
|
115
117
|
Three strategies for selecting which AI model executes a prompt:
|
|
@@ -200,6 +202,47 @@ Hierarchical credential resolution for API keys:
|
|
|
200
202
|
|
|
201
203
|
When a model fails due to rate limiting, authentication errors, or other transient issues, the runner can automatically retry with alternate models from the selection candidates.
|
|
202
204
|
|
|
205
|
+
### Media Runners
|
|
206
|
+
|
|
207
|
+
Non-chat models run through runners built on the same `BaseModelRunner`, so they get the same model selection, credential resolution, failover and `MJ: AI Prompt Runs` row as a chat prompt. Each requires one model type, and uses a default carrier prompt, found by name, unless the caller passes a `PromptID`. A `ModelID` pins the model, and failover then stays within its vendors. The run row never holds media bytes.
|
|
208
|
+
|
|
209
|
+
| Runner | Model type | Method | Default prompt | Usage recorded |
|
|
210
|
+
|---|---|---|---|---|
|
|
211
|
+
| `AIImageGenerationRunner` | `Image Generator` | `RunImageGeneration`, `RunImageEdit` | `Default Image Generation` | the driver's, else images returned (`Images`) |
|
|
212
|
+
| `AITextToSpeechRunner` | `TTS` | `RunTextToSpeech` | `Default Text To Speech` | the driver's, else characters sent (`Characters`) |
|
|
213
|
+
| `AISpeechToTextRunner` | `Speech to Text` | `RunSpeechToText` | `Default Speech To Text` | the driver's: audio seconds (`Seconds`) when reported, otherwise nothing |
|
|
214
|
+
| `AIVideoRunner` | `Video` | `RunAvatarVideo` | `Default Video Generation` | the driver's: video seconds (`Seconds`) when reported, otherwise nothing |
|
|
215
|
+
|
|
216
|
+
The text-to-speech, speech-to-text and video runners share their lifecycle through `BaseMediaRunner`. It follows the carrier prompt's `FailoverStrategy`, narrowing `SameModelDifferentVendor` to the selected model's vendors.
|
|
217
|
+
|
|
218
|
+
Whether a failed call fails over is `ErrorAnalyzer`'s decision, as it is for chat prompts. The shipped audio and video drivers report it on `SpeechResult.errorInfo` / `VideoResult.errorInfo`, from the error their SDK threw, so it keeps the HTTP status:
|
|
219
|
+
|
|
220
|
+
- a rate limit, an outage, a server error or a timeout fails over;
|
|
221
|
+
- a request the vendor rejected as invalid (a 400 or 422, such as an unknown voice or an unsupported audio format) does not, since every other candidate would reject it too;
|
|
222
|
+
- a 400 whose message reads as vendor-specific validation (`required`, `must be`, …) fails over to another vendor, as it does for chat;
|
|
223
|
+
- a 401 stops failover, as it does for chat.
|
|
224
|
+
|
|
225
|
+
A driver that reports only `errorMessage` is classified from the message, which recognizes rate limits, outages and a few malformed requests; any other message reads as `Unknown`, which fails over. A driver should set `errorInfo` with `ErrorAnalyzer.AnalyzeError(error)` on the error it caught.
|
|
226
|
+
|
|
227
|
+
`TimeoutMS` and `CancellationToken` bound each driver call as `timeoutMS` and `cancellationToken` bound a chat call. A call that exceeds `TimeoutMS` fails with an `AIPromptTimeoutError` and fails over. A cancelled call ends at once, never fails over, and its run row is recorded as `Cancelled`. The media drivers take no abort signal, so in both cases the request already sent is abandoned rather than torn down.
|
|
228
|
+
|
|
229
|
+
Video generation is asynchronous at the provider, and the primitive offers no status call, so `AIVideoRunner` does not wait for a render. A successful run means the provider accepted the request; the row records the video ID (for HeyGen, the render job's ID) and records the video's length only if a driver reports it.
|
|
230
|
+
|
|
231
|
+
Usage is recorded with `BaseModelRunner.ApplyUsageToRunRecord`, which writes non-token quantities as `InputUnitsUsed` / `OutputUnitsUsed` with the `MJ: AI Usage Types` row that names their measure. The runners never set a cost. The row's save prices it from the model's cost rows, and declines when no price unit type claims the measure, as none yet does for `Characters`.
|
|
232
|
+
|
|
233
|
+
```typescript
|
|
234
|
+
import { AITextToSpeechRunner } from '@memberjunction/ai-prompts';
|
|
235
|
+
|
|
236
|
+
const result = await new AITextToSpeechRunner().RunTextToSpeech({
|
|
237
|
+
text: 'Your report is ready.',
|
|
238
|
+
voice: 'alloy',
|
|
239
|
+
ContextUser: contextUser,
|
|
240
|
+
});
|
|
241
|
+
if (result.Success) {
|
|
242
|
+
const audio = result.SpeechResult?.data; // the run row describes the audio but never stores it
|
|
243
|
+
}
|
|
244
|
+
```
|
|
245
|
+
|
|
203
246
|
## Usage
|
|
204
247
|
|
|
205
248
|
### Basic Prompt Execution
|
package/dist/AIModelRunner.d.ts
CHANGED
|
@@ -1,109 +1,30 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
*/
|
|
5
|
-
export interface EmbeddingRunResult {
|
|
6
|
-
/** Whether the embedding call succeeded */
|
|
7
|
-
Success: boolean;
|
|
8
|
-
/** The embedding vectors (one per input text) */
|
|
9
|
-
Vectors: number[][];
|
|
10
|
-
/** The AIPromptRun ID created for tracking */
|
|
11
|
-
PromptRunID: string | null;
|
|
12
|
-
/** Total tokens used */
|
|
13
|
-
TokensUsed: number;
|
|
14
|
-
/** Cost of the call */
|
|
15
|
-
Cost: number;
|
|
16
|
-
/** Error message if failed */
|
|
17
|
-
ErrorMessage: string | null;
|
|
18
|
-
/** Execution time in milliseconds */
|
|
19
|
-
ExecutionTimeMs: number;
|
|
20
|
-
}
|
|
21
|
-
/**
|
|
22
|
-
* Parameters for running an embedding via AIModelRunner.
|
|
23
|
-
*/
|
|
24
|
-
export interface EmbeddingRunParams {
|
|
25
|
-
/** The texts to embed */
|
|
26
|
-
Texts: string[];
|
|
27
|
-
/** Optional: specific AIPrompt ID for this embedding operation (type=Embedding).
|
|
28
|
-
* If not provided, uses the first active Embedding prompt found. */
|
|
29
|
-
PromptID?: string;
|
|
30
|
-
/** Optional: specific model ID to use. If not provided, uses the prompt's model configuration. */
|
|
31
|
-
ModelID?: string;
|
|
32
|
-
/** The user context for permissions and audit */
|
|
33
|
-
ContextUser: UserInfo;
|
|
34
|
-
/** Optional: parent run ID (e.g., agent run, classification run) for hierarchical tracking */
|
|
35
|
-
ParentRunID?: string;
|
|
36
|
-
/** Optional: human-readable description for the AIPromptRun record */
|
|
37
|
-
Description?: string;
|
|
38
|
-
/**
|
|
39
|
-
* Optional: reduced embedding dimensions. Forwarded to the embedding provider's EmbedTexts call
|
|
40
|
-
* for models that support dimension reduction (e.g. OpenAI text-embedding-3-*); ignored by
|
|
41
|
-
* models that don't. The authoritative source is `MJ: Vector Indexes.Dimensions` — callers
|
|
42
|
-
* should read it from there and pass it here.
|
|
43
|
-
*/
|
|
44
|
-
Dimensions?: number;
|
|
45
|
-
}
|
|
1
|
+
import { IMetadataProvider } from '@memberjunction/core';
|
|
2
|
+
import { EmbeddingRunParams, EmbeddingRunResult } from './embedding/embedding-runner.types.js';
|
|
3
|
+
export type { EmbeddingRunResult, EmbeddingRunParams } from './embedding/embedding-runner.types.js';
|
|
46
4
|
/**
|
|
47
5
|
* AIModelRunner — Lightweight AI model execution tracker for non-LLM model types.
|
|
48
6
|
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
* infrastructure as AIPromptRunner but without template rendering or validation.
|
|
52
|
-
*
|
|
53
|
-
* Architecture:
|
|
54
|
-
* - @memberjunction/ai (no MJ infra): BaseEmbeddings, ModelUsage
|
|
55
|
-
* - @memberjunction/ai-core-plus (this class): AIModelRunner — creates AIPromptRun records
|
|
56
|
-
* - @memberjunction/ai-prompts: AIPromptRunner — LLM-specific (template rendering, validation)
|
|
57
|
-
*
|
|
58
|
-
* Usage:
|
|
59
|
-
* ```typescript
|
|
60
|
-
* const runner = new AIModelRunner();
|
|
61
|
-
* const result = await runner.RunEmbedding({
|
|
62
|
-
* Texts: ['Hello world', 'How are you'],
|
|
63
|
-
* ContextUser: currentUser,
|
|
64
|
-
* Description: 'Content item vectorization'
|
|
65
|
-
* });
|
|
66
|
-
* // result.PromptRunID links to the AIPromptRun record with token/cost data
|
|
67
|
-
* ```
|
|
7
|
+
* @deprecated Use {@link AIEmbeddingRunner} instead. AIModelRunner is retained for backward
|
|
8
|
+
* compatibility and delegates all execution and prompt-run queuing to AIEmbeddingRunner.
|
|
68
9
|
*/
|
|
69
10
|
export declare class AIModelRunner {
|
|
70
|
-
private
|
|
11
|
+
private _embeddingRunner;
|
|
71
12
|
/**
|
|
72
|
-
*
|
|
73
|
-
* embedding/model run record is observability — the caller gets its vectors regardless of whether
|
|
74
|
-
* the tracking row persists — so saves are queued, not awaited, and the embedding call is never
|
|
75
|
-
* blocked on a DB round-trip. The queue sequences saves per entity (the initial 'Running' INSERT
|
|
76
|
-
* always completes before the 'Completed'/'Failed' UPDATE). `PromptRunID` is returned immediately
|
|
77
|
-
* because `NewRecord()` client-generates the UUID.
|
|
78
|
-
*/
|
|
79
|
-
private _promptRunQueue;
|
|
80
|
-
/**
|
|
81
|
-
* Optional metadata provider override. Callers should set
|
|
82
|
-
* `instance.Provider = providerToUse` before invoking run methods
|
|
83
|
-
* in multi-provider contexts. Falls back to the global default provider when unset.
|
|
13
|
+
* Optional metadata provider override.
|
|
84
14
|
*/
|
|
85
15
|
get Provider(): IMetadataProvider;
|
|
86
16
|
set Provider(value: IMetadataProvider | null);
|
|
87
17
|
/**
|
|
88
18
|
* Execute an embedding call with full AIPromptRun tracking.
|
|
89
|
-
*
|
|
90
|
-
* Creates an AIPromptRun record before the call, invokes the embedding model,
|
|
91
|
-
* stores tokens/cost/timing on the run record, and returns the result.
|
|
19
|
+
* Delegates to {@link AIEmbeddingRunner.RunEmbedding}.
|
|
92
20
|
*
|
|
93
21
|
* @param params - Embedding execution parameters
|
|
94
22
|
* @returns Result with vectors, run ID, and usage metrics
|
|
95
23
|
*/
|
|
96
24
|
RunEmbedding(params: EmbeddingRunParams): Promise<EmbeddingRunResult>;
|
|
97
|
-
private resolveEmbeddingModel;
|
|
98
|
-
private findBestVendor;
|
|
99
|
-
private createRunRecord;
|
|
100
25
|
/**
|
|
101
|
-
* Awaits all in-flight prompt-run saves queued by this runner.
|
|
102
|
-
* call this — persistence is intentionally fire-and-forget. For tests / durability needs.
|
|
26
|
+
* Awaits all in-flight prompt-run saves queued by this runner.
|
|
103
27
|
*/
|
|
104
28
|
WaitForPendingPromptRunSaves(): Promise<void>;
|
|
105
|
-
private completeRunRecord;
|
|
106
|
-
private failRunRecord;
|
|
107
|
-
private errorResult;
|
|
108
29
|
}
|
|
109
30
|
//# sourceMappingURL=AIModelRunner.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AIModelRunner.d.ts","sourceRoot":"","sources":["../src/AIModelRunner.ts"],"names":[],"mappings":"AAAA,OAAO,
|
|
1
|
+
{"version":3,"file":"AIModelRunner.d.ts","sourceRoot":"","sources":["../src/AIModelRunner.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,iBAAiB,EAAE,MAAM,sBAAsB,CAAC;AAEzD,OAAO,EAAE,kBAAkB,EAAE,kBAAkB,EAAE,MAAM,oCAAoC,CAAC;AAE5F,YAAY,EAAE,kBAAkB,EAAE,kBAAkB,EAAE,MAAM,oCAAoC,CAAC;AAEjG;;;;;GAKG;AACH,qBAAa,aAAa;IACtB,OAAO,CAAC,gBAAgB,CAA2B;IAEnD;;OAEG;IACH,IAAW,QAAQ,IAAI,iBAAiB,CAEvC;IACD,IAAW,QAAQ,CAAC,KAAK,EAAE,iBAAiB,GAAG,IAAI,EAElD;IAED;;;;;;OAMG;IACU,YAAY,CAAC,MAAM,EAAE,kBAAkB,GAAG,OAAO,CAAC,kBAAkB,CAAC;IAIlF;;OAEG;IACU,4BAA4B,IAAI,OAAO,CAAC,IAAI,CAAC;CAG7D"}
|
package/dist/AIModelRunner.js
CHANGED
|
@@ -1,308 +1,38 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { MJGlobal, UUIDsEqual } from '@memberjunction/global';
|
|
3
|
-
import { BaseEmbeddings, GetAIAPIKey } from '@memberjunction/ai';
|
|
4
|
-
import { AIEngineBase } from '@memberjunction/ai-engine-base';
|
|
1
|
+
import { AIEmbeddingRunner } from './embedding/AIEmbeddingRunner.js';
|
|
5
2
|
/**
|
|
6
3
|
* AIModelRunner — Lightweight AI model execution tracker for non-LLM model types.
|
|
7
4
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* infrastructure as AIPromptRunner but without template rendering or validation.
|
|
11
|
-
*
|
|
12
|
-
* Architecture:
|
|
13
|
-
* - @memberjunction/ai (no MJ infra): BaseEmbeddings, ModelUsage
|
|
14
|
-
* - @memberjunction/ai-core-plus (this class): AIModelRunner — creates AIPromptRun records
|
|
15
|
-
* - @memberjunction/ai-prompts: AIPromptRunner — LLM-specific (template rendering, validation)
|
|
16
|
-
*
|
|
17
|
-
* Usage:
|
|
18
|
-
* ```typescript
|
|
19
|
-
* const runner = new AIModelRunner();
|
|
20
|
-
* const result = await runner.RunEmbedding({
|
|
21
|
-
* Texts: ['Hello world', 'How are you'],
|
|
22
|
-
* ContextUser: currentUser,
|
|
23
|
-
* Description: 'Content item vectorization'
|
|
24
|
-
* });
|
|
25
|
-
* // result.PromptRunID links to the AIPromptRun record with token/cost data
|
|
26
|
-
* ```
|
|
5
|
+
* @deprecated Use {@link AIEmbeddingRunner} instead. AIModelRunner is retained for backward
|
|
6
|
+
* compatibility and delegates all execution and prompt-run queuing to AIEmbeddingRunner.
|
|
27
7
|
*/
|
|
28
8
|
export class AIModelRunner {
|
|
29
9
|
constructor() {
|
|
30
|
-
this.
|
|
31
|
-
/**
|
|
32
|
-
* Fire-and-forget AIPromptRun persistence via the shared {@link BaseEntitySaveQueue}. The
|
|
33
|
-
* embedding/model run record is observability — the caller gets its vectors regardless of whether
|
|
34
|
-
* the tracking row persists — so saves are queued, not awaited, and the embedding call is never
|
|
35
|
-
* blocked on a DB round-trip. The queue sequences saves per entity (the initial 'Running' INSERT
|
|
36
|
-
* always completes before the 'Completed'/'Failed' UPDATE). `PromptRunID` is returned immediately
|
|
37
|
-
* because `NewRecord()` client-generates the UUID.
|
|
38
|
-
*/
|
|
39
|
-
this._promptRunQueue = new BaseEntitySaveQueue();
|
|
10
|
+
this._embeddingRunner = new AIEmbeddingRunner();
|
|
40
11
|
}
|
|
41
12
|
/**
|
|
42
|
-
* Optional metadata provider override.
|
|
43
|
-
* `instance.Provider = providerToUse` before invoking run methods
|
|
44
|
-
* in multi-provider contexts. Falls back to the global default provider when unset.
|
|
13
|
+
* Optional metadata provider override.
|
|
45
14
|
*/
|
|
46
15
|
get Provider() {
|
|
47
|
-
return this.
|
|
16
|
+
return this._embeddingRunner.Provider;
|
|
48
17
|
}
|
|
49
18
|
set Provider(value) {
|
|
50
|
-
this.
|
|
19
|
+
this._embeddingRunner.Provider = value;
|
|
51
20
|
}
|
|
52
21
|
/**
|
|
53
22
|
* Execute an embedding call with full AIPromptRun tracking.
|
|
54
|
-
*
|
|
55
|
-
* Creates an AIPromptRun record before the call, invokes the embedding model,
|
|
56
|
-
* stores tokens/cost/timing on the run record, and returns the result.
|
|
23
|
+
* Delegates to {@link AIEmbeddingRunner.RunEmbedding}.
|
|
57
24
|
*
|
|
58
25
|
* @param params - Embedding execution parameters
|
|
59
26
|
* @returns Result with vectors, run ID, and usage metrics
|
|
60
27
|
*/
|
|
61
28
|
async RunEmbedding(params) {
|
|
62
|
-
|
|
63
|
-
try {
|
|
64
|
-
// Ensure AIEngine is loaded
|
|
65
|
-
await AIEngineBase.Instance.Config(false, params.ContextUser);
|
|
66
|
-
// Resolve the embedding prompt and model
|
|
67
|
-
const { prompt, modelInfo } = await this.resolveEmbeddingModel(params);
|
|
68
|
-
if (!modelInfo) {
|
|
69
|
-
return this.errorResult('No embedding model available', startTime);
|
|
70
|
-
}
|
|
71
|
-
// Create AIPromptRun record (before the call)
|
|
72
|
-
const promptRun = await this.createRunRecord(prompt, modelInfo.ModelID, modelInfo.VendorID, params, startTime);
|
|
73
|
-
// Execute the embedding call
|
|
74
|
-
const embeddingInstance = MJGlobal.Instance.ClassFactory.CreateInstance(BaseEmbeddings, modelInfo.DriverClass, GetAIAPIKey(modelInfo.DriverClass));
|
|
75
|
-
if (!embeddingInstance) {
|
|
76
|
-
await this.failRunRecord(promptRun, 'Failed to create embedding instance', startTime);
|
|
77
|
-
return this.errorResult(`No embedding provider for driver: ${modelInfo.DriverClass}`, startTime);
|
|
78
|
-
}
|
|
79
|
-
const embedResult = await embeddingInstance.EmbedTexts({
|
|
80
|
-
texts: params.Texts,
|
|
81
|
-
model: modelInfo.APIName,
|
|
82
|
-
dimensions: params.Dimensions
|
|
83
|
-
});
|
|
84
|
-
if (!embedResult || !embedResult.vectors || embedResult.vectors.length === 0) {
|
|
85
|
-
await this.failRunRecord(promptRun, 'Embedding returned no vectors', startTime);
|
|
86
|
-
return this.errorResult('Embedding returned no vectors', startTime);
|
|
87
|
-
}
|
|
88
|
-
// Update AIPromptRun with usage data
|
|
89
|
-
await this.completeRunRecord(promptRun, embedResult, startTime);
|
|
90
|
-
return {
|
|
91
|
-
Success: true,
|
|
92
|
-
Vectors: embedResult.vectors,
|
|
93
|
-
PromptRunID: promptRun?.ID ?? null,
|
|
94
|
-
TokensUsed: embedResult.ModelUsage?.totalTokens ?? 0,
|
|
95
|
-
Cost: embedResult.ModelUsage?.cost ?? 0,
|
|
96
|
-
ErrorMessage: null,
|
|
97
|
-
ExecutionTimeMs: Date.now() - startTime,
|
|
98
|
-
};
|
|
99
|
-
}
|
|
100
|
-
catch (error) {
|
|
101
|
-
const msg = error instanceof Error ? error.message : String(error);
|
|
102
|
-
LogError(`AIModelRunner.RunEmbedding failed: ${msg}`);
|
|
103
|
-
return this.errorResult(msg, startTime);
|
|
104
|
-
}
|
|
105
|
-
}
|
|
106
|
-
// ================================================================
|
|
107
|
-
// Private: Model Resolution
|
|
108
|
-
// ================================================================
|
|
109
|
-
async resolveEmbeddingModel(params) {
|
|
110
|
-
const aiEngine = AIEngineBase.Instance;
|
|
111
|
-
// Find the embedding prompt
|
|
112
|
-
let prompt = null;
|
|
113
|
-
if (params.PromptID) {
|
|
114
|
-
prompt = aiEngine.Prompts.find(p => UUIDsEqual(p.ID, params.PromptID)) ?? null;
|
|
115
|
-
}
|
|
116
|
-
else {
|
|
117
|
-
// Find first active Embedding-type prompt
|
|
118
|
-
prompt = aiEngine.Prompts.find(p => p.Type?.toLowerCase() === 'embedding' && p.Status === 'Active') ?? null;
|
|
119
|
-
}
|
|
120
|
-
// If explicit model ID provided, use it directly
|
|
121
|
-
if (params.ModelID) {
|
|
122
|
-
const model = aiEngine.Models.find(m => UUIDsEqual(m.ID, params.ModelID));
|
|
123
|
-
if (model) {
|
|
124
|
-
const vendor = this.findBestVendor(model.ID);
|
|
125
|
-
if (vendor) {
|
|
126
|
-
return {
|
|
127
|
-
prompt,
|
|
128
|
-
modelInfo: {
|
|
129
|
-
ModelID: model.ID,
|
|
130
|
-
VendorID: vendor.VendorID,
|
|
131
|
-
DriverClass: vendor.DriverClass,
|
|
132
|
-
APIName: vendor.APIName ?? model.APIName ?? model.Name,
|
|
133
|
-
}
|
|
134
|
-
};
|
|
135
|
-
}
|
|
136
|
-
}
|
|
137
|
-
}
|
|
138
|
-
// Use prompt's model configuration if available
|
|
139
|
-
if (prompt) {
|
|
140
|
-
const promptModels = aiEngine.PromptModels
|
|
141
|
-
.filter(pm => pm.PromptID === prompt.ID)
|
|
142
|
-
.sort((a, b) => (b.Priority ?? 0) - (a.Priority ?? 0));
|
|
143
|
-
for (const pm of promptModels) {
|
|
144
|
-
// Look up model vendor to get driver class
|
|
145
|
-
const vendor = this.findBestVendor(pm.ModelID);
|
|
146
|
-
if (vendor) {
|
|
147
|
-
return {
|
|
148
|
-
prompt,
|
|
149
|
-
modelInfo: {
|
|
150
|
-
ModelID: pm.ModelID,
|
|
151
|
-
VendorID: vendor.VendorID,
|
|
152
|
-
DriverClass: vendor.DriverClass,
|
|
153
|
-
APIName: vendor.APIName ?? '',
|
|
154
|
-
}
|
|
155
|
-
};
|
|
156
|
-
}
|
|
157
|
-
}
|
|
158
|
-
}
|
|
159
|
-
// Fallback: find smallest active embedding model
|
|
160
|
-
const embeddingModels = aiEngine.Models.filter(m => {
|
|
161
|
-
const modelType = typeof m.AIModelType === 'string' ? m.AIModelType.trim().toLowerCase() : '';
|
|
162
|
-
return modelType === 'embeddings';
|
|
163
|
-
});
|
|
164
|
-
if (embeddingModels.length === 0)
|
|
165
|
-
return { prompt, modelInfo: null };
|
|
166
|
-
const sorted = [...embeddingModels].sort((a, b) => (a.InputTokenLimit ?? Number.MAX_SAFE_INTEGER) - (b.InputTokenLimit ?? Number.MAX_SAFE_INTEGER));
|
|
167
|
-
for (const model of sorted) {
|
|
168
|
-
const vendor = this.findBestVendor(model.ID);
|
|
169
|
-
if (vendor) {
|
|
170
|
-
return {
|
|
171
|
-
prompt,
|
|
172
|
-
modelInfo: {
|
|
173
|
-
ModelID: model.ID,
|
|
174
|
-
VendorID: vendor.VendorID,
|
|
175
|
-
DriverClass: vendor.DriverClass,
|
|
176
|
-
APIName: vendor.APIName ?? model.APIName ?? model.Name,
|
|
177
|
-
}
|
|
178
|
-
};
|
|
179
|
-
}
|
|
180
|
-
}
|
|
181
|
-
return { prompt, modelInfo: null };
|
|
182
|
-
}
|
|
183
|
-
findBestVendor(modelID) {
|
|
184
|
-
const aiEngine = AIEngineBase.Instance;
|
|
185
|
-
const vendors = aiEngine.ModelVendors
|
|
186
|
-
.filter(mv => mv.ModelID === modelID && mv.Status === 'Active' && mv.DriverClass != null)
|
|
187
|
-
.sort((a, b) => (b.Priority ?? 0) - (a.Priority ?? 0));
|
|
188
|
-
for (const v of vendors) {
|
|
189
|
-
const apiKey = GetAIAPIKey(v.DriverClass);
|
|
190
|
-
if (apiKey) {
|
|
191
|
-
return { VendorID: v.VendorID ?? '', DriverClass: v.DriverClass, APIName: v.APIName };
|
|
192
|
-
}
|
|
193
|
-
}
|
|
194
|
-
return null;
|
|
195
|
-
}
|
|
196
|
-
// ================================================================
|
|
197
|
-
// Private: AIPromptRun CRUD
|
|
198
|
-
// ================================================================
|
|
199
|
-
async createRunRecord(prompt, modelID, vendorID, params, startTime) {
|
|
200
|
-
try {
|
|
201
|
-
const md = this.Provider;
|
|
202
|
-
const promptRun = await md.GetEntityObject('MJ: AI Prompt Runs', params.ContextUser);
|
|
203
|
-
promptRun.NewRecord();
|
|
204
|
-
if (prompt) {
|
|
205
|
-
promptRun.PromptID = prompt.ID;
|
|
206
|
-
}
|
|
207
|
-
promptRun.ModelID = modelID;
|
|
208
|
-
promptRun.VendorID = vendorID || null;
|
|
209
|
-
promptRun.Status = 'Running';
|
|
210
|
-
promptRun.RunAt = new Date(startTime);
|
|
211
|
-
promptRun.Cancelled = false;
|
|
212
|
-
promptRun.CacheHit = false;
|
|
213
|
-
promptRun.StreamingEnabled = false;
|
|
214
|
-
if (params.ParentRunID) {
|
|
215
|
-
promptRun.ParentID = params.ParentRunID;
|
|
216
|
-
}
|
|
217
|
-
// Store description in Messages field as context
|
|
218
|
-
if (params.Description) {
|
|
219
|
-
promptRun.Messages = JSON.stringify({
|
|
220
|
-
description: params.Description,
|
|
221
|
-
textCount: params.Texts.length,
|
|
222
|
-
totalChars: params.Texts.reduce((sum, t) => sum + t.length, 0),
|
|
223
|
-
});
|
|
224
|
-
}
|
|
225
|
-
// Fire-and-forget the initial 'Running' INSERT — ID is already assigned by NewRecord()
|
|
226
|
-
// so the returned PromptRunID is valid immediately; the UPDATE chains after this.
|
|
227
|
-
this._promptRunQueue.Insert(promptRun);
|
|
228
|
-
return promptRun;
|
|
229
|
-
}
|
|
230
|
-
catch (error) {
|
|
231
|
-
LogError(`AIModelRunner: Error creating AIPromptRun: ${error}`);
|
|
232
|
-
return null;
|
|
233
|
-
}
|
|
29
|
+
return this._embeddingRunner.RunEmbedding(params);
|
|
234
30
|
}
|
|
235
31
|
/**
|
|
236
|
-
* Awaits all in-flight prompt-run saves queued by this runner.
|
|
237
|
-
* call this — persistence is intentionally fire-and-forget. For tests / durability needs.
|
|
32
|
+
* Awaits all in-flight prompt-run saves queued by this runner.
|
|
238
33
|
*/
|
|
239
34
|
async WaitForPendingPromptRunSaves() {
|
|
240
|
-
await this.
|
|
241
|
-
}
|
|
242
|
-
async completeRunRecord(promptRun, embedResult, startTime) {
|
|
243
|
-
if (!promptRun)
|
|
244
|
-
return;
|
|
245
|
-
try {
|
|
246
|
-
promptRun.Status = 'Completed';
|
|
247
|
-
promptRun.Success = true;
|
|
248
|
-
promptRun.CompletedAt = new Date();
|
|
249
|
-
promptRun.ExecutionTimeMS = Date.now() - startTime;
|
|
250
|
-
// Store token/cost from ModelUsage
|
|
251
|
-
if (embedResult.ModelUsage) {
|
|
252
|
-
// TokensPrompt = UNCACHED ("net-new") input; cache reads/writes tracked separately.
|
|
253
|
-
// TokensUsed = totalTokens = promptTokens + completionTokens (EXCLUDES cache), to
|
|
254
|
-
// satisfy the AIPromptRun invariant TokensUsed === TokensPrompt + TokensCompletion.
|
|
255
|
-
// (Embeddings don't cache, so cache buckets are 0 here regardless.)
|
|
256
|
-
promptRun.TokensPrompt = embedResult.ModelUsage.promptTokens ?? 0;
|
|
257
|
-
promptRun.TokensCompletion = embedResult.ModelUsage.completionTokens ?? 0;
|
|
258
|
-
promptRun.TokensUsed = embedResult.ModelUsage.totalTokens ?? 0;
|
|
259
|
-
promptRun.TokensCacheRead = embedResult.ModelUsage.cacheReadTokens ?? 0;
|
|
260
|
-
promptRun.TokensCacheWrite = embedResult.ModelUsage.cacheWriteTokens ?? 0;
|
|
261
|
-
promptRun.Cost = embedResult.ModelUsage.cost ?? 0;
|
|
262
|
-
promptRun.CostCurrency = embedResult.ModelUsage.costCurrency ?? 'USD';
|
|
263
|
-
promptRun.QueueTime = embedResult.ModelUsage.queueTime ?? 0;
|
|
264
|
-
promptRun.PromptTime = embedResult.ModelUsage.promptTime ?? 0;
|
|
265
|
-
promptRun.CompletionTime = embedResult.ModelUsage.completionTime ?? 0;
|
|
266
|
-
}
|
|
267
|
-
// Store vector count in Result
|
|
268
|
-
promptRun.Result = JSON.stringify({
|
|
269
|
-
vectorCount: embedResult.vectors?.length ?? 0,
|
|
270
|
-
dimensions: embedResult.vectors?.[0]?.length ?? 0,
|
|
271
|
-
});
|
|
272
|
-
this._promptRunQueue.Update(promptRun); // fire-and-forget UPDATE; the INSERT landed during the embedding call
|
|
273
|
-
}
|
|
274
|
-
catch (error) {
|
|
275
|
-
LogError(`AIModelRunner: Error completing AIPromptRun: ${error}`);
|
|
276
|
-
}
|
|
277
|
-
}
|
|
278
|
-
async failRunRecord(promptRun, errorMessage, startTime) {
|
|
279
|
-
if (!promptRun)
|
|
280
|
-
return;
|
|
281
|
-
try {
|
|
282
|
-
promptRun.Status = 'Failed';
|
|
283
|
-
promptRun.Success = false;
|
|
284
|
-
promptRun.ErrorMessage = errorMessage;
|
|
285
|
-
promptRun.CompletedAt = new Date();
|
|
286
|
-
promptRun.ExecutionTimeMS = Date.now() - startTime;
|
|
287
|
-
this._promptRunQueue.Update(promptRun); // fire-and-forget UPDATE; the INSERT landed during the embedding call
|
|
288
|
-
}
|
|
289
|
-
catch (error) {
|
|
290
|
-
LogError(`AIModelRunner: Error failing AIPromptRun: ${error}`);
|
|
291
|
-
}
|
|
292
|
-
}
|
|
293
|
-
// ================================================================
|
|
294
|
-
// Private: Helpers
|
|
295
|
-
// ================================================================
|
|
296
|
-
errorResult(message, startTime) {
|
|
297
|
-
return {
|
|
298
|
-
Success: false,
|
|
299
|
-
Vectors: [],
|
|
300
|
-
PromptRunID: null,
|
|
301
|
-
TokensUsed: 0,
|
|
302
|
-
Cost: 0,
|
|
303
|
-
ErrorMessage: message,
|
|
304
|
-
ExecutionTimeMs: Date.now() - startTime,
|
|
305
|
-
};
|
|
35
|
+
await this._embeddingRunner.WaitForPendingPromptRunSaves();
|
|
306
36
|
}
|
|
307
37
|
}
|
|
308
38
|
//# sourceMappingURL=AIModelRunner.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AIModelRunner.js","sourceRoot":"","sources":["../src/AIModelRunner.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"AIModelRunner.js","sourceRoot":"","sources":["../src/AIModelRunner.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,iBAAiB,EAAE,MAAM,+BAA+B,CAAC;AAKlE;;;;;GAKG;AACH,MAAM,OAAO,aAAa;IAA1B;QACY,qBAAgB,GAAG,IAAI,iBAAiB,EAAE,CAAC;IA6BvD,CAAC;IA3BG;;OAEG;IACH,IAAW,QAAQ;QACf,OAAO,IAAI,CAAC,gBAAgB,CAAC,QAAQ,CAAC;IAC1C,CAAC;IACD,IAAW,QAAQ,CAAC,KAA+B;QAC/C,IAAI,CAAC,gBAAgB,CAAC,QAAQ,GAAG,KAAK,CAAC;IAC3C,CAAC;IAED;;;;;;OAMG;IACI,KAAK,CAAC,YAAY,CAAC,MAA0B;QAChD,OAAO,IAAI,CAAC,gBAAgB,CAAC,YAAY,CAAC,MAAM,CAAC,CAAC;IACtD,CAAC;IAED;;OAEG;IACI,KAAK,CAAC,4BAA4B;QACrC,MAAM,IAAI,CAAC,gBAAgB,CAAC,4BAA4B,EAAE,CAAC;IAC/D,CAAC;CACJ"}
|