@memberjunction/ai-prompts 2.44.0 → 2.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -13,8 +13,12 @@ The MemberJunction AI Prompts package provides sophisticated prompt management,
13
13
  - **🔄 Template Integration**: Dynamic prompt generation with MemberJunction template system
14
14
  - **📊 Execution Analytics**: Comprehensive metrics, token usage tracking, and performance monitoring
15
15
  - **🎯 Result Selection**: AI-powered selection of best results from parallel executions
16
- - **🔧 Output Validation**: Structured output validation with retry logic
16
+ - **🔧 Enhanced Output Validation**: JSON schema validation against OutputExample with intelligent retry logic
17
17
  - **⚙️ Configuration-Driven**: Metadata-driven prompt configuration and execution
18
+ - **🗃️ Hierarchical Logging**: Parent-child relationship tracking for parallel executions
19
+ - **🚫 Cancellation Support**: AbortSignal integration for graceful execution cancellation
20
+ - **📈 Progress Updates**: Real-time progress callbacks and streaming response support
21
+ - **🔄 Streaming Integration**: Compatible with BaseLLM streaming capabilities
18
22
 
19
23
  ## Installation
20
24
 
@@ -146,6 +150,144 @@ if (result.promptRun?.Messages) {
146
150
  }
147
151
  ```
148
152
 
153
+ ### 4. Complete Example with All New Features
154
+
155
+ ```typescript
156
+ import { AIPromptRunner } from '@memberjunction/ai-prompts';
157
+ import { AIEngine } from '@memberjunction/aiengine';
158
+
159
+ // Complete example showcasing all Phase 6 enhancements
160
+ async function comprehensivePromptExecution() {
161
+ // Initialize
162
+ await AIEngine.Instance.Config(false, currentUser);
163
+ const runner = new AIPromptRunner();
164
+
165
+ // Set up cancellation (e.g., from user clicking cancel button)
166
+ const controller = new AbortController();
167
+ const timeoutId = setTimeout(() => {
168
+ controller.abort();
169
+ console.log('Operation timed out after 2 minutes');
170
+ }, 120000);
171
+
172
+ try {
173
+ const result = await runner.ExecutePrompt({
174
+ prompt: complexAnalysisPrompt, // ParallelizationMode: 'ModelSpecific'
175
+ data: {
176
+ document: largeDocument,
177
+ analysisType: 'comprehensive',
178
+ outputFormat: 'structured'
179
+ },
180
+ contextUser: currentUser,
181
+
182
+ // Enable cancellation
183
+ cancellationToken: controller.signal,
184
+
185
+ // Track progress throughout execution
186
+ onProgress: (progress) => {
187
+ console.log(`[${progress.step}] ${progress.percentage}% - ${progress.message}`);
188
+
189
+ // Handle parallel execution progress
190
+ if (progress.metadata?.parallelExecution) {
191
+ const parallel = progress.metadata.parallelExecution;
192
+ console.log(` → Group ${parallel.currentGroup + 1}/${parallel.totalGroups}, Tasks: ${parallel.completedTasks}/${parallel.totalTasks}`);
193
+ }
194
+
195
+ // Update UI
196
+ updateProgressBar(progress.percentage);
197
+ updateStatusText(progress.message);
198
+ },
199
+
200
+ // Receive streaming content updates
201
+ onStreaming: (chunk) => {
202
+ if (chunk.isComplete) {
203
+ console.log(`Streaming complete for ${chunk.modelName}`);
204
+ finalizeOutput();
205
+ } else {
206
+ // Show real-time content generation
207
+ console.log(`[${chunk.modelName}]: ${chunk.content.substring(0, 50)}...`);
208
+ appendToDisplay(chunk.content, chunk.taskId);
209
+ }
210
+ }
211
+ });
212
+
213
+ // Clear timeout since we completed successfully
214
+ clearTimeout(timeoutId);
215
+
216
+ // Handle different result scenarios
217
+ if (result.cancelled) {
218
+ console.log(`Execution cancelled: ${result.cancellationReason}`);
219
+ // May still have partial results available
220
+ if (result.additionalResults && result.additionalResults.length > 0) {
221
+ console.log(`${result.additionalResults.length} partial results available`);
222
+ }
223
+ } else if (result.success) {
224
+ console.log('Execution completed successfully!');
225
+ console.log(`Primary result from ${result.modelInfo?.modelName}: ${result.result}`);
226
+
227
+ // Analyze judge selection if multiple results
228
+ if (result.ranking && result.judgeRationale) {
229
+ console.log(`Selected as #${result.ranking} by AI judge: ${result.judgeRationale}`);
230
+ }
231
+
232
+ // Review alternative results from parallel execution
233
+ if (result.additionalResults) {
234
+ console.log(`${result.additionalResults.length} alternative results ranked by judge:`);
235
+ result.additionalResults.forEach((altResult, index) => {
236
+ console.log(` ${altResult.ranking}. ${altResult.modelInfo?.modelName}: ${altResult.judgeRationale}`);
237
+ });
238
+ }
239
+
240
+ // Analyze execution performance using hierarchical logging
241
+ if (result.promptRun?.RunType === 'ParallelParent') {
242
+ await analyzeParallelExecutionPerformance(result.promptRun.ID);
243
+ }
244
+
245
+ // Check streaming and caching
246
+ if (result.wasStreamed) {
247
+ console.log('Response was streamed in real-time');
248
+ }
249
+ if (result.cacheInfo?.cacheHit) {
250
+ console.log(`Result served from cache: ${result.cacheInfo.cacheSource}`);
251
+ }
252
+ } else {
253
+ console.error(`Execution failed: ${result.errorMessage}`);
254
+ }
255
+
256
+ } catch (error) {
257
+ clearTimeout(timeoutId);
258
+ console.error('Execution error:', error.message);
259
+ }
260
+ }
261
+
262
+ // Helper function to analyze parallel execution performance
263
+ async function analyzeParallelExecutionPerformance(parentPromptRunId: string) {
264
+ // Query hierarchical logs to understand execution breakdown
265
+ console.log('Analyzing parallel execution performance...');
266
+
267
+ // This would typically be a database query or API call
268
+ // For demonstration, showing the concept:
269
+ const analysisQuery = `
270
+ SELECT
271
+ pr.RunType,
272
+ pr.ExecutionOrder,
273
+ pr.Success,
274
+ pr.ExecutionTimeMS,
275
+ pr.TokensUsed,
276
+ m.Name as ModelName
277
+ FROM AIPromptRun pr
278
+ JOIN AIModel m ON pr.ModelID = m.ID
279
+ WHERE pr.ParentID = '${parentPromptRunId}' OR pr.ID = '${parentPromptRunId}'
280
+ ORDER BY pr.RunType, pr.ExecutionOrder
281
+ `;
282
+
283
+ console.log('Performance analysis query:', analysisQuery);
284
+ // Execute query and analyze results...
285
+ }
286
+
287
+ // Execute the comprehensive example
288
+ comprehensivePromptExecution().catch(console.error);
289
+ ```
290
+
149
291
  ## Advanced Features
150
292
 
151
293
  ### Intelligent Caching
@@ -392,6 +534,578 @@ if (result.promptRun) {
392
534
  }
393
535
  ```
394
536
 
537
+ ## AI Prompt Run Logging
538
+
539
+ The AI Prompt Runner implements a sophisticated hierarchical logging system that tracks all execution activities in the database through the `AIPromptRun` entity. This system provides complete traceability and analytics for both simple and complex parallel executions.
540
+
541
+ ### Hierarchical Logging Structure
542
+
543
+ The logging system uses a parent-child relationship model with different `RunType` values to represent the execution hierarchy:
544
+
545
+ - **`Single`**: Standard single-model execution
546
+ - **`ParallelParent`**: Parent record for parallel execution coordinating multiple models
547
+ - **`ParallelChild`**: Individual model execution within a parallel run
548
+ - **`ResultSelector`**: AI judge execution that selects the best result from parallel executions
549
+
550
+ ### RunType Values and Relationships
551
+
552
+ ```typescript
553
+ // Single execution - no parent relationship
554
+ {
555
+ RunType: 'Single',
556
+ ParentID: null,
557
+ ExecutionOrder: null
558
+ }
559
+
560
+ // Parallel execution creates a hierarchical structure:
561
+ // 1. Parent record coordinates the overall execution
562
+ {
563
+ RunType: 'ParallelParent',
564
+ ParentID: null,
565
+ ExecutionOrder: null
566
+ }
567
+
568
+ // 2. Child records for each model execution
569
+ {
570
+ RunType: 'ParallelChild',
571
+ ParentID: '12345-parent-id',
572
+ ExecutionOrder: 0 // Order within execution group
573
+ }
574
+
575
+ // 3. Result selector judges the best result
576
+ {
577
+ RunType: 'ResultSelector',
578
+ ParentID: '12345-parent-id',
579
+ ExecutionOrder: 5 // After all parallel children
580
+ }
581
+ ```
582
+
583
+ ### Database Schema Fields
584
+
585
+ Key fields in the `AIPromptRun` entity for hierarchical logging:
586
+
587
+ ```sql
588
+ -- Core execution tracking
589
+ PromptID uniqueidentifier -- Prompt being executed
590
+ ModelID uniqueidentifier -- AI model used
591
+ VendorID uniqueidentifier -- Vendor providing the model
592
+ RunAt datetime2 -- Execution start time
593
+ CompletedAt datetime2 -- Execution completion time
594
+
595
+ -- Hierarchical logging fields
596
+ RunType nvarchar(50) -- 'Single', 'ParallelParent', 'ParallelChild', 'ResultSelector'
597
+ ParentID uniqueidentifier -- Parent prompt run ID (NULL for top-level)
598
+ ExecutionOrder int -- Order within parallel execution group
599
+
600
+ -- Results and metrics
601
+ Success bit -- Whether execution succeeded
602
+ Result nvarchar(max) -- Raw result from AI model
603
+ ErrorMessage nvarchar(500) -- Error message if failed
604
+ ExecutionTimeMS int -- Total execution time
605
+ TokensUsed int -- Total tokens consumed
606
+ TokensPrompt int -- Prompt tokens used
607
+ TokensCompletion int -- Completion tokens generated
608
+
609
+ -- Context and configuration
610
+ Messages nvarchar(max) -- JSON with input data and metadata
611
+ ConfigurationID uniqueidentifier -- Environment configuration used
612
+ ```
613
+
614
+ ### Querying Hierarchical Log Data
615
+
616
+ The hierarchical structure enables powerful analytics queries:
617
+
618
+ ```sql
619
+ -- Get all executions for a parallel run
620
+ SELECT
621
+ pr.ID,
622
+ pr.RunType,
623
+ pr.ExecutionOrder,
624
+ pr.Success,
625
+ pr.ExecutionTimeMS,
626
+ pr.TokensUsed,
627
+ m.Name as ModelName,
628
+ p.Name as PromptName
629
+ FROM AIPromptRun pr
630
+ JOIN AIModel m ON pr.ModelID = m.ID
631
+ JOIN AIPrompt p ON pr.PromptID = p.ID
632
+ WHERE pr.ParentID = '12345-parent-id'
633
+ OR pr.ID = '12345-parent-id'
634
+ ORDER BY pr.RunType, pr.ExecutionOrder;
635
+
636
+ -- Analyze parallel execution performance
637
+ WITH ParallelStats AS (
638
+ SELECT
639
+ ParentID,
640
+ COUNT(*) as TotalChildren,
641
+ SUM(CASE WHEN Success = 1 THEN 1 ELSE 0 END) as SuccessfulChildren,
642
+ AVG(ExecutionTimeMS) as AvgExecutionTime,
643
+ SUM(TokensUsed) as TotalTokens
644
+ FROM AIPromptRun
645
+ WHERE RunType = 'ParallelChild'
646
+ AND ParentID IS NOT NULL
647
+ GROUP BY ParentID
648
+ )
649
+ SELECT
650
+ parent.ID as ParentRunID,
651
+ parent.RunAt,
652
+ parent.ExecutionTimeMS as ParentExecutionTime,
653
+ stats.TotalChildren,
654
+ stats.SuccessfulChildren,
655
+ stats.AvgExecutionTime,
656
+ stats.TotalTokens,
657
+ prompt.Name as PromptName
658
+ FROM AIPromptRun parent
659
+ JOIN ParallelStats stats ON parent.ID = stats.ParentID
660
+ JOIN AIPrompt prompt ON parent.PromptID = prompt.ID
661
+ WHERE parent.RunType = 'ParallelParent'
662
+ ORDER BY parent.RunAt DESC;
663
+
664
+ -- Find failed executions with context
665
+ SELECT
666
+ pr.ID,
667
+ pr.RunType,
668
+ pr.ParentID,
669
+ pr.ErrorMessage,
670
+ pr.ExecutionTimeMS,
671
+ m.Name as ModelName,
672
+ v.Name as VendorName,
673
+ p.Name as PromptName
674
+ FROM AIPromptRun pr
675
+ JOIN AIModel m ON pr.ModelID = m.ID
676
+ LEFT JOIN AIVendor v ON pr.VendorID = v.ID
677
+ JOIN AIPrompt p ON pr.PromptID = p.ID
678
+ WHERE pr.Success = 0
679
+ ORDER BY pr.RunAt DESC;
680
+ ```
681
+
682
+ ## Cancellation Support
683
+
684
+ The AI Prompt Runner provides comprehensive cancellation support through the standard JavaScript `AbortSignal` and `AbortController` pattern, enabling graceful termination of long-running operations.
685
+
686
+ ### Understanding AbortSignal in Prompt Execution
687
+
688
+ The `AbortSignal` pattern separates **cancellation control** from **cancellation handling**:
689
+
690
+ - **Your Code (Controller)**: Creates the `AbortController` and decides **when** to cancel
691
+ - **Prompt Runner (Worker)**: Receives the `AbortSignal` token and handles **how** to cancel gracefully
692
+
693
+ This separation allows for flexible cancellation from multiple sources (user actions, timeouts, resource limits) while the Prompt Runner handles the complex cleanup across parallel executions, model calls, and result selection.
694
+
695
+ **The Pattern Flow:**
696
+ ```
697
+ Controller (Your Code) → AbortController.signal → AIPromptRunner
698
+ ↓ ↓ ↓
699
+ Decides WHEN The "Red Phone" Handles HOW
700
+ to cancel Token to stop
701
+ ```
702
+
703
+ ### Basic Cancellation Usage
704
+
705
+ ```typescript
706
+ import { AIPromptRunner } from '@memberjunction/ai-prompts';
707
+
708
+ // Create cancellation controller
709
+ const controller = new AbortController();
710
+ const cancellationToken = controller.signal;
711
+
712
+ // Set up cancellation after 30 seconds
713
+ setTimeout(() => {
714
+ controller.abort();
715
+ console.log('Prompt execution cancelled due to timeout');
716
+ }, 30000);
717
+
718
+ // Execute prompt with cancellation support
719
+ const runner = new AIPromptRunner();
720
+ const result = await runner.ExecutePrompt({
721
+ prompt: myPrompt,
722
+ data: { query: 'Long running analysis...' },
723
+ contextUser: currentUser,
724
+ cancellationToken: cancellationToken
725
+ });
726
+
727
+ // Check if execution was cancelled
728
+ if (result.cancelled) {
729
+ console.log(`Execution cancelled: ${result.cancellationReason}`);
730
+ console.log('Partial results may be available');
731
+ } else if (result.success) {
732
+ console.log('Execution completed successfully');
733
+ }
734
+ ```
735
+
736
+ ### Cancellation in Parallel Execution
737
+
738
+ Cancellation works seamlessly with parallel execution, allowing you to stop all running tasks:
739
+
740
+ ```typescript
741
+ const controller = new AbortController();
742
+
743
+ // User clicks cancel button
744
+ document.getElementById('cancelButton').onclick = () => {
745
+ controller.abort();
746
+ };
747
+
748
+ // Execute parallel prompt with multiple models
749
+ const result = await runner.ExecutePrompt({
750
+ prompt: parallelPrompt, // ParallelizationMode: 'ModelSpecific'
751
+ data: analysisData,
752
+ contextUser: currentUser,
753
+ cancellationToken: controller.signal
754
+ });
755
+
756
+ // Parallel cancellation behavior:
757
+ // - Tasks not yet started will be marked as cancelled
758
+ // - Currently executing tasks will be terminated
759
+ // - Completed tasks remain in the results
760
+ // - Partial results may still be available for analysis
761
+ ```
762
+
763
+ ### Multiple Cancellation Sources
764
+
765
+ One of the powerful aspects of the AbortSignal pattern is that multiple sources can cancel the same operation:
766
+
767
+ ```typescript
768
+ async function intelligentPromptExecution() {
769
+ const controller = new AbortController();
770
+ const signal = controller.signal;
771
+
772
+ // 1. User cancel button
773
+ document.getElementById('cancelBtn')?.addEventListener('click', () => {
774
+ controller.abort(); // User-initiated cancellation
775
+ console.log('User cancelled the operation');
776
+ });
777
+
778
+ // 2. Timeout cancellation (prevent runaway prompts)
779
+ const timeout = setTimeout(() => {
780
+ controller.abort(); // Timeout cancellation
781
+ console.log('Operation timed out after 2 minutes');
782
+ }, 120000);
783
+
784
+ // 3. Resource limit cancellation
785
+ const memoryCheck = setInterval(async () => {
786
+ if (await getMemoryUsage() > MAX_MEMORY_THRESHOLD) {
787
+ controller.abort(); // Resource limit cancellation
788
+ console.log('Cancelled due to memory limits');
789
+ }
790
+ }, 5000);
791
+
792
+ // 4. Window unload cancellation (cleanup on page close)
793
+ window.addEventListener('beforeunload', () => {
794
+ controller.abort(); // Page closing cancellation
795
+ });
796
+
797
+ try {
798
+ const result = await runner.ExecutePrompt({
799
+ prompt: complexAnalysisPrompt,
800
+ data: largeDataset,
801
+ cancellationToken: signal // One token, many cancel sources!
802
+ });
803
+
804
+ // Clean up timers if successful
805
+ clearTimeout(timeout);
806
+ clearInterval(memoryCheck);
807
+
808
+ return result;
809
+ } catch (error) {
810
+ // The Prompt Runner doesn't know WHY it was cancelled
811
+ // It just knows it should stop gracefully
812
+ console.log('Prompt execution was cancelled:', error.message);
813
+ } finally {
814
+ clearTimeout(timeout);
815
+ clearInterval(memoryCheck);
816
+ }
817
+ }
818
+ ```
819
+
820
+ ### Cancellation in Component-Based UIs
821
+
822
+ Perfect for React, Angular, or Vue components:
823
+
824
+ ```typescript
825
+ class PromptExecutionComponent {
826
+ private currentController: AbortController | null = null;
827
+ private isExecuting: boolean = false;
828
+
829
+ async executePrompt(prompt: AIPromptEntity, data: any) {
830
+ // Cancel any existing execution
831
+ this.cancelCurrentExecution();
832
+
833
+ // Create new controller for this execution
834
+ this.currentController = new AbortController();
835
+ this.isExecuting = true;
836
+
837
+ try {
838
+ const result = await this.runner.ExecutePrompt({
839
+ prompt,
840
+ data,
841
+ cancellationToken: this.currentController.signal,
842
+ onProgress: (progress) => {
843
+ this.updateUI(`${progress.step}: ${progress.percentage}%`);
844
+ },
845
+ onStreaming: (chunk) => {
846
+ this.appendStreamingContent(chunk.content);
847
+ }
848
+ });
849
+
850
+ this.handleSuccess(result);
851
+ } catch (error) {
852
+ if (error.message.includes('cancelled')) {
853
+ this.handleCancellation();
854
+ } else {
855
+ this.handleError(error);
856
+ }
857
+ } finally {
858
+ this.isExecuting = false;
859
+ this.currentController = null;
860
+ }
861
+ }
862
+
863
+ // Called when user clicks "Cancel" or navigates away
864
+ cancelCurrentExecution() {
865
+ if (this.currentController && this.isExecuting) {
866
+ this.currentController.abort();
867
+ console.log('Cancelled current prompt execution');
868
+ }
869
+ }
870
+
871
+ // Component cleanup
872
+ ngOnDestroy() { // Angular example
873
+ this.cancelCurrentExecution();
874
+ }
875
+ }
876
+ ```
877
+
878
+ ### Integration with BaseLLM Cancellation
879
+
880
+ The cancellation token is automatically propagated through the entire execution chain:
881
+
882
+ ```typescript
883
+ // Cancellation Flow in MemberJunction AI Architecture:
884
+ //
885
+ // 1. User Code (AbortController.signal)
886
+ // ↓
887
+ // 2. AIPromptRunner.ExecutePrompt(cancellationToken)
888
+ // ↓
889
+ // 3. ParallelExecutionCoordinator.executeTasksInParallel(cancellationToken)
890
+ // ↓
891
+ // 4. Individual Task Execution with cancellation
892
+ // ↓
893
+ // 5. BaseLLM.ChatCompletion({ cancellationToken })
894
+ // ↓
895
+ // 6. Provider-specific cancellation (fetch signal, Promise.race)
896
+ // ↓
897
+ // 7. AI Model API cancellation (if supported)
898
+
899
+ // At each level, cancellation is handled appropriately:
900
+ const internalFlow = {
901
+ // Level 1: Prompt Runner checks before major operations
902
+ promptRunner: () => {
903
+ if (cancellationToken?.aborted) {
904
+ return { success: false, cancelled: true };
905
+ }
906
+ },
907
+
908
+ // Level 2: Parallel coordinator cancels remaining tasks
909
+ parallelCoordinator: () => {
910
+ tasks.forEach(task => {
911
+ if (cancellationToken?.aborted) {
912
+ task.cancelled = true;
913
+ }
914
+ });
915
+ },
916
+
917
+ // Level 3: BaseLLM uses Promise.race for instant cancellation
918
+ baseLLM: () => {
919
+ return Promise.race([
920
+ actualModelCall(params),
921
+ cancellationPromise(cancellationToken)
922
+ ]);
923
+ },
924
+
925
+ // Level 4: Native provider cancellation (where supported)
926
+ provider: () => {
927
+ fetch(apiUrl, {
928
+ signal: cancellationToken // Native browser/Node.js cancellation
929
+ });
930
+ }
931
+ };
932
+ ```
933
+
934
+ ### Cancellation Guarantees
935
+
936
+ The AI Prompt Runner provides these cancellation guarantees:
937
+
938
+ 1. **🚫 Instant Recognition**: Cancellation requests are checked at multiple points throughout execution
939
+ 2. **🧹 Graceful Cleanup**: Partial results are preserved and returned when possible
940
+ 3. **📊 Proper Logging**: Cancelled operations are logged with appropriate status and metadata
941
+ 4. **💾 Resource Release**: Network connections and memory are cleaned up promptly
942
+ 5. **🔄 State Consistency**: The system remains in a consistent state after cancellation
943
+
944
+ **Key Benefits:**
945
+ - **Responsive UI**: Users get immediate feedback when cancelling operations
946
+ - **Resource Efficiency**: Prevents wasted compute and API costs
947
+ - **System Stability**: Avoids memory leaks and hanging operations
948
+ - **Standard Pattern**: Uses native JavaScript APIs - no custom cancellation logic needed
949
+
950
+ ### Cancellation Result Properties
951
+
952
+ When execution is cancelled, the result includes detailed cancellation information:
953
+
954
+ ```typescript
955
+ interface AIPromptRunResult {
956
+ success: boolean;
957
+ cancelled?: boolean; // True if execution was cancelled
958
+ cancellationReason?: CancellationReason; // Why it was cancelled
959
+ status?: ExecutionStatus; // Current execution status
960
+ // ... other properties
961
+ }
962
+
963
+ type CancellationReason = 'user_requested' | 'timeout' | 'error' | 'resource_limit';
964
+ type ExecutionStatus = 'pending' | 'running' | 'completed' | 'failed' | 'cancelled';
965
+ ```
966
+
967
+ ## Progress Updates & Streaming
968
+
969
+ The AI Prompt Runner provides real-time progress updates and streaming support for long-running executions, enabling responsive user interfaces and monitoring dashboards.
970
+
971
+ ### Progress Callbacks
972
+
973
+ Track execution progress through different phases:
974
+
975
+ ```typescript
976
+ const runner = new AIPromptRunner();
977
+
978
+ const result = await runner.ExecutePrompt({
979
+ prompt: complexPrompt,
980
+ data: { document: longDocument },
981
+ contextUser: currentUser,
982
+
983
+ // Progress callback receives updates throughout execution
984
+ onProgress: (progress) => {
985
+ console.log(`${progress.step}: ${progress.percentage}% - ${progress.message}`);
986
+
987
+ // Update UI progress bar
988
+ updateProgressBar(progress.percentage);
989
+ updateStatusMessage(progress.message);
990
+
991
+ // Access additional metadata
992
+ if (progress.metadata) {
993
+ console.log('Execution metadata:', progress.metadata);
994
+ }
995
+ }
996
+ });
997
+ ```
998
+
999
+ ### Execution Progress Phases
1000
+
1001
+ The progress callback receives updates for these execution phases:
1002
+
1003
+ ```typescript
1004
+ type ProgressPhase =
1005
+ | 'template_rendering' // Rendering prompt template with data
1006
+ | 'model_selection' // Selecting appropriate AI model
1007
+ | 'execution' // Executing AI model
1008
+ | 'validation' // Validating and parsing results
1009
+ | 'parallel_coordination' // Coordinating parallel executions
1010
+ | 'result_selection'; // AI judge selecting best result
1011
+
1012
+ // Example progress updates:
1013
+ // template_rendering: 20% - "Rendering prompt template with provided data"
1014
+ // model_selection: 40% - "Selected GPT-4 model based on prompt configuration"
1015
+ // execution: 60% - "Executing AI model..."
1016
+ // validation: 80% - "Validating output against expected format"
1017
+ // result_selection: 90% - "AI judge selecting best result from 3 candidates"
1018
+ ```
1019
+
1020
+ ### Streaming Response Support
1021
+
1022
+ Receive real-time content updates as AI models generate responses:
1023
+
1024
+ ```typescript
1025
+ const result = await runner.ExecutePrompt({
1026
+ prompt: streamingPrompt,
1027
+ data: { query: 'Generate a detailed report...' },
1028
+ contextUser: currentUser,
1029
+
1030
+ // Streaming callback receives content chunks as they arrive
1031
+ onStreaming: (chunk) => {
1032
+ if (chunk.isComplete) {
1033
+ console.log('Streaming complete');
1034
+ finalizeDocument();
1035
+ } else {
1036
+ // Append content chunk to UI
1037
+ appendToDocument(chunk.content);
1038
+
1039
+ // Show which model is generating content (for parallel execution)
1040
+ if (chunk.modelName) {
1041
+ showActiveModel(chunk.modelName);
1042
+ }
1043
+ }
1044
+ }
1045
+ });
1046
+ ```
1047
+
1048
+ ### Progress Updates in Parallel Execution
1049
+
1050
+ Progress tracking works seamlessly with parallel execution:
1051
+
1052
+ ```typescript
1053
+ const result = await runner.ExecutePrompt({
1054
+ prompt: parallelPrompt, // Uses multiple models
1055
+ data: analysisData,
1056
+ contextUser: currentUser,
1057
+
1058
+ onProgress: (progress) => {
1059
+ // Parallel execution provides additional metadata
1060
+ if (progress.metadata?.parallelExecution) {
1061
+ const parallel = progress.metadata.parallelExecution;
1062
+ console.log(`Group ${parallel.currentGroup}/${parallel.totalGroups}`);
1063
+ console.log(`Tasks: ${parallel.completedTasks}/${parallel.totalTasks}`);
1064
+ console.log(`Successful: ${parallel.successfulTasks}`);
1065
+ }
1066
+ },
1067
+
1068
+ onStreaming: (chunk) => {
1069
+ // Multiple models may stream simultaneously
1070
+ console.log(`${chunk.modelName}: ${chunk.content}`);
1071
+
1072
+ // Update model-specific UI sections
1073
+ updateModelSection(chunk.taskId, chunk.content);
1074
+ }
1075
+ });
1076
+ ```
1077
+
1078
+ ### Advanced Streaming Configuration
1079
+
1080
+ Fine-tune streaming behavior for optimal performance:
1081
+
1082
+ ```typescript
1083
+ // Streaming configuration can be applied globally or per-prompt
1084
+ const streamingConfig = {
1085
+ enabled: true,
1086
+ aggregateParallelUpdates: false, // Separate updates per parallel task
1087
+ progressUpdateIntervalMS: 250 // Limit update frequency
1088
+ };
1089
+
1090
+ // Progress updates are automatically throttled to prevent UI flooding
1091
+ // Minimum interval between updates prevents performance issues
1092
+ ```
1093
+
1094
+ ### Integration with BaseLLM Streaming
1095
+
1096
+ The streaming system integrates seamlessly with BaseLLM capabilities:
1097
+
1098
+ ```typescript
1099
+ // The AI Prompt Runner automatically detects streaming support:
1100
+ // 1. Checks if the selected model supports streaming
1101
+ // 2. Configures BaseLLM streaming callbacks
1102
+ // 3. Aggregates streaming updates from multiple models in parallel execution
1103
+ // 4. Provides unified streaming interface regardless of underlying model
1104
+
1105
+ // Models that support streaming will automatically use it when callbacks are provided
1106
+ // Models without streaming support will provide content in the final result
1107
+ ```
1108
+
395
1109
  ## API Reference
396
1110
 
397
1111
  ### Exported Classes and Types
@@ -415,15 +1129,45 @@ Handles execution of AI prompts with advanced parallel processing, template rend
415
1129
 
416
1130
  ```typescript
417
1131
  interface AIPromptParams {
418
- prompt: AIPromptEntity; // The prompt to execute
419
- data?: any; // Template and context data
420
- modelId?: string; // Override model selection
421
- vendorId?: string; // Override vendor selection
422
- configurationId?: string; // Environment-specific config
423
- contextUser?: UserInfo; // User context
424
- skipValidation?: boolean; // Skip output validation
425
- templateData?: any; // Additional template data that augments the main data context
1132
+ prompt: AIPromptEntity; // The prompt to execute
1133
+ data?: any; // Template and context data
1134
+ modelId?: string; // Override model selection
1135
+ vendorId?: string; // Override vendor selection
1136
+ configurationId?: string; // Environment-specific config
1137
+ contextUser?: UserInfo; // User context
1138
+ skipValidation?: boolean; // Skip output validation
1139
+ templateData?: any; // Additional template data that augments the main data context
1140
+ conversationMessages?: ChatMessage[]; // Multi-turn conversation messages
1141
+ templateMessageRole?: TemplateMessageRole; // How to use rendered template ('system'|'user'|'none')
1142
+ cancellationToken?: AbortSignal; // Cancellation token for aborting execution
1143
+ onProgress?: ExecutionProgressCallback; // Progress update callback
1144
+ onStreaming?: ExecutionStreamingCallback; // Streaming content callback
426
1145
  }
1146
+
1147
+ /**
1148
+ * Progress callback function type
1149
+ */
1150
+ type ExecutionProgressCallback = (progress: {
1151
+ step: 'template_rendering' | 'model_selection' | 'execution' | 'validation' | 'parallel_coordination' | 'result_selection';
1152
+ percentage: number; // Progress percentage (0-100)
1153
+ message: string; // Human-readable status message
1154
+ metadata?: Record<string, any>; // Additional metadata about the current step
1155
+ }) => void;
1156
+
1157
+ /**
1158
+ * Streaming callback function type
1159
+ */
1160
+ type ExecutionStreamingCallback = (chunk: {
1161
+ content: string; // The content chunk received
1162
+ isComplete: boolean; // Whether this is the final chunk
1163
+ taskId?: string; // Which task/model is producing this content (for parallel execution)
1164
+ modelName?: string; // Model name producing this content
1165
+ }) => void;
1166
+
1167
+ /**
1168
+ * Template message role type
1169
+ */
1170
+ type TemplateMessageRole = 'system' | 'user' | 'none';
427
1171
  ```
428
1172
 
429
1173
  ### Extended Entity Classes
@@ -443,6 +1187,9 @@ class AIPromptCategoryEntityExtended extends AIPromptCategoryEntity {
443
1187
  ```typescript
444
1188
  interface AIPromptRunResult {
445
1189
  success: boolean; // Whether the execution was successful
1190
+ status?: ExecutionStatus; // Current execution status
1191
+ cancelled?: boolean; // Whether the execution was cancelled
1192
+ cancellationReason?: CancellationReason; // Reason for cancellation if applicable
446
1193
  rawResult?: string; // The raw result from the AI model
447
1194
  result?: any; // The parsed/validated result based on OutputType
448
1195
  errorMessage?: string; // Error message if execution failed
@@ -450,6 +1197,42 @@ interface AIPromptRunResult {
450
1197
  executionTimeMS?: number; // Total execution time in milliseconds
451
1198
  tokensUsed?: number; // Tokens used in the execution
452
1199
  validationResult?: ValidationResult; // Validation result if output validation was performed
1200
+ additionalResults?: AIPromptRunResult[]; // Additional results from parallel execution, ranked by judge
1201
+ ranking?: number; // Ranking assigned by judge (1 = best, 2 = second best, etc.)
1202
+ judgeRationale?: string; // Judge's rationale for this ranking
1203
+ modelInfo?: ModelInfo; // Model information for this result
1204
+ judgeMetadata?: JudgeMetadata; // Metadata about the judging process (only present on the main result)
1205
+ wasStreamed?: boolean; // Whether streaming was used for this execution
1206
+ cacheInfo?: { // Cache information if caching was involved
1207
+ cacheHit: boolean;
1208
+ cacheKey?: string;
1209
+ cacheSource?: string;
1210
+ };
1211
+ }
1212
+
1213
+ // Execution status enumeration
1214
+ type ExecutionStatus = 'pending' | 'running' | 'completed' | 'failed' | 'cancelled';
1215
+
1216
+ // Cancellation reason enumeration
1217
+ type CancellationReason = 'user_requested' | 'timeout' | 'error' | 'resource_limit';
1218
+
1219
+ // Model information interface
1220
+ interface ModelInfo {
1221
+ modelId: string;
1222
+ modelName: string;
1223
+ vendorId?: string;
1224
+ vendorName?: string;
1225
+ powerRank?: number;
1226
+ modelType?: string;
1227
+ }
1228
+
1229
+ // Judge metadata interface
1230
+ interface JudgeMetadata {
1231
+ judgePromptId: string;
1232
+ judgeExecutionTimeMS: number;
1233
+ judgeTokensUsed?: number;
1234
+ judgeCancelled?: boolean;
1235
+ judgeErrorMessage?: string;
453
1236
  }
454
1237
 
455
1238
  // Parallelization strategies supported by the system
@@ -566,6 +1349,65 @@ const result = await runner.ExecutePrompt({
566
1349
  4. **Monitor Performance**: Track token usage and execution times
567
1350
  5. **Parallel Wisely**: Use parallel execution for independent tasks, not dependent ones
568
1351
  6. **Handle Errors**: Implement proper retry logic and error handling
1352
+ 7. **Implement Cancellation**: Always provide cancellation tokens for user-facing operations
1353
+ 8. **Use Progress Callbacks**: Provide progress feedback for long-running operations
1354
+ 9. **Leverage Hierarchical Logging**: Use the logging hierarchy for debugging and analytics
1355
+ 10. **Configure Streaming Appropriately**: Enable streaming for responsive user experiences
1356
+ 11. **Optimize Judge Selection**: Use efficient judge prompts for parallel result selection
1357
+ 12. **Monitor Resource Usage**: Track token consumption and execution times across hierarchical runs
1358
+
1359
+ ### Implementation Guidelines
1360
+
1361
+ ```typescript
1362
+ // Comprehensive prompt execution with all new features
1363
+ const controller = new AbortController();
1364
+
1365
+ const result = await runner.ExecutePrompt({
1366
+ prompt: myPrompt,
1367
+ data: executionData,
1368
+ contextUser: currentUser,
1369
+
1370
+ // Cancellation support
1371
+ cancellationToken: controller.signal,
1372
+
1373
+ // Progress tracking
1374
+ onProgress: (progress) => {
1375
+ updateProgressIndicator(progress.percentage, progress.message);
1376
+ if (progress.metadata?.parallelExecution) {
1377
+ updateParallelStatus(progress.metadata.parallelExecution);
1378
+ }
1379
+ },
1380
+
1381
+ // Streaming for real-time updates
1382
+ onStreaming: (chunk) => {
1383
+ if (chunk.isComplete) {
1384
+ finalizePage();
1385
+ } else {
1386
+ appendContent(chunk.content);
1387
+ }
1388
+ }
1389
+ });
1390
+
1391
+ // Always check for cancellation in results
1392
+ if (result.cancelled) {
1393
+ handleCancellation(result.cancellationReason);
1394
+ } else if (result.success) {
1395
+ processResults(result);
1396
+
1397
+ // Analyze additional results from parallel execution
1398
+ if (result.additionalResults) {
1399
+ analyzeAlternativeResults(result.additionalResults);
1400
+ }
1401
+ }
1402
+
1403
+ // Use hierarchical logging data for analytics
1404
+ if (result.promptRun) {
1405
+ trackExecutionMetrics(result.promptRun);
1406
+ if (result.promptRun.RunType === 'ParallelParent') {
1407
+ analyzeParallelPerformance(result.promptRun.ID);
1408
+ }
1409
+ }
1410
+ ```
569
1411
 
570
1412
  ## Troubleshooting
571
1413
 
@@ -591,6 +1433,92 @@ const result = await runner.ExecutePrompt({
591
1433
  - Provide a valid OutputExample for structured data
592
1434
  - Consider increasing MaxRetries for complex outputs
593
1435
 
1436
+ 5. **Cancellation Not Working**
1437
+ - Verify the AbortController is properly created and signal is passed
1438
+ - Check that the cancellation token is not already aborted before execution
1439
+ - Ensure model implementations support cancellation (older models may not)
1440
+ - Review cancellation timing - very fast executions may complete before cancellation
1441
+
1442
+ 6. **Progress Updates Not Received**
1443
+ - Confirm onProgress callback is properly defined and passed to ExecutePrompt
1444
+ - Check that the callback function doesn't throw errors (which can stop updates)
1445
+ - Progress updates are throttled - very fast operations may have fewer updates
1446
+ - Parallel execution provides more detailed progress metadata
1447
+
1448
+ 7. **Streaming Not Working**
1449
+ - Verify the selected AI model supports streaming (not all models do)
1450
+ - Ensure onStreaming callback is provided in AIPromptParams
1451
+ - Check BaseLLM implementation supports streaming for the specific model
1452
+ - Review model configuration - some vendors require specific settings for streaming
1453
+
1454
+ 8. **Hierarchical Logging Missing**
1455
+ - Ensure database schema includes RunType, ParentID, and ExecutionOrder fields
1456
+ - Check that user has permissions to create AIPromptRun records
1457
+ - Verify prompt run creation isn't being skipped due to errors
1458
+ - Review logs for save failures on prompt run entities
1459
+
1460
+ 9. **Judge Selection Failing**
1461
+ - Confirm ResultSelectorPromptID is set and points to a valid, active prompt
1462
+ - Verify the judge prompt returns valid JSON with rankings array
1463
+ - Check that judge prompt has proper model associations
1464
+ - Review judge prompt timeout settings for complex evaluations
1465
+
1466
+ ### Performance Optimization
1467
+
1468
+ For optimal performance with the new features:
1469
+
1470
+ ```typescript
1471
+ // Minimize progress update frequency for high-performance scenarios
1472
+ const result = await runner.ExecutePrompt({
1473
+ prompt: myPrompt,
1474
+ data: myData,
1475
+ onProgress: (progress) => {
1476
+ // Throttle UI updates
1477
+ if (progress.percentage % 10 === 0) {
1478
+ updateUI(progress);
1479
+ }
1480
+ }
1481
+ });
1482
+
1483
+ // Use cancellation for long-running operations
1484
+ const controller = new AbortController();
1485
+ setTimeout(() => controller.abort(), 60000); // 1 minute timeout
1486
+
1487
+ // Configure parallel execution for optimal throughput
1488
+ const parallelPrompt = {
1489
+ ParallelizationMode: "ModelSpecific",
1490
+ // Configure specific models with different execution groups for coordination
1491
+ };
1492
+ ```
1493
+
1494
+ ### Debugging Hierarchical Logs
1495
+
1496
+ Use these queries to troubleshoot execution issues:
1497
+
1498
+ ```sql
1499
+ -- Find incomplete executions
1500
+ SELECT * FROM AIPromptRun
1501
+ WHERE CompletedAt IS NULL
1502
+ AND RunAt < DATEADD(minute, -5, GETDATE());
1503
+
1504
+ -- Check parallel execution hierarchy
1505
+ SELECT
1506
+ ID, RunType, ParentID, ExecutionOrder, Success, ErrorMessage
1507
+ FROM AIPromptRun
1508
+ WHERE ParentID = 'your-parent-id' OR ID = 'your-parent-id'
1509
+ ORDER BY RunType, ExecutionOrder;
1510
+
1511
+ -- Find resource usage patterns
1512
+ SELECT
1513
+ RunType,
1514
+ AVG(ExecutionTimeMS) as AvgTimeMS,
1515
+ AVG(TokensUsed) as AvgTokens,
1516
+ COUNT(*) as ExecutionCount
1517
+ FROM AIPromptRun
1518
+ WHERE RunAt > DATEADD(day, -7, GETDATE())
1519
+ GROUP BY RunType;
1520
+ ```
1521
+
594
1522
  ## License
595
1523
 
596
1524
  ISC