@blockrun/llm 3.13.1 → 3.13.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +219 -187
- package/dist/index.cjs +1121 -522
- package/dist/index.d.cts +16 -0
- package/dist/index.d.ts +16 -0
- package/dist/index.js +1121 -522
- package/package.json +2 -2
package/dist/index.js
CHANGED
|
@@ -31,7 +31,7 @@ var APIError = class extends BlockrunError {
|
|
|
31
31
|
}
|
|
32
32
|
};
|
|
33
33
|
|
|
34
|
-
// node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+
|
|
34
|
+
// node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+5d911879d7f1e_fbldkf3f3jmlwtwu53ftq2mj24/node_modules/@blockrun/router-core/dist/index.js
|
|
35
35
|
function scoreTokenCount(estimatedTokens, thresholds) {
|
|
36
36
|
if (estimatedTokens < thresholds.simple) {
|
|
37
37
|
return { name: "tokenCount", score: -1, signal: `short (${estimatedTokens} tokens)` };
|
|
@@ -365,6 +365,21 @@ function applyPromotions(tierConfigs, promotions, profile, now = /* @__PURE__ */
|
|
|
365
365
|
}
|
|
366
366
|
return result;
|
|
367
367
|
}
|
|
368
|
+
function applyUnavailableModels(tierConfigs, unavailableModels) {
|
|
369
|
+
if (!unavailableModels || unavailableModels.length === 0) return tierConfigs;
|
|
370
|
+
const dead = new Set(unavailableModels);
|
|
371
|
+
let result = tierConfigs;
|
|
372
|
+
for (const tier of Object.keys(tierConfigs)) {
|
|
373
|
+
const config = tierConfigs[tier];
|
|
374
|
+
const alive = [config.primary, ...config.fallback].filter((model) => !dead.has(model));
|
|
375
|
+
if (alive.length === 0 || alive[0] === config.primary && alive.length === config.fallback.length + 1) {
|
|
376
|
+
continue;
|
|
377
|
+
}
|
|
378
|
+
if (result === tierConfigs) result = { ...tierConfigs };
|
|
379
|
+
result[tier] = { primary: alive[0], fallback: alive.slice(1) };
|
|
380
|
+
}
|
|
381
|
+
return result;
|
|
382
|
+
}
|
|
368
383
|
var RulesStrategy = class {
|
|
369
384
|
name = "rules";
|
|
370
385
|
route(prompt, systemPrompt, maxOutputTokens, options) {
|
|
@@ -416,6 +431,7 @@ ${value.slice(-(scanLimit - prefixLength))}`;
|
|
|
416
431
|
profile = useAgenticTiers ? "agentic" : "auto";
|
|
417
432
|
}
|
|
418
433
|
tierConfigs = applyPromotions(tierConfigs, config.promotions, profile, options.now);
|
|
434
|
+
tierConfigs = applyUnavailableModels(tierConfigs, options.unavailableModels);
|
|
419
435
|
const agenticScoreValue = ruleResult.agenticScore;
|
|
420
436
|
if (estimatedTokens > config.overrides.maxTokensForceComplex) {
|
|
421
437
|
const decision2 = selectModel(
|
|
@@ -489,14 +505,15 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
489
505
|
supportsVision: true
|
|
490
506
|
},
|
|
491
507
|
"anthropic/claude-haiku-4.5": {
|
|
508
|
+
// override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
|
|
492
509
|
contextWindow: 2e5,
|
|
493
|
-
maxOutputTokens:
|
|
510
|
+
maxOutputTokens: 64e3,
|
|
494
511
|
supportsTools: true,
|
|
495
512
|
supportsVision: true
|
|
496
513
|
},
|
|
497
|
-
"anthropic/claude-opus-4.
|
|
498
|
-
contextWindow:
|
|
499
|
-
maxOutputTokens:
|
|
514
|
+
"anthropic/claude-opus-4.5": {
|
|
515
|
+
contextWindow: 2e5,
|
|
516
|
+
maxOutputTokens: 64e3,
|
|
500
517
|
supportsTools: true,
|
|
501
518
|
supportsVision: true
|
|
502
519
|
},
|
|
@@ -518,12 +535,19 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
518
535
|
supportsTools: true,
|
|
519
536
|
supportsVision: true
|
|
520
537
|
},
|
|
521
|
-
"anthropic/claude-sonnet-4.
|
|
538
|
+
"anthropic/claude-sonnet-4.5": {
|
|
522
539
|
contextWindow: 2e5,
|
|
523
540
|
maxOutputTokens: 64e3,
|
|
524
541
|
supportsTools: true,
|
|
525
542
|
supportsVision: true
|
|
526
543
|
},
|
|
544
|
+
"anthropic/claude-sonnet-4.6": {
|
|
545
|
+
// override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
|
|
546
|
+
contextWindow: 1e6,
|
|
547
|
+
maxOutputTokens: 128e3,
|
|
548
|
+
supportsTools: true,
|
|
549
|
+
supportsVision: true
|
|
550
|
+
},
|
|
527
551
|
"anthropic/claude-sonnet-5": {
|
|
528
552
|
contextWindow: 1e6,
|
|
529
553
|
maxOutputTokens: 128e3,
|
|
@@ -531,14 +555,14 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
531
555
|
supportsVision: true
|
|
532
556
|
},
|
|
533
557
|
"deepseek/deepseek-chat": {
|
|
534
|
-
contextWindow:
|
|
535
|
-
maxOutputTokens:
|
|
558
|
+
contextWindow: 1048576,
|
|
559
|
+
maxOutputTokens: 65536,
|
|
536
560
|
supportsTools: true,
|
|
537
561
|
supportsVision: false
|
|
538
562
|
},
|
|
539
563
|
"deepseek/deepseek-reasoner": {
|
|
540
|
-
contextWindow:
|
|
541
|
-
maxOutputTokens:
|
|
564
|
+
contextWindow: 1048576,
|
|
565
|
+
maxOutputTokens: 65536,
|
|
542
566
|
supportsTools: true,
|
|
543
567
|
supportsVision: false
|
|
544
568
|
},
|
|
@@ -548,62 +572,38 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
548
572
|
supportsTools: true,
|
|
549
573
|
supportsVision: false
|
|
550
574
|
},
|
|
551
|
-
"free/deepseek-v4-flash": {
|
|
552
|
-
contextWindow: 1e6,
|
|
553
|
-
maxOutputTokens: 16384,
|
|
554
|
-
supportsTools: false,
|
|
555
|
-
supportsVision: false
|
|
556
|
-
},
|
|
557
|
-
"free/gpt-oss-120b": {
|
|
558
|
-
contextWindow: 128e3,
|
|
559
|
-
maxOutputTokens: 16384,
|
|
560
|
-
supportsTools: false,
|
|
561
|
-
supportsVision: false
|
|
562
|
-
},
|
|
563
|
-
"free/gpt-oss-20b": {
|
|
564
|
-
contextWindow: 128e3,
|
|
565
|
-
maxOutputTokens: 16384,
|
|
566
|
-
supportsTools: false,
|
|
567
|
-
supportsVision: false
|
|
568
|
-
},
|
|
569
|
-
"free/seed-oss-36b": {
|
|
570
|
-
contextWindow: 131072,
|
|
571
|
-
maxOutputTokens: 16384,
|
|
572
|
-
supportsTools: false,
|
|
573
|
-
supportsVision: false
|
|
574
|
-
},
|
|
575
575
|
"google/gemini-2.5-flash": {
|
|
576
|
-
contextWindow:
|
|
576
|
+
contextWindow: 1048576,
|
|
577
577
|
maxOutputTokens: 65536,
|
|
578
578
|
supportsTools: true,
|
|
579
579
|
supportsVision: true
|
|
580
580
|
},
|
|
581
581
|
"google/gemini-2.5-flash-lite": {
|
|
582
|
-
contextWindow:
|
|
582
|
+
contextWindow: 1048576,
|
|
583
583
|
maxOutputTokens: 65536,
|
|
584
584
|
supportsTools: true,
|
|
585
585
|
supportsVision: false
|
|
586
586
|
},
|
|
587
587
|
"google/gemini-2.5-pro": {
|
|
588
|
-
contextWindow:
|
|
588
|
+
contextWindow: 1048576,
|
|
589
589
|
maxOutputTokens: 65536,
|
|
590
590
|
supportsTools: true,
|
|
591
591
|
supportsVision: true
|
|
592
592
|
},
|
|
593
593
|
"google/gemini-3-flash-preview": {
|
|
594
|
-
contextWindow:
|
|
594
|
+
contextWindow: 1048576,
|
|
595
595
|
maxOutputTokens: 65536,
|
|
596
|
-
supportsTools:
|
|
596
|
+
supportsTools: true,
|
|
597
597
|
supportsVision: true
|
|
598
598
|
},
|
|
599
599
|
"google/gemini-3.1-flash-lite": {
|
|
600
|
-
contextWindow:
|
|
601
|
-
maxOutputTokens:
|
|
600
|
+
contextWindow: 1048576,
|
|
601
|
+
maxOutputTokens: 65536,
|
|
602
602
|
supportsTools: true,
|
|
603
603
|
supportsVision: false
|
|
604
604
|
},
|
|
605
605
|
"google/gemini-3.1-pro": {
|
|
606
|
-
contextWindow:
|
|
606
|
+
contextWindow: 1048576,
|
|
607
607
|
maxOutputTokens: 65536,
|
|
608
608
|
supportsTools: true,
|
|
609
609
|
supportsVision: true
|
|
@@ -614,23 +614,29 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
614
614
|
supportsTools: true,
|
|
615
615
|
supportsVision: true
|
|
616
616
|
},
|
|
617
|
-
"
|
|
618
|
-
contextWindow:
|
|
619
|
-
maxOutputTokens:
|
|
617
|
+
"google/gemini-3.5-flash-lite": {
|
|
618
|
+
contextWindow: 1048576,
|
|
619
|
+
maxOutputTokens: 65536,
|
|
620
620
|
supportsTools: true,
|
|
621
|
-
supportsVision:
|
|
621
|
+
supportsVision: false
|
|
622
622
|
},
|
|
623
|
-
"
|
|
624
|
-
contextWindow:
|
|
623
|
+
"google/gemini-3.6-flash": {
|
|
624
|
+
contextWindow: 1048576,
|
|
625
625
|
maxOutputTokens: 65536,
|
|
626
626
|
supportsTools: true,
|
|
627
627
|
supportsVision: true
|
|
628
628
|
},
|
|
629
|
-
"
|
|
630
|
-
contextWindow:
|
|
629
|
+
"minimax/minimax-m2.7": {
|
|
630
|
+
contextWindow: 204800,
|
|
631
|
+
maxOutputTokens: 16384,
|
|
632
|
+
supportsTools: true,
|
|
633
|
+
supportsVision: false
|
|
634
|
+
},
|
|
635
|
+
"minimax/minimax-m3": {
|
|
636
|
+
contextWindow: 1048576,
|
|
631
637
|
maxOutputTokens: 65536,
|
|
632
638
|
supportsTools: true,
|
|
633
|
-
supportsVision:
|
|
639
|
+
supportsVision: false
|
|
634
640
|
},
|
|
635
641
|
"moonshot/kimi-k3": {
|
|
636
642
|
contextWindow: 1048576,
|
|
@@ -638,7 +644,63 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
638
644
|
supportsTools: true,
|
|
639
645
|
supportsVision: true
|
|
640
646
|
},
|
|
647
|
+
"nvidia/mistral-nemotron": {
|
|
648
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
649
|
+
contextWindow: 131072,
|
|
650
|
+
maxOutputTokens: 16384,
|
|
651
|
+
supportsTools: false,
|
|
652
|
+
supportsVision: false
|
|
653
|
+
},
|
|
654
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
|
|
655
|
+
contextWindow: 256e3,
|
|
656
|
+
maxOutputTokens: 16384,
|
|
657
|
+
supportsTools: false,
|
|
658
|
+
supportsVision: true
|
|
659
|
+
},
|
|
660
|
+
"nvidia/nemotron-nano-12b-v2-vl": {
|
|
661
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
662
|
+
contextWindow: 131072,
|
|
663
|
+
maxOutputTokens: 16384,
|
|
664
|
+
supportsTools: false,
|
|
665
|
+
supportsVision: true
|
|
666
|
+
},
|
|
667
|
+
"nvidia/nemotron-nano-9b-v2": {
|
|
668
|
+
contextWindow: 131072,
|
|
669
|
+
maxOutputTokens: 16384,
|
|
670
|
+
supportsTools: false,
|
|
671
|
+
supportsVision: false
|
|
672
|
+
},
|
|
673
|
+
"nvidia/step-3.7-flash": {
|
|
674
|
+
contextWindow: 131072,
|
|
675
|
+
maxOutputTokens: 16384,
|
|
676
|
+
supportsTools: false,
|
|
677
|
+
supportsVision: false
|
|
678
|
+
},
|
|
679
|
+
"openai/chat-latest": {
|
|
680
|
+
contextWindow: 128e3,
|
|
681
|
+
maxOutputTokens: 128e3,
|
|
682
|
+
supportsTools: true,
|
|
683
|
+
supportsVision: true
|
|
684
|
+
},
|
|
641
685
|
"openai/gpt-4.1": {
|
|
686
|
+
contextWindow: 128e3,
|
|
687
|
+
maxOutputTokens: 32768,
|
|
688
|
+
supportsTools: true,
|
|
689
|
+
supportsVision: true
|
|
690
|
+
},
|
|
691
|
+
"openai/gpt-4.1-mini": {
|
|
692
|
+
contextWindow: 128e3,
|
|
693
|
+
maxOutputTokens: 32768,
|
|
694
|
+
supportsTools: true,
|
|
695
|
+
supportsVision: false
|
|
696
|
+
},
|
|
697
|
+
"openai/gpt-4.1-nano": {
|
|
698
|
+
contextWindow: 128e3,
|
|
699
|
+
maxOutputTokens: 32768,
|
|
700
|
+
supportsTools: true,
|
|
701
|
+
supportsVision: false
|
|
702
|
+
},
|
|
703
|
+
"openai/gpt-4o": {
|
|
642
704
|
contextWindow: 128e3,
|
|
643
705
|
maxOutputTokens: 16384,
|
|
644
706
|
supportsTools: true,
|
|
@@ -652,17 +714,44 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
652
714
|
},
|
|
653
715
|
"openai/gpt-5-mini": {
|
|
654
716
|
contextWindow: 2e5,
|
|
655
|
-
maxOutputTokens:
|
|
717
|
+
maxOutputTokens: 128e3,
|
|
656
718
|
supportsTools: true,
|
|
657
719
|
supportsVision: false
|
|
658
720
|
},
|
|
721
|
+
"openai/gpt-5.2": {
|
|
722
|
+
contextWindow: 4e5,
|
|
723
|
+
maxOutputTokens: 128e3,
|
|
724
|
+
supportsTools: true,
|
|
725
|
+
supportsVision: true
|
|
726
|
+
},
|
|
727
|
+
"openai/gpt-5.2-pro": {
|
|
728
|
+
// supportsTools: not probed — fails closed
|
|
729
|
+
contextWindow: 4e5,
|
|
730
|
+
maxOutputTokens: 128e3,
|
|
731
|
+
supportsTools: false,
|
|
732
|
+
supportsVision: true
|
|
733
|
+
},
|
|
734
|
+
"openai/gpt-5.3": {
|
|
735
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
736
|
+
contextWindow: 128e3,
|
|
737
|
+
maxOutputTokens: 128e3,
|
|
738
|
+
supportsTools: false,
|
|
739
|
+
supportsVision: true
|
|
740
|
+
},
|
|
659
741
|
"openai/gpt-5.3-codex": {
|
|
742
|
+
// supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe measured an incident, not the model. Codex's function calling is established by the 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot claiming the model cannot call tools.
|
|
660
743
|
contextWindow: 4e5,
|
|
661
744
|
maxOutputTokens: 128e3,
|
|
662
745
|
supportsTools: true,
|
|
663
746
|
supportsVision: false
|
|
664
747
|
},
|
|
665
748
|
"openai/gpt-5.4": {
|
|
749
|
+
contextWindow: 105e4,
|
|
750
|
+
maxOutputTokens: 128e3,
|
|
751
|
+
supportsTools: true,
|
|
752
|
+
supportsVision: true
|
|
753
|
+
},
|
|
754
|
+
"openai/gpt-5.4-mini": {
|
|
666
755
|
contextWindow: 4e5,
|
|
667
756
|
maxOutputTokens: 128e3,
|
|
668
757
|
supportsTools: true,
|
|
@@ -670,30 +759,92 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
670
759
|
},
|
|
671
760
|
"openai/gpt-5.4-nano": {
|
|
672
761
|
contextWindow: 105e4,
|
|
673
|
-
maxOutputTokens:
|
|
762
|
+
maxOutputTokens: 128e3,
|
|
674
763
|
supportsTools: true,
|
|
675
764
|
supportsVision: false
|
|
676
765
|
},
|
|
766
|
+
"openai/gpt-5.4-pro": {
|
|
767
|
+
// supportsTools: not probed — fails closed
|
|
768
|
+
contextWindow: 105e4,
|
|
769
|
+
maxOutputTokens: 128e3,
|
|
770
|
+
supportsTools: false,
|
|
771
|
+
supportsVision: true
|
|
772
|
+
},
|
|
677
773
|
"openai/gpt-5.5": {
|
|
678
774
|
contextWindow: 105e4,
|
|
679
775
|
maxOutputTokens: 128e3,
|
|
680
776
|
supportsTools: true,
|
|
681
777
|
supportsVision: true
|
|
682
778
|
},
|
|
779
|
+
"openai/gpt-5.5-pro": {
|
|
780
|
+
// supportsTools: not probed — fails closed
|
|
781
|
+
contextWindow: 105e4,
|
|
782
|
+
maxOutputTokens: 128e3,
|
|
783
|
+
supportsTools: false,
|
|
784
|
+
supportsVision: true
|
|
785
|
+
},
|
|
786
|
+
"openai/gpt-5.6-luna": {
|
|
787
|
+
contextWindow: 105e4,
|
|
788
|
+
maxOutputTokens: 128e3,
|
|
789
|
+
supportsTools: true,
|
|
790
|
+
supportsVision: true
|
|
791
|
+
},
|
|
792
|
+
"openai/gpt-5.6-luna-pro": {
|
|
793
|
+
contextWindow: 105e4,
|
|
794
|
+
maxOutputTokens: 128e3,
|
|
795
|
+
supportsTools: false,
|
|
796
|
+
supportsVision: true
|
|
797
|
+
},
|
|
798
|
+
"openai/gpt-5.6-sol": {
|
|
799
|
+
contextWindow: 105e4,
|
|
800
|
+
maxOutputTokens: 128e3,
|
|
801
|
+
supportsTools: true,
|
|
802
|
+
supportsVision: true
|
|
803
|
+
},
|
|
804
|
+
"openai/gpt-5.6-sol-pro": {
|
|
805
|
+
contextWindow: 105e4,
|
|
806
|
+
maxOutputTokens: 128e3,
|
|
807
|
+
supportsTools: true,
|
|
808
|
+
supportsVision: true
|
|
809
|
+
},
|
|
683
810
|
"openai/gpt-5.6-terra": {
|
|
684
811
|
contextWindow: 105e4,
|
|
685
812
|
maxOutputTokens: 128e3,
|
|
686
813
|
supportsTools: true,
|
|
687
814
|
supportsVision: true
|
|
688
815
|
},
|
|
816
|
+
"openai/gpt-5.6-terra-pro": {
|
|
817
|
+
contextWindow: 105e4,
|
|
818
|
+
maxOutputTokens: 128e3,
|
|
819
|
+
supportsTools: true,
|
|
820
|
+
supportsVision: true
|
|
821
|
+
},
|
|
822
|
+
"openai/o1": {
|
|
823
|
+
contextWindow: 2e5,
|
|
824
|
+
maxOutputTokens: 1e5,
|
|
825
|
+
supportsTools: true,
|
|
826
|
+
supportsVision: false
|
|
827
|
+
},
|
|
689
828
|
"openai/o3": {
|
|
690
829
|
contextWindow: 2e5,
|
|
691
830
|
maxOutputTokens: 1e5,
|
|
692
831
|
supportsTools: true,
|
|
693
832
|
supportsVision: false
|
|
694
833
|
},
|
|
834
|
+
"openai/o3-mini": {
|
|
835
|
+
contextWindow: 128e3,
|
|
836
|
+
maxOutputTokens: 1e5,
|
|
837
|
+
supportsTools: true,
|
|
838
|
+
supportsVision: false
|
|
839
|
+
},
|
|
695
840
|
"openai/o4-mini": {
|
|
696
841
|
contextWindow: 128e3,
|
|
842
|
+
maxOutputTokens: 1e5,
|
|
843
|
+
supportsTools: true,
|
|
844
|
+
supportsVision: false
|
|
845
|
+
},
|
|
846
|
+
"qwen/qwen3.7-flash": {
|
|
847
|
+
contextWindow: 1e6,
|
|
697
848
|
maxOutputTokens: 65536,
|
|
698
849
|
supportsTools: true,
|
|
699
850
|
supportsVision: false
|
|
@@ -704,299 +855,605 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
704
855
|
supportsTools: true,
|
|
705
856
|
supportsVision: false
|
|
706
857
|
},
|
|
707
|
-
"
|
|
708
|
-
contextWindow:
|
|
709
|
-
maxOutputTokens:
|
|
858
|
+
"qwen/qwen3.7-plus": {
|
|
859
|
+
contextWindow: 1e6,
|
|
860
|
+
maxOutputTokens: 131072,
|
|
710
861
|
supportsTools: true,
|
|
711
862
|
supportsVision: false
|
|
712
863
|
},
|
|
713
|
-
"
|
|
714
|
-
contextWindow:
|
|
715
|
-
maxOutputTokens:
|
|
864
|
+
"tencent/hy3": {
|
|
865
|
+
contextWindow: 262144,
|
|
866
|
+
maxOutputTokens: 128e3,
|
|
716
867
|
supportsTools: true,
|
|
717
868
|
supportsVision: false
|
|
718
869
|
},
|
|
719
|
-
"xai/grok-4
|
|
720
|
-
contextWindow:
|
|
870
|
+
"xai/grok-4.3": {
|
|
871
|
+
contextWindow: 1e6,
|
|
721
872
|
maxOutputTokens: 16384,
|
|
722
873
|
supportsTools: true,
|
|
723
|
-
supportsVision:
|
|
874
|
+
supportsVision: true
|
|
724
875
|
},
|
|
725
|
-
"xai/grok-4
|
|
726
|
-
contextWindow:
|
|
876
|
+
"xai/grok-4.5": {
|
|
877
|
+
contextWindow: 5e5,
|
|
727
878
|
maxOutputTokens: 16384,
|
|
728
879
|
supportsTools: true,
|
|
729
|
-
supportsVision:
|
|
880
|
+
supportsVision: true
|
|
730
881
|
},
|
|
731
|
-
"xai/grok-
|
|
732
|
-
contextWindow:
|
|
882
|
+
"xai/grok-build-0.1": {
|
|
883
|
+
contextWindow: 256e3,
|
|
733
884
|
maxOutputTokens: 16384,
|
|
734
885
|
supportsTools: true,
|
|
735
886
|
supportsVision: false
|
|
736
887
|
},
|
|
737
|
-
"
|
|
738
|
-
contextWindow:
|
|
739
|
-
maxOutputTokens:
|
|
740
|
-
supportsTools: true,
|
|
741
|
-
supportsVision: false
|
|
888
|
+
"xiaomi/mimo-v2.5-pro": {
|
|
889
|
+
contextWindow: 1048576,
|
|
890
|
+
maxOutputTokens: 131072,
|
|
891
|
+
supportsTools: true,
|
|
892
|
+
supportsVision: false
|
|
893
|
+
},
|
|
894
|
+
"zai/glm-5": {
|
|
895
|
+
contextWindow: 2e5,
|
|
896
|
+
maxOutputTokens: 128e3,
|
|
897
|
+
supportsTools: true,
|
|
898
|
+
supportsVision: false
|
|
899
|
+
},
|
|
900
|
+
"zai/glm-5-turbo": {
|
|
901
|
+
contextWindow: 2e5,
|
|
902
|
+
maxOutputTokens: 128e3,
|
|
903
|
+
supportsTools: true,
|
|
904
|
+
supportsVision: false
|
|
905
|
+
},
|
|
906
|
+
"zai/glm-5.1": {
|
|
907
|
+
contextWindow: 2e5,
|
|
908
|
+
maxOutputTokens: 128e3,
|
|
909
|
+
supportsTools: true,
|
|
910
|
+
supportsVision: false
|
|
911
|
+
},
|
|
912
|
+
"zai/glm-5.2": {
|
|
913
|
+
contextWindow: 1e6,
|
|
914
|
+
maxOutputTokens: 131072,
|
|
915
|
+
supportsTools: true,
|
|
916
|
+
supportsVision: false
|
|
917
|
+
},
|
|
918
|
+
"zai/glm-5.3": {
|
|
919
|
+
contextWindow: 1e6,
|
|
920
|
+
maxOutputTokens: 131072,
|
|
921
|
+
supportsTools: true,
|
|
922
|
+
supportsVision: false
|
|
923
|
+
},
|
|
924
|
+
"zai/glm-5.3-flash": {
|
|
925
|
+
contextWindow: 1e6,
|
|
926
|
+
maxOutputTokens: 131072,
|
|
927
|
+
supportsTools: true,
|
|
928
|
+
supportsVision: true
|
|
929
|
+
}
|
|
930
|
+
});
|
|
931
|
+
var model_profiles_generated_default = {
|
|
932
|
+
"anthropic/claude-fable-5": {
|
|
933
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
934
|
+
latencyMs: 9298.5,
|
|
935
|
+
p95LatencyMs: 9873.4,
|
|
936
|
+
outputTokensPerSecond: 55.17,
|
|
937
|
+
errorRate: 0,
|
|
938
|
+
samples: 3
|
|
939
|
+
},
|
|
940
|
+
"anthropic/claude-haiku-4.5": {
|
|
941
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
942
|
+
latencyMs: 3157.4,
|
|
943
|
+
p95LatencyMs: 3170.7,
|
|
944
|
+
outputTokensPerSecond: 162.16,
|
|
945
|
+
errorRate: 0,
|
|
946
|
+
samples: 3
|
|
947
|
+
},
|
|
948
|
+
"anthropic/claude-opus-4.5": {
|
|
949
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
950
|
+
latencyMs: 6497.7,
|
|
951
|
+
p95LatencyMs: 6953.7,
|
|
952
|
+
outputTokensPerSecond: 78.99,
|
|
953
|
+
errorRate: 0,
|
|
954
|
+
samples: 3
|
|
955
|
+
},
|
|
956
|
+
"anthropic/claude-opus-4.7": {
|
|
957
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
958
|
+
latencyMs: 5316.5,
|
|
959
|
+
p95LatencyMs: 6121.5,
|
|
960
|
+
outputTokensPerSecond: 97.34,
|
|
961
|
+
errorRate: 0,
|
|
962
|
+
samples: 3
|
|
963
|
+
},
|
|
964
|
+
"anthropic/claude-opus-4.8": {
|
|
965
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
966
|
+
latencyMs: 6216.1,
|
|
967
|
+
p95LatencyMs: 6847.7,
|
|
968
|
+
outputTokensPerSecond: 82.81,
|
|
969
|
+
errorRate: 0,
|
|
970
|
+
samples: 3
|
|
971
|
+
},
|
|
972
|
+
"anthropic/claude-opus-5": {
|
|
973
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
974
|
+
latencyMs: 7309,
|
|
975
|
+
p95LatencyMs: 7745.2,
|
|
976
|
+
outputTokensPerSecond: 70.17,
|
|
977
|
+
errorRate: 0,
|
|
978
|
+
samples: 3
|
|
979
|
+
},
|
|
980
|
+
"anthropic/claude-sonnet-4.5": {
|
|
981
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
982
|
+
latencyMs: 6330.4,
|
|
983
|
+
p95LatencyMs: 6631.6,
|
|
984
|
+
outputTokensPerSecond: 81.03,
|
|
985
|
+
errorRate: 0,
|
|
986
|
+
samples: 3
|
|
987
|
+
},
|
|
988
|
+
"anthropic/claude-sonnet-4.6": {
|
|
989
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
990
|
+
latencyMs: 6508,
|
|
991
|
+
p95LatencyMs: 6698.3,
|
|
992
|
+
outputTokensPerSecond: 78.6,
|
|
993
|
+
errorRate: 0,
|
|
994
|
+
samples: 3
|
|
995
|
+
},
|
|
996
|
+
"anthropic/claude-sonnet-5": {
|
|
997
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
998
|
+
latencyMs: 6165.4,
|
|
999
|
+
p95LatencyMs: 6582.9,
|
|
1000
|
+
outputTokensPerSecond: 83.62,
|
|
1001
|
+
errorRate: 0,
|
|
1002
|
+
samples: 3
|
|
1003
|
+
},
|
|
1004
|
+
"deepseek/deepseek-chat": {
|
|
1005
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1006
|
+
latencyMs: 4351.4,
|
|
1007
|
+
p95LatencyMs: 4543.7,
|
|
1008
|
+
outputTokensPerSecond: 117.78,
|
|
1009
|
+
errorRate: 0,
|
|
1010
|
+
samples: 3
|
|
1011
|
+
},
|
|
1012
|
+
"deepseek/deepseek-reasoner": {
|
|
1013
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1014
|
+
latencyMs: 5201.2,
|
|
1015
|
+
p95LatencyMs: 6079.6,
|
|
1016
|
+
outputTokensPerSecond: 99.77,
|
|
1017
|
+
errorRate: 0,
|
|
1018
|
+
samples: 3
|
|
1019
|
+
},
|
|
1020
|
+
"deepseek/deepseek-v4-pro": {
|
|
1021
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1022
|
+
latencyMs: 8781.2,
|
|
1023
|
+
p95LatencyMs: 9881.1,
|
|
1024
|
+
outputTokensPerSecond: 58.98,
|
|
1025
|
+
errorRate: 0,
|
|
1026
|
+
samples: 3
|
|
1027
|
+
},
|
|
1028
|
+
"google/gemini-2.5-flash": {
|
|
1029
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1030
|
+
latencyMs: 5416.4,
|
|
1031
|
+
p95LatencyMs: 6442.8,
|
|
1032
|
+
outputTokensPerSecond: 213.07,
|
|
1033
|
+
errorRate: 0,
|
|
1034
|
+
samples: 3
|
|
1035
|
+
},
|
|
1036
|
+
"google/gemini-2.5-flash-lite": {
|
|
1037
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1038
|
+
latencyMs: 5002.6,
|
|
1039
|
+
p95LatencyMs: 5780.3,
|
|
1040
|
+
outputTokensPerSecond: 408.43,
|
|
1041
|
+
errorRate: 0,
|
|
1042
|
+
samples: 3
|
|
1043
|
+
},
|
|
1044
|
+
"google/gemini-2.5-pro": {
|
|
1045
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1046
|
+
latencyMs: 28169.5,
|
|
1047
|
+
p95LatencyMs: 29491.4,
|
|
1048
|
+
outputTokensPerSecond: 147.3,
|
|
1049
|
+
errorRate: 0,
|
|
1050
|
+
samples: 3
|
|
1051
|
+
},
|
|
1052
|
+
"google/gemini-3-flash-preview": {
|
|
1053
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1054
|
+
latencyMs: 4717.1,
|
|
1055
|
+
p95LatencyMs: 5037.1,
|
|
1056
|
+
outputTokensPerSecond: 198.71,
|
|
1057
|
+
errorRate: 0,
|
|
1058
|
+
samples: 3
|
|
1059
|
+
},
|
|
1060
|
+
"google/gemini-3.1-flash-lite": {
|
|
1061
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1062
|
+
latencyMs: 2855.8,
|
|
1063
|
+
p95LatencyMs: 3172.7,
|
|
1064
|
+
outputTokensPerSecond: 286.91,
|
|
1065
|
+
errorRate: 0,
|
|
1066
|
+
samples: 3
|
|
1067
|
+
},
|
|
1068
|
+
"google/gemini-3.1-pro": {
|
|
1069
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1070
|
+
latencyMs: 24194.1,
|
|
1071
|
+
p95LatencyMs: 27269.6,
|
|
1072
|
+
outputTokensPerSecond: 109.47,
|
|
1073
|
+
errorRate: 0,
|
|
1074
|
+
samples: 3
|
|
1075
|
+
},
|
|
1076
|
+
"google/gemini-3.5-flash": {
|
|
1077
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1078
|
+
latencyMs: 5320.6,
|
|
1079
|
+
p95LatencyMs: 5429.8,
|
|
1080
|
+
outputTokensPerSecond: 226.21,
|
|
1081
|
+
errorRate: 0,
|
|
1082
|
+
samples: 3
|
|
1083
|
+
},
|
|
1084
|
+
"google/gemini-3.5-flash-lite": {
|
|
1085
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1086
|
+
latencyMs: 3515.8,
|
|
1087
|
+
p95LatencyMs: 4363.4,
|
|
1088
|
+
outputTokensPerSecond: 248.9,
|
|
1089
|
+
errorRate: 0,
|
|
1090
|
+
samples: 3
|
|
1091
|
+
},
|
|
1092
|
+
"google/gemini-3.6-flash": {
|
|
1093
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1094
|
+
latencyMs: 13020,
|
|
1095
|
+
p95LatencyMs: 15383.1,
|
|
1096
|
+
outputTokensPerSecond: 187.87,
|
|
1097
|
+
errorRate: 0,
|
|
1098
|
+
samples: 3
|
|
1099
|
+
},
|
|
1100
|
+
"minimax/minimax-m2.7": {
|
|
1101
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1102
|
+
latencyMs: 8761.1,
|
|
1103
|
+
p95LatencyMs: 10199.3,
|
|
1104
|
+
outputTokensPerSecond: 59.18,
|
|
1105
|
+
errorRate: 0,
|
|
1106
|
+
samples: 3
|
|
1107
|
+
},
|
|
1108
|
+
"minimax/minimax-m3": {
|
|
1109
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1110
|
+
latencyMs: 11101.9,
|
|
1111
|
+
p95LatencyMs: 26087.1,
|
|
1112
|
+
outputTokensPerSecond: 101.12,
|
|
1113
|
+
errorRate: 0,
|
|
1114
|
+
samples: 3
|
|
1115
|
+
},
|
|
1116
|
+
"moonshot/kimi-k3": {
|
|
1117
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1118
|
+
latencyMs: 24498.9,
|
|
1119
|
+
p95LatencyMs: 40365.3,
|
|
1120
|
+
outputTokensPerSecond: 25.11,
|
|
1121
|
+
errorRate: 0,
|
|
1122
|
+
samples: 3
|
|
1123
|
+
},
|
|
1124
|
+
"nvidia/mistral-nemotron": {
|
|
1125
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1126
|
+
latencyMs: 7349.6,
|
|
1127
|
+
p95LatencyMs: 9932.3,
|
|
1128
|
+
outputTokensPerSecond: 79.48,
|
|
1129
|
+
errorRate: 0.3333,
|
|
1130
|
+
samples: 3
|
|
1131
|
+
},
|
|
1132
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
|
|
1133
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1134
|
+
latencyMs: 9324.6,
|
|
1135
|
+
p95LatencyMs: 12992,
|
|
1136
|
+
outputTokensPerSecond: 64.96,
|
|
1137
|
+
errorRate: 0.3333,
|
|
1138
|
+
samples: 3
|
|
1139
|
+
},
|
|
1140
|
+
"nvidia/nemotron-nano-12b-v2-vl": {
|
|
1141
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1142
|
+
latencyMs: 5846.9,
|
|
1143
|
+
p95LatencyMs: 5846.9,
|
|
1144
|
+
outputTokensPerSecond: 87.57,
|
|
1145
|
+
errorRate: 0.6667,
|
|
1146
|
+
samples: 3
|
|
1147
|
+
},
|
|
1148
|
+
"nvidia/nemotron-nano-9b-v2": {
|
|
1149
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1150
|
+
latencyMs: 5282.5,
|
|
1151
|
+
p95LatencyMs: 5282.5,
|
|
1152
|
+
outputTokensPerSecond: 96.92,
|
|
1153
|
+
errorRate: 0.6667,
|
|
1154
|
+
samples: 3
|
|
1155
|
+
},
|
|
1156
|
+
"nvidia/step-3.7-flash": {
|
|
1157
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1158
|
+
latencyMs: 4617.4,
|
|
1159
|
+
p95LatencyMs: 5237.4,
|
|
1160
|
+
outputTokensPerSecond: 112.92,
|
|
1161
|
+
errorRate: 0.3333,
|
|
1162
|
+
samples: 3
|
|
1163
|
+
},
|
|
1164
|
+
"openai/chat-latest": {
|
|
1165
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1166
|
+
latencyMs: 3690.9,
|
|
1167
|
+
p95LatencyMs: 4344,
|
|
1168
|
+
outputTokensPerSecond: 111.85,
|
|
1169
|
+
errorRate: 0,
|
|
1170
|
+
samples: 3
|
|
1171
|
+
},
|
|
1172
|
+
"openai/gpt-4.1": {
|
|
1173
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1174
|
+
latencyMs: 3527.9,
|
|
1175
|
+
p95LatencyMs: 3831.7,
|
|
1176
|
+
outputTokensPerSecond: 141.27,
|
|
1177
|
+
errorRate: 0,
|
|
1178
|
+
samples: 3
|
|
1179
|
+
},
|
|
1180
|
+
"openai/gpt-4.1-mini": {
|
|
1181
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1182
|
+
latencyMs: 4268.2,
|
|
1183
|
+
p95LatencyMs: 5101.5,
|
|
1184
|
+
outputTokensPerSecond: 103.42,
|
|
1185
|
+
errorRate: 0,
|
|
1186
|
+
samples: 3
|
|
742
1187
|
},
|
|
743
|
-
"
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
1188
|
+
"openai/gpt-4.1-nano": {
|
|
1189
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1190
|
+
latencyMs: 3088.3,
|
|
1191
|
+
p95LatencyMs: 3369.2,
|
|
1192
|
+
outputTokensPerSecond: 150.31,
|
|
1193
|
+
errorRate: 0,
|
|
1194
|
+
samples: 3
|
|
748
1195
|
},
|
|
749
|
-
"
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
1196
|
+
"openai/gpt-4o": {
|
|
1197
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1198
|
+
latencyMs: 2995.2,
|
|
1199
|
+
p95LatencyMs: 3174.2,
|
|
1200
|
+
outputTokensPerSecond: 171.32,
|
|
1201
|
+
errorRate: 0,
|
|
1202
|
+
samples: 3
|
|
754
1203
|
},
|
|
755
|
-
"
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
}
|
|
761
|
-
});
|
|
762
|
-
var model_profiles_generated_default = {
|
|
763
|
-
"openai/gpt-5.5": {
|
|
764
|
-
measuredAt: "2026-07-21T10:21:31Z",
|
|
765
|
-
latencyMs: 6243.1,
|
|
766
|
-
p95LatencyMs: 9865,
|
|
767
|
-
outputTokensPerSecond: 12.53,
|
|
1204
|
+
"openai/gpt-4o-mini": {
|
|
1205
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1206
|
+
latencyMs: 4751.5,
|
|
1207
|
+
p95LatencyMs: 4930.4,
|
|
1208
|
+
outputTokensPerSecond: 107.84,
|
|
768
1209
|
errorRate: 0,
|
|
769
1210
|
samples: 3
|
|
770
1211
|
},
|
|
771
|
-
"openai/gpt-5
|
|
772
|
-
measuredAt: "2026-
|
|
773
|
-
latencyMs:
|
|
774
|
-
p95LatencyMs:
|
|
775
|
-
outputTokensPerSecond:
|
|
1212
|
+
"openai/gpt-5-mini": {
|
|
1213
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1214
|
+
latencyMs: 4558.1,
|
|
1215
|
+
p95LatencyMs: 5081.9,
|
|
1216
|
+
outputTokensPerSecond: 113.25,
|
|
776
1217
|
errorRate: 0,
|
|
777
1218
|
samples: 3
|
|
778
1219
|
},
|
|
779
|
-
"openai/gpt-5.
|
|
780
|
-
measuredAt: "2026-
|
|
781
|
-
latencyMs:
|
|
782
|
-
p95LatencyMs:
|
|
783
|
-
outputTokensPerSecond:
|
|
784
|
-
errorRate: 0
|
|
1220
|
+
"openai/gpt-5.2": {
|
|
1221
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1222
|
+
latencyMs: 5436.6,
|
|
1223
|
+
p95LatencyMs: 5928.8,
|
|
1224
|
+
outputTokensPerSecond: 95.47,
|
|
1225
|
+
errorRate: 0,
|
|
785
1226
|
samples: 3
|
|
786
1227
|
},
|
|
787
1228
|
"openai/gpt-5.3-codex": {
|
|
788
|
-
measuredAt: "2026-
|
|
789
|
-
latencyMs:
|
|
790
|
-
p95LatencyMs:
|
|
791
|
-
outputTokensPerSecond:
|
|
792
|
-
errorRate: 0,
|
|
1229
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1230
|
+
latencyMs: 15290.4,
|
|
1231
|
+
p95LatencyMs: 15290.4,
|
|
1232
|
+
outputTokensPerSecond: 33.49,
|
|
1233
|
+
errorRate: 0.6667,
|
|
793
1234
|
samples: 3
|
|
794
1235
|
},
|
|
795
|
-
"
|
|
796
|
-
measuredAt: "2026-
|
|
797
|
-
latencyMs:
|
|
798
|
-
p95LatencyMs:
|
|
799
|
-
outputTokensPerSecond:
|
|
1236
|
+
"openai/gpt-5.4": {
|
|
1237
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1238
|
+
latencyMs: 5596,
|
|
1239
|
+
p95LatencyMs: 5919.4,
|
|
1240
|
+
outputTokensPerSecond: 91.67,
|
|
800
1241
|
errorRate: 0,
|
|
801
1242
|
samples: 3
|
|
802
1243
|
},
|
|
803
|
-
"
|
|
804
|
-
measuredAt: "2026-
|
|
805
|
-
latencyMs:
|
|
806
|
-
p95LatencyMs:
|
|
807
|
-
outputTokensPerSecond:
|
|
1244
|
+
"openai/gpt-5.4-mini": {
|
|
1245
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1246
|
+
latencyMs: 3377.8,
|
|
1247
|
+
p95LatencyMs: 3646.8,
|
|
1248
|
+
outputTokensPerSecond: 138.08,
|
|
808
1249
|
errorRate: 0,
|
|
809
1250
|
samples: 3
|
|
810
1251
|
},
|
|
811
|
-
"
|
|
812
|
-
measuredAt: "2026-
|
|
813
|
-
latencyMs:
|
|
814
|
-
p95LatencyMs:
|
|
815
|
-
outputTokensPerSecond:
|
|
1252
|
+
"openai/gpt-5.4-nano": {
|
|
1253
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1254
|
+
latencyMs: 4040.4,
|
|
1255
|
+
p95LatencyMs: 4205.9,
|
|
1256
|
+
outputTokensPerSecond: 118.52,
|
|
816
1257
|
errorRate: 0,
|
|
817
1258
|
samples: 3
|
|
818
1259
|
},
|
|
819
|
-
"
|
|
820
|
-
measuredAt: "2026-
|
|
821
|
-
latencyMs:
|
|
822
|
-
p95LatencyMs:
|
|
823
|
-
outputTokensPerSecond:
|
|
1260
|
+
"openai/gpt-5.5": {
|
|
1261
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1262
|
+
latencyMs: 6367.8,
|
|
1263
|
+
p95LatencyMs: 7330.7,
|
|
1264
|
+
outputTokensPerSecond: 81.29,
|
|
824
1265
|
errorRate: 0,
|
|
825
1266
|
samples: 3
|
|
826
1267
|
},
|
|
827
|
-
"
|
|
828
|
-
measuredAt: "2026-
|
|
829
|
-
latencyMs:
|
|
830
|
-
p95LatencyMs:
|
|
831
|
-
outputTokensPerSecond:
|
|
1268
|
+
"openai/gpt-5.6-luna": {
|
|
1269
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1270
|
+
latencyMs: 6064.5,
|
|
1271
|
+
p95LatencyMs: 7347.3,
|
|
1272
|
+
outputTokensPerSecond: 87.93,
|
|
832
1273
|
errorRate: 0,
|
|
833
1274
|
samples: 3
|
|
834
1275
|
},
|
|
835
|
-
"
|
|
836
|
-
measuredAt: "2026-
|
|
837
|
-
latencyMs:
|
|
838
|
-
p95LatencyMs:
|
|
839
|
-
outputTokensPerSecond:
|
|
1276
|
+
"openai/gpt-5.6-luna-pro": {
|
|
1277
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1278
|
+
latencyMs: 13914.9,
|
|
1279
|
+
p95LatencyMs: 13914.9,
|
|
1280
|
+
outputTokensPerSecond: 36.79,
|
|
1281
|
+
errorRate: 0.6667,
|
|
1282
|
+
samples: 3
|
|
1283
|
+
},
|
|
1284
|
+
"openai/gpt-5.6-sol": {
|
|
1285
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1286
|
+
latencyMs: 7720.2,
|
|
1287
|
+
p95LatencyMs: 9108.2,
|
|
1288
|
+
outputTokensPerSecond: 67.47,
|
|
840
1289
|
errorRate: 0,
|
|
841
1290
|
samples: 3
|
|
842
1291
|
},
|
|
843
|
-
"
|
|
844
|
-
measuredAt: "2026-
|
|
845
|
-
latencyMs:
|
|
846
|
-
p95LatencyMs:
|
|
847
|
-
outputTokensPerSecond:
|
|
1292
|
+
"openai/gpt-5.6-sol-pro": {
|
|
1293
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1294
|
+
latencyMs: 11442.7,
|
|
1295
|
+
p95LatencyMs: 13363.1,
|
|
1296
|
+
outputTokensPerSecond: 148.75,
|
|
848
1297
|
errorRate: 0,
|
|
849
1298
|
samples: 3
|
|
850
1299
|
},
|
|
851
|
-
"
|
|
852
|
-
measuredAt: "2026-
|
|
853
|
-
latencyMs:
|
|
854
|
-
p95LatencyMs:
|
|
855
|
-
outputTokensPerSecond:
|
|
1300
|
+
"openai/gpt-5.6-terra": {
|
|
1301
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1302
|
+
latencyMs: 4941,
|
|
1303
|
+
p95LatencyMs: 5095.3,
|
|
1304
|
+
outputTokensPerSecond: 103.69,
|
|
856
1305
|
errorRate: 0,
|
|
857
1306
|
samples: 3
|
|
858
1307
|
},
|
|
859
|
-
"
|
|
860
|
-
measuredAt: "2026-
|
|
861
|
-
latencyMs:
|
|
862
|
-
p95LatencyMs:
|
|
863
|
-
outputTokensPerSecond:
|
|
1308
|
+
"openai/gpt-5.6-terra-pro": {
|
|
1309
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1310
|
+
latencyMs: 3574.1,
|
|
1311
|
+
p95LatencyMs: 4126.3,
|
|
1312
|
+
outputTokensPerSecond: 133.59,
|
|
864
1313
|
errorRate: 0,
|
|
865
1314
|
samples: 3
|
|
866
1315
|
},
|
|
867
|
-
"
|
|
868
|
-
measuredAt: "2026-
|
|
869
|
-
latencyMs:
|
|
870
|
-
p95LatencyMs:
|
|
871
|
-
outputTokensPerSecond:
|
|
1316
|
+
"openai/o1": {
|
|
1317
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1318
|
+
latencyMs: 4324.9,
|
|
1319
|
+
p95LatencyMs: 5838.1,
|
|
1320
|
+
outputTokensPerSecond: 125.86,
|
|
872
1321
|
errorRate: 0,
|
|
873
1322
|
samples: 3
|
|
874
1323
|
},
|
|
875
|
-
"
|
|
876
|
-
measuredAt: "2026-
|
|
877
|
-
latencyMs:
|
|
878
|
-
p95LatencyMs:
|
|
879
|
-
outputTokensPerSecond:
|
|
1324
|
+
"openai/o3": {
|
|
1325
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1326
|
+
latencyMs: 5463.4,
|
|
1327
|
+
p95LatencyMs: 5613.1,
|
|
1328
|
+
outputTokensPerSecond: 93.8,
|
|
880
1329
|
errorRate: 0,
|
|
881
1330
|
samples: 3
|
|
882
1331
|
},
|
|
883
|
-
"
|
|
884
|
-
measuredAt: "2026-
|
|
885
|
-
latencyMs:
|
|
886
|
-
p95LatencyMs:
|
|
887
|
-
outputTokensPerSecond:
|
|
1332
|
+
"openai/o3-mini": {
|
|
1333
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1334
|
+
latencyMs: 2912.7,
|
|
1335
|
+
p95LatencyMs: 3092.1,
|
|
1336
|
+
outputTokensPerSecond: 176.49,
|
|
888
1337
|
errorRate: 0,
|
|
889
1338
|
samples: 3
|
|
890
1339
|
},
|
|
891
|
-
"
|
|
892
|
-
measuredAt: "2026-
|
|
893
|
-
latencyMs:
|
|
894
|
-
p95LatencyMs:
|
|
895
|
-
outputTokensPerSecond:
|
|
896
|
-
errorRate: 0
|
|
1340
|
+
"openai/o4-mini": {
|
|
1341
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1342
|
+
latencyMs: 4958.7,
|
|
1343
|
+
p95LatencyMs: 5313,
|
|
1344
|
+
outputTokensPerSecond: 103.81,
|
|
1345
|
+
errorRate: 0,
|
|
897
1346
|
samples: 3
|
|
898
1347
|
},
|
|
899
|
-
"
|
|
900
|
-
measuredAt: "2026-
|
|
901
|
-
latencyMs:
|
|
902
|
-
p95LatencyMs:
|
|
903
|
-
outputTokensPerSecond:
|
|
1348
|
+
"qwen/qwen3.7-flash": {
|
|
1349
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1350
|
+
latencyMs: 3385.5,
|
|
1351
|
+
p95LatencyMs: 4042.7,
|
|
1352
|
+
outputTokensPerSecond: 153.94,
|
|
904
1353
|
errorRate: 0,
|
|
905
1354
|
samples: 3
|
|
906
1355
|
},
|
|
907
|
-
"
|
|
908
|
-
measuredAt: "2026-
|
|
909
|
-
latencyMs:
|
|
910
|
-
p95LatencyMs:
|
|
911
|
-
outputTokensPerSecond:
|
|
1356
|
+
"qwen/qwen3.7-max": {
|
|
1357
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1358
|
+
latencyMs: 9387.1,
|
|
1359
|
+
p95LatencyMs: 10490.2,
|
|
1360
|
+
outputTokensPerSecond: 54.92,
|
|
912
1361
|
errorRate: 0,
|
|
913
1362
|
samples: 3
|
|
914
1363
|
},
|
|
915
|
-
"
|
|
916
|
-
measuredAt: "2026-
|
|
917
|
-
latencyMs:
|
|
918
|
-
p95LatencyMs:
|
|
919
|
-
outputTokensPerSecond:
|
|
920
|
-
errorRate: 0
|
|
1364
|
+
"qwen/qwen3.7-plus": {
|
|
1365
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1366
|
+
latencyMs: 9766.6,
|
|
1367
|
+
p95LatencyMs: 9798.2,
|
|
1368
|
+
outputTokensPerSecond: 52.42,
|
|
1369
|
+
errorRate: 0,
|
|
921
1370
|
samples: 3
|
|
922
1371
|
},
|
|
923
|
-
"
|
|
924
|
-
measuredAt: "2026-
|
|
925
|
-
latencyMs:
|
|
926
|
-
p95LatencyMs:
|
|
927
|
-
outputTokensPerSecond:
|
|
1372
|
+
"tencent/hy3": {
|
|
1373
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1374
|
+
latencyMs: 6062.3,
|
|
1375
|
+
p95LatencyMs: 7070.2,
|
|
1376
|
+
outputTokensPerSecond: 87.3,
|
|
928
1377
|
errorRate: 0,
|
|
929
1378
|
samples: 3
|
|
930
1379
|
},
|
|
931
|
-
"
|
|
932
|
-
measuredAt: "2026-
|
|
933
|
-
latencyMs:
|
|
934
|
-
p95LatencyMs:
|
|
935
|
-
outputTokensPerSecond:
|
|
1380
|
+
"xai/grok-4.3": {
|
|
1381
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1382
|
+
latencyMs: 9467.7,
|
|
1383
|
+
p95LatencyMs: 10087.9,
|
|
1384
|
+
outputTokensPerSecond: 48.36,
|
|
936
1385
|
errorRate: 0,
|
|
937
1386
|
samples: 3
|
|
938
1387
|
},
|
|
939
|
-
"
|
|
940
|
-
measuredAt: "2026-
|
|
941
|
-
latencyMs:
|
|
942
|
-
p95LatencyMs:
|
|
943
|
-
outputTokensPerSecond:
|
|
1388
|
+
"xai/grok-4.5": {
|
|
1389
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1390
|
+
latencyMs: 13564.8,
|
|
1391
|
+
p95LatencyMs: 17351.9,
|
|
1392
|
+
outputTokensPerSecond: 60.71,
|
|
944
1393
|
errorRate: 0,
|
|
945
1394
|
samples: 3
|
|
946
1395
|
},
|
|
947
|
-
"
|
|
948
|
-
measuredAt: "2026-
|
|
949
|
-
latencyMs:
|
|
950
|
-
p95LatencyMs:
|
|
951
|
-
outputTokensPerSecond:
|
|
1396
|
+
"xai/grok-build-0.1": {
|
|
1397
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1398
|
+
latencyMs: 16394.8,
|
|
1399
|
+
p95LatencyMs: 18035.4,
|
|
1400
|
+
outputTokensPerSecond: 96.86,
|
|
952
1401
|
errorRate: 0,
|
|
953
1402
|
samples: 3
|
|
954
1403
|
},
|
|
955
|
-
"
|
|
956
|
-
measuredAt: "2026-
|
|
957
|
-
latencyMs:
|
|
958
|
-
p95LatencyMs:
|
|
959
|
-
outputTokensPerSecond:
|
|
1404
|
+
"xiaomi/mimo-v2.5-pro": {
|
|
1405
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1406
|
+
latencyMs: 12070.7,
|
|
1407
|
+
p95LatencyMs: 12386.8,
|
|
1408
|
+
outputTokensPerSecond: 42.44,
|
|
960
1409
|
errorRate: 0,
|
|
961
1410
|
samples: 3
|
|
962
1411
|
},
|
|
963
1412
|
"zai/glm-5": {
|
|
964
|
-
measuredAt: "2026-
|
|
965
|
-
latencyMs:
|
|
966
|
-
p95LatencyMs:
|
|
967
|
-
outputTokensPerSecond:
|
|
1413
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1414
|
+
latencyMs: 6839.7,
|
|
1415
|
+
p95LatencyMs: 7261.4,
|
|
1416
|
+
outputTokensPerSecond: 75.16,
|
|
1417
|
+
errorRate: 0,
|
|
1418
|
+
samples: 3
|
|
1419
|
+
},
|
|
1420
|
+
"zai/glm-5-turbo": {
|
|
1421
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1422
|
+
latencyMs: 55348.5,
|
|
1423
|
+
p95LatencyMs: 114086.6,
|
|
1424
|
+
outputTokensPerSecond: 14.64,
|
|
968
1425
|
errorRate: 0,
|
|
969
1426
|
samples: 3
|
|
970
1427
|
},
|
|
971
|
-
"
|
|
972
|
-
measuredAt: "2026-
|
|
973
|
-
latencyMs:
|
|
974
|
-
p95LatencyMs:
|
|
975
|
-
outputTokensPerSecond:
|
|
1428
|
+
"zai/glm-5.1": {
|
|
1429
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1430
|
+
latencyMs: 15658.4,
|
|
1431
|
+
p95LatencyMs: 17307.1,
|
|
1432
|
+
outputTokensPerSecond: 32.9,
|
|
976
1433
|
errorRate: 0,
|
|
977
1434
|
samples: 3
|
|
978
1435
|
},
|
|
979
|
-
"
|
|
980
|
-
measuredAt: "2026-
|
|
981
|
-
latencyMs:
|
|
982
|
-
p95LatencyMs:
|
|
983
|
-
outputTokensPerSecond:
|
|
1436
|
+
"zai/glm-5.2": {
|
|
1437
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1438
|
+
latencyMs: 10308.5,
|
|
1439
|
+
p95LatencyMs: 15127.6,
|
|
1440
|
+
outputTokensPerSecond: 54.87,
|
|
984
1441
|
errorRate: 0,
|
|
985
1442
|
samples: 3
|
|
986
1443
|
},
|
|
987
|
-
"
|
|
988
|
-
measuredAt: "2026-
|
|
989
|
-
latencyMs:
|
|
990
|
-
p95LatencyMs:
|
|
991
|
-
outputTokensPerSecond:
|
|
1444
|
+
"zai/glm-5.3": {
|
|
1445
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1446
|
+
latencyMs: 7272.4,
|
|
1447
|
+
p95LatencyMs: 7998.1,
|
|
1448
|
+
outputTokensPerSecond: 71.09,
|
|
992
1449
|
errorRate: 0,
|
|
993
1450
|
samples: 3
|
|
994
1451
|
},
|
|
995
|
-
"
|
|
996
|
-
measuredAt: "2026-
|
|
997
|
-
latencyMs:
|
|
998
|
-
p95LatencyMs:
|
|
999
|
-
outputTokensPerSecond:
|
|
1452
|
+
"zai/glm-5.3-flash": {
|
|
1453
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1454
|
+
latencyMs: 10545.3,
|
|
1455
|
+
p95LatencyMs: 11672.4,
|
|
1456
|
+
outputTokensPerSecond: 49.01,
|
|
1000
1457
|
errorRate: 0,
|
|
1001
1458
|
samples: 3
|
|
1002
1459
|
}
|
|
@@ -1010,11 +1467,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1010
1467
|
latencyMs: 2305,
|
|
1011
1468
|
outputTokensPerSecond: 140.6
|
|
1012
1469
|
},
|
|
1013
|
-
"anthropic/claude-opus-4.6": {
|
|
1014
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1015
|
-
latencyMs: 2139,
|
|
1016
|
-
outputTokensPerSecond: 119.7
|
|
1017
|
-
},
|
|
1018
1470
|
"anthropic/claude-sonnet-4.6": {
|
|
1019
1471
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1020
1472
|
latencyMs: 2110,
|
|
@@ -1048,11 +1500,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1048
1500
|
latencyMs: 1609,
|
|
1049
1501
|
outputTokensPerSecond: 167.2
|
|
1050
1502
|
},
|
|
1051
|
-
"moonshot/kimi-k2.5": {
|
|
1052
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1053
|
-
latencyMs: 1646,
|
|
1054
|
-
outputTokensPerSecond: 155.7
|
|
1055
|
-
},
|
|
1056
1503
|
"openai/gpt-4o-mini": {
|
|
1057
1504
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1058
1505
|
latencyMs: 2764,
|
|
@@ -1062,18 +1509,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1062
1509
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1063
1510
|
latencyMs: 7935,
|
|
1064
1511
|
outputTokensPerSecond: 32.3
|
|
1065
|
-
},
|
|
1066
|
-
"xai/grok-4-1-fast-non-reasoning": {
|
|
1067
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1068
|
-
latencyMs: 1244,
|
|
1069
|
-
outputTokensPerSecond: 205.8,
|
|
1070
|
-
intelligenceIndex: 41
|
|
1071
|
-
},
|
|
1072
|
-
"xai/grok-4-1-fast-reasoning": {
|
|
1073
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1074
|
-
latencyMs: 1454,
|
|
1075
|
-
outputTokensPerSecond: 176.2,
|
|
1076
|
-
intelligenceIndex: 41
|
|
1077
1512
|
}
|
|
1078
1513
|
});
|
|
1079
1514
|
function inferToolRequirement(prompt, _systemPrompt, toolChoice) {
|
|
@@ -1566,7 +2001,7 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1566
2001
|
match(["gpt-5.3-codex"], 1),
|
|
1567
2002
|
match(["claude-sonnet-4.6"], 0.94),
|
|
1568
2003
|
match(["glm-5.2"], 0.9),
|
|
1569
|
-
match(["
|
|
2004
|
+
match(["deepseek-v4-pro"], 0.86)
|
|
1570
2005
|
);
|
|
1571
2006
|
case "reasoning":
|
|
1572
2007
|
return Math.max(
|
|
@@ -1590,14 +2025,13 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1590
2025
|
base,
|
|
1591
2026
|
match(["gemini-3.5-flash"], 1),
|
|
1592
2027
|
match(["grok-4.5"], 0.93),
|
|
1593
|
-
match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
|
|
1594
|
-
match(["kimi-k2.7"], 0.84)
|
|
2028
|
+
match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
|
|
1595
2029
|
);
|
|
1596
2030
|
case "vision":
|
|
1597
2031
|
return Math.max(
|
|
1598
2032
|
base,
|
|
1599
2033
|
match(["gemini-3.1-pro"], 0.96),
|
|
1600
|
-
match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-
|
|
2034
|
+
match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9)
|
|
1601
2035
|
);
|
|
1602
2036
|
case "long_context":
|
|
1603
2037
|
return Math.max(
|
|
@@ -1609,17 +2043,18 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1609
2043
|
);
|
|
1610
2044
|
case "extraction": {
|
|
1611
2045
|
const kimiExtractionAffinity = language === "zh" ? 1 : 0.9;
|
|
2046
|
+
const otherExtractionAffinity = language === "zh" ? 0.88 : 0.9;
|
|
1612
2047
|
return Math.max(
|
|
1613
2048
|
base,
|
|
1614
|
-
match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"],
|
|
1615
|
-
match(["claude-sonnet-5", "claude-sonnet-4.6"],
|
|
1616
|
-
match(["kimi-k3"
|
|
2049
|
+
match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], otherExtractionAffinity),
|
|
2050
|
+
match(["claude-sonnet-5", "claude-sonnet-4.6"], otherExtractionAffinity),
|
|
2051
|
+
match(["kimi-k3"], kimiExtractionAffinity)
|
|
1617
2052
|
);
|
|
1618
2053
|
}
|
|
1619
2054
|
default:
|
|
1620
2055
|
return Math.max(
|
|
1621
2056
|
base,
|
|
1622
|
-
match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"
|
|
2057
|
+
match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)
|
|
1623
2058
|
);
|
|
1624
2059
|
}
|
|
1625
2060
|
}
|
|
@@ -1678,6 +2113,9 @@ function evidenceCandidates(task) {
|
|
|
1678
2113
|
"deepseek/deepseek-v4-pro"
|
|
1679
2114
|
];
|
|
1680
2115
|
}
|
|
2116
|
+
if (task === "extraction") {
|
|
2117
|
+
return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"];
|
|
2118
|
+
}
|
|
1681
2119
|
if (task === "reasoning_math") {
|
|
1682
2120
|
return [
|
|
1683
2121
|
"google/gemini-3.5-flash",
|
|
@@ -1734,9 +2172,12 @@ var PortfolioStrategy = class {
|
|
|
1734
2172
|
const targetTier = (features.taskType === "reasoning_mcq" || features.taskType === "reasoning_math") && (base.tier === "SIMPLE" || base.tier === "MEDIUM") ? "REASONING" : base.tier;
|
|
1735
2173
|
const tierConfig = tierConfigs[targetTier];
|
|
1736
2174
|
const configuredCandidates = tierConfig ? getFallbackChain(targetTier, tierConfigs) : [];
|
|
2175
|
+
const unavailable = new Set(options.unavailableModels ?? []);
|
|
1737
2176
|
const chain = [
|
|
1738
2177
|
.../* @__PURE__ */ new Set([...configuredCandidates, ...evidenceCandidates(features.taskType)])
|
|
1739
|
-
].filter(
|
|
2178
|
+
].filter(
|
|
2179
|
+
(model2) => typeof model2 === "string" && model2.length > 0 && !unavailable.has(model2)
|
|
2180
|
+
);
|
|
1740
2181
|
const eligible = chain.filter(
|
|
1741
2182
|
(model2) => isEligible(model2, features, maxOutputTokens, options)
|
|
1742
2183
|
);
|
|
@@ -1813,13 +2254,7 @@ var PortfolioStrategy = class {
|
|
|
1813
2254
|
...eligibleCandidates.filter(
|
|
1814
2255
|
(model2) => !scoredModels.includes(model2) && !webResearchFallbackOrder.includes(model2)
|
|
1815
2256
|
)
|
|
1816
|
-
] :
|
|
1817
|
-
...scoredModels,
|
|
1818
|
-
...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
|
|
1819
|
-
] : [
|
|
1820
|
-
...scoredModels,
|
|
1821
|
-
...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
|
|
1822
|
-
];
|
|
2257
|
+
] : [...scoredModels, ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))];
|
|
1823
2258
|
const model = ranked[0] ?? base.model;
|
|
1824
2259
|
const selectedTierConfigs = {
|
|
1825
2260
|
...tierConfigs,
|
|
@@ -1856,7 +2291,7 @@ var PortfolioStrategy = class {
|
|
|
1856
2291
|
}
|
|
1857
2292
|
};
|
|
1858
2293
|
var DEFAULT_ROUTING_CONFIG = {
|
|
1859
|
-
version: "3.
|
|
2294
|
+
version: "3.5",
|
|
1860
2295
|
strategy: "portfolio",
|
|
1861
2296
|
portfolio: {
|
|
1862
2297
|
auto: {
|
|
@@ -2910,186 +3345,249 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
2910
3345
|
// Below this confidence → ambiguous (null tier)
|
|
2911
3346
|
confidenceThreshold: 0.7
|
|
2912
3347
|
},
|
|
3348
|
+
// ─── Tier chains ───
|
|
3349
|
+
//
|
|
3350
|
+
// Catalog refresh 2026-08-29 (V3.5). Every chain below names only models
|
|
3351
|
+
// the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the
|
|
3352
|
+
// gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast
|
|
3353
|
+
// and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview,
|
|
3354
|
+
// the whole `free/*` namespace — were removed everywhere, including fallback
|
|
3355
|
+
// rungs, so a routed model is always one a user can find on blockrun.ai/models.
|
|
3356
|
+
//
|
|
3357
|
+
// Primaries moved only where portfolio.ts already carries calibration
|
|
3358
|
+
// evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for
|
|
3359
|
+
// agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no
|
|
3360
|
+
// trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash,
|
|
3361
|
+
// gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs;
|
|
3362
|
+
// promotion waits for a calibration run, because version recency is not a
|
|
3363
|
+
// quality signal.
|
|
3364
|
+
//
|
|
3365
|
+
// Latency figures in comments are the 2026-08-29 gateway probe
|
|
3366
|
+
// (model-profiles.generated.json); prices are the catalog list.
|
|
2913
3367
|
// Auto (balanced) tier configs - current default smart routing
|
|
2914
|
-
// Benchmark-tuned 2026-03-16: balancing quality (retention) + latency
|
|
2915
3368
|
tiers: {
|
|
2916
3369
|
SIMPLE: {
|
|
2917
3370
|
primary: "google/gemini-2.5-flash",
|
|
2918
|
-
//
|
|
3371
|
+
// $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer
|
|
2919
3372
|
fallback: [
|
|
2920
3373
|
"google/gemini-3-flash-preview",
|
|
2921
|
-
//
|
|
3374
|
+
// $0.50/$3 — GPQA 5/6 in the 2026-07 calibration
|
|
3375
|
+
"google/gemini-3.5-flash-lite",
|
|
3376
|
+
// $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation
|
|
2922
3377
|
"deepseek/deepseek-chat",
|
|
2923
|
-
//
|
|
2924
|
-
"moonshot/kimi-k2.5",
|
|
2925
|
-
// 1,646ms, IQ 47, strong quality
|
|
3378
|
+
// $0.14/$0.28, 1M ctx
|
|
2926
3379
|
"google/gemini-3.1-flash-lite",
|
|
2927
|
-
// $0.25/$1.50, 1M
|
|
2928
|
-
"
|
|
2929
|
-
//
|
|
3380
|
+
// $0.25/$1.50, 1M ctx
|
|
3381
|
+
"openai/gpt-5.6-luna",
|
|
3382
|
+
// $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30)
|
|
2930
3383
|
"openai/gpt-5.4-nano",
|
|
2931
|
-
// $0.20/$1.25, 1M
|
|
2932
|
-
"
|
|
2933
|
-
//
|
|
2934
|
-
"
|
|
2935
|
-
//
|
|
3384
|
+
// $0.20/$1.25, 1M ctx
|
|
3385
|
+
"google/gemini-2.5-flash-lite",
|
|
3386
|
+
// $0.10/$0.40
|
|
3387
|
+
"nvidia/step-3.7-flash"
|
|
3388
|
+
// FREE backstop — NVIDIA free tier (probed 2026-08-21)
|
|
2936
3389
|
]
|
|
2937
3390
|
},
|
|
2938
3391
|
MEDIUM: {
|
|
2939
|
-
|
|
2940
|
-
//
|
|
3392
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the
|
|
3393
|
+
// calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts).
|
|
3394
|
+
primary: "google/gemini-3.5-flash",
|
|
3395
|
+
// $1.50/$9, 1M ctx, vision + tools
|
|
2941
3396
|
fallback: [
|
|
2942
|
-
"
|
|
2943
|
-
//
|
|
2944
|
-
"
|
|
2945
|
-
// $0.
|
|
3397
|
+
"google/gemini-3.6-flash",
|
|
3398
|
+
// $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration
|
|
3399
|
+
"zai/glm-5.3-flash",
|
|
3400
|
+
// $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27
|
|
3401
|
+
"openai/gpt-5.6-terra",
|
|
3402
|
+
// $2/$12, 1M ctx — GPT-5.6 balanced tier
|
|
2946
3403
|
"google/gemini-3-flash-preview",
|
|
2947
|
-
//
|
|
3404
|
+
// $0.50/$3
|
|
2948
3405
|
"deepseek/deepseek-chat",
|
|
2949
|
-
//
|
|
3406
|
+
// $0.14/$0.28
|
|
2950
3407
|
"google/gemini-2.5-flash",
|
|
2951
|
-
//
|
|
3408
|
+
// $0.30/$2.50
|
|
3409
|
+
"minimax/minimax-m3",
|
|
3410
|
+
// $0.30/$1.20, 1M ctx
|
|
2952
3411
|
"google/gemini-3.1-flash-lite",
|
|
2953
|
-
// $0.25/$1.50
|
|
2954
|
-
"
|
|
2955
|
-
//
|
|
2956
|
-
"
|
|
2957
|
-
//
|
|
2958
|
-
"xai/grok-3-mini"
|
|
2959
|
-
// 1,202ms, $0.30/$0.50
|
|
3412
|
+
// $0.25/$1.50
|
|
3413
|
+
"openai/gpt-5.6-luna",
|
|
3414
|
+
// $0.20/$1.20
|
|
3415
|
+
"google/gemini-2.5-flash-lite"
|
|
3416
|
+
// $0.10/$0.40
|
|
2960
3417
|
]
|
|
2961
3418
|
},
|
|
2962
3419
|
COMPLEX: {
|
|
2963
3420
|
primary: "google/gemini-3.1-pro",
|
|
2964
|
-
//
|
|
3421
|
+
// $2/$12 — proven long-context flagship (portfolio.ts long_context lead)
|
|
2965
3422
|
fallback: [
|
|
2966
|
-
"google/gemini-3-flash
|
|
2967
|
-
// 1
|
|
2968
|
-
"
|
|
2969
|
-
// 1
|
|
2970
|
-
"google/gemini-2.5-pro",
|
|
2971
|
-
// 1,294ms
|
|
3423
|
+
"google/gemini-3.6-flash",
|
|
3424
|
+
// $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here)
|
|
3425
|
+
"google/gemini-3.5-flash",
|
|
3426
|
+
// $1.50/$9 — calibrated
|
|
2972
3427
|
"anthropic/claude-sonnet-5",
|
|
2973
|
-
// near-Opus quality
|
|
3428
|
+
// $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated
|
|
3429
|
+
"xai/grok-4.5",
|
|
3430
|
+
// $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden)
|
|
3431
|
+
"google/gemini-2.5-pro",
|
|
3432
|
+
// $1.25/$10
|
|
2974
3433
|
"anthropic/claude-sonnet-4.6",
|
|
2975
|
-
//
|
|
2976
|
-
"deepseek/deepseek-chat",
|
|
2977
|
-
// 1,431ms, IQ 32
|
|
2978
|
-
"google/gemini-2.5-flash",
|
|
2979
|
-
// 1,238ms, IQ 20 — cheap last resort
|
|
3434
|
+
// $3/$15
|
|
2980
3435
|
"openai/gpt-5.6-terra",
|
|
2981
|
-
// GPT-5.6 balanced tier
|
|
3436
|
+
// $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202)
|
|
2982
3437
|
"openai/gpt-5.5",
|
|
2983
|
-
//
|
|
2984
|
-
"openai/gpt-5.4"
|
|
2985
|
-
//
|
|
3438
|
+
// $5/$30 — prior OpenAI flagship
|
|
3439
|
+
"openai/gpt-5.4",
|
|
3440
|
+
// $2.50/$15 — previous flagship, benchmarked
|
|
3441
|
+
"zai/glm-5.3",
|
|
3442
|
+
// $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19
|
|
3443
|
+
"moonshot/kimi-k3",
|
|
3444
|
+
// $3/$15, 1M ctx — Moonshot flagship (K2.7 successor)
|
|
3445
|
+
"deepseek/deepseek-v4-pro",
|
|
3446
|
+
// $0.435/$0.87 — strongest open-weight reasoner
|
|
3447
|
+
"deepseek/deepseek-chat",
|
|
3448
|
+
// $0.14/$0.28 — cheap last resort
|
|
3449
|
+
"google/gemini-2.5-flash"
|
|
3450
|
+
// $0.30/$2.50
|
|
2986
3451
|
]
|
|
2987
3452
|
},
|
|
2988
3453
|
REASONING: {
|
|
2989
|
-
|
|
2990
|
-
//
|
|
3454
|
+
// Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek
|
|
3455
|
+
// Reasoner is the cheapest listed reasoner at the same 1M context.
|
|
3456
|
+
primary: "deepseek/deepseek-reasoner",
|
|
3457
|
+
// $0.14/$0.28, 1M ctx
|
|
2991
3458
|
fallback: [
|
|
2992
|
-
"xai/grok-4-fast-reasoning",
|
|
2993
|
-
// 1,298ms, $0.20/$0.50
|
|
2994
|
-
"deepseek/deepseek-reasoner",
|
|
2995
|
-
// V4 Flash thinking ($0.20/$0.40, 1M ctx)
|
|
2996
3459
|
"deepseek/deepseek-v4-pro",
|
|
2997
|
-
//
|
|
3460
|
+
// $0.435/$0.87 — calibrated reasoning band 0.95
|
|
3461
|
+
"xai/grok-4.3",
|
|
3462
|
+
// $1.50/$4, 1M ctx — xAI reasoning model, vision
|
|
3463
|
+
"qwen/qwen3.7-plus",
|
|
3464
|
+
// $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed)
|
|
3465
|
+
"google/gemini-3.5-flash",
|
|
3466
|
+
// $1.50/$9 — MGSM 5/5
|
|
2998
3467
|
"openai/o4-mini",
|
|
2999
|
-
//
|
|
3468
|
+
// $1.10/$4.40
|
|
3000
3469
|
"openai/o3"
|
|
3001
|
-
// 2
|
|
3470
|
+
// $2/$8
|
|
3002
3471
|
]
|
|
3003
3472
|
}
|
|
3004
3473
|
},
|
|
3005
3474
|
// Eco tier configs - absolute cheapest (blockrun/eco)
|
|
3006
3475
|
ecoTiers: {
|
|
3007
3476
|
SIMPLE: {
|
|
3008
|
-
primary: "
|
|
3009
|
-
// FREE
|
|
3477
|
+
primary: "nvidia/step-3.7-flash",
|
|
3478
|
+
// FREE — NVIDIA free tier flagship
|
|
3010
3479
|
fallback: [
|
|
3011
|
-
"
|
|
3012
|
-
// FREE —
|
|
3013
|
-
|
|
3014
|
-
//
|
|
3015
|
-
//
|
|
3016
|
-
//
|
|
3017
|
-
"google/gemini-3.1-flash-lite",
|
|
3018
|
-
// $0.25/$1.50 — newest flash-lite
|
|
3019
|
-
"openai/gpt-5.4-nano",
|
|
3020
|
-
// $0.20/$1.25 — fast nano
|
|
3480
|
+
"nvidia/nemotron-nano-9b-v2",
|
|
3481
|
+
// FREE — compact + fast, high-volume light tasks
|
|
3482
|
+
// The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash
|
|
3483
|
+
// 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400
|
|
3484
|
+
// 2026-08-21). Each retirement retargets the two free rungs to the
|
|
3485
|
+
// current free tier; the paid rungs below never move.
|
|
3021
3486
|
"google/gemini-2.5-flash-lite",
|
|
3022
|
-
// $0.10/$0.40
|
|
3023
|
-
"
|
|
3024
|
-
// $0.
|
|
3487
|
+
// $0.10/$0.40 — cheapest paid rung
|
|
3488
|
+
"zai/glm-5.3-flash",
|
|
3489
|
+
// $0.15/$0.50, 1M ctx, vision + tools
|
|
3490
|
+
"openai/gpt-5.6-luna",
|
|
3491
|
+
// $0.20/$1.20, 1M ctx
|
|
3492
|
+
"openai/gpt-5.4-nano",
|
|
3493
|
+
// $0.20/$1.25
|
|
3494
|
+
"google/gemini-3.1-flash-lite"
|
|
3495
|
+
// $0.25/$1.50
|
|
3025
3496
|
]
|
|
3026
3497
|
},
|
|
3027
3498
|
MEDIUM: {
|
|
3028
|
-
primary: "
|
|
3029
|
-
// $0.
|
|
3499
|
+
primary: "zai/glm-5.3-flash",
|
|
3500
|
+
// $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model
|
|
3030
3501
|
fallback: [
|
|
3502
|
+
"deepseek/deepseek-chat",
|
|
3503
|
+
// $0.14/$0.28
|
|
3504
|
+
"google/gemini-3.1-flash-lite",
|
|
3505
|
+
// $0.25/$1.50
|
|
3506
|
+
"openai/gpt-5.6-luna",
|
|
3507
|
+
// $0.20/$1.20
|
|
3031
3508
|
"openai/gpt-5.4-nano",
|
|
3032
3509
|
// $0.20/$1.25
|
|
3033
3510
|
"google/gemini-2.5-flash-lite",
|
|
3034
3511
|
// $0.10/$0.40
|
|
3035
|
-
"xai/grok-4-fast-non-reasoning",
|
|
3036
3512
|
"google/gemini-2.5-flash"
|
|
3513
|
+
// $0.30/$2.50
|
|
3037
3514
|
]
|
|
3038
3515
|
},
|
|
3039
3516
|
COMPLEX: {
|
|
3040
|
-
primary: "
|
|
3041
|
-
// $0.
|
|
3517
|
+
primary: "zai/glm-5.3-flash",
|
|
3518
|
+
// $0.15/$0.50, 1M ctx
|
|
3042
3519
|
fallback: [
|
|
3043
|
-
"
|
|
3044
|
-
|
|
3045
|
-
"
|
|
3046
|
-
|
|
3520
|
+
"deepseek/deepseek-chat",
|
|
3521
|
+
// $0.14/$0.28, 1M ctx
|
|
3522
|
+
"minimax/minimax-m3",
|
|
3523
|
+
// $0.30/$1.20, 1M ctx
|
|
3524
|
+
"deepseek/deepseek-v4-pro",
|
|
3525
|
+
// $0.435/$0.87
|
|
3526
|
+
"google/gemini-3.1-flash-lite",
|
|
3527
|
+
// $0.25/$1.50
|
|
3528
|
+
"google/gemini-2.5-flash"
|
|
3529
|
+
// $0.30/$2.50
|
|
3047
3530
|
]
|
|
3048
3531
|
},
|
|
3049
3532
|
REASONING: {
|
|
3050
|
-
primary: "
|
|
3051
|
-
// $0.
|
|
3533
|
+
primary: "deepseek/deepseek-reasoner",
|
|
3534
|
+
// $0.14/$0.28, 1M ctx — cheapest listed reasoner
|
|
3052
3535
|
fallback: [
|
|
3053
|
-
"
|
|
3054
|
-
|
|
3055
|
-
|
|
3056
|
-
|
|
3057
|
-
|
|
3536
|
+
"deepseek/deepseek-v4-pro",
|
|
3537
|
+
// $0.435/$0.87
|
|
3538
|
+
"qwen/qwen3.7-plus",
|
|
3539
|
+
// $0.32/$1.28 — reasoning
|
|
3540
|
+
"minimax/minimax-m3",
|
|
3541
|
+
// $0.30/$1.20 — reasoning + coding
|
|
3542
|
+
"zai/glm-5.3-flash"
|
|
3543
|
+
// $0.15/$0.50 — reasoning tokens alongside content
|
|
3058
3544
|
]
|
|
3059
3545
|
}
|
|
3060
3546
|
},
|
|
3061
3547
|
// Premium tier configs - best quality (blockrun/premium)
|
|
3062
|
-
// codex=complex coding,
|
|
3548
|
+
// codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits
|
|
3063
3549
|
premiumTiers: {
|
|
3064
3550
|
SIMPLE: {
|
|
3065
|
-
|
|
3066
|
-
|
|
3551
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08).
|
|
3552
|
+
primary: "google/gemini-3.5-flash",
|
|
3553
|
+
// $1.50/$9, 1M ctx, vision + tools — calibrated
|
|
3067
3554
|
fallback: [
|
|
3068
|
-
"
|
|
3069
|
-
//
|
|
3070
|
-
"moonshot/kimi-k2.5",
|
|
3071
|
-
// $0.60/$3.00 - proven reliable backstop when Moonshot direct API falters
|
|
3072
|
-
"google/gemini-2.5-flash",
|
|
3073
|
-
// 60% retention, fast growth
|
|
3555
|
+
"google/gemini-3.6-flash",
|
|
3556
|
+
// $1.50/$7.50 — newest Flash
|
|
3074
3557
|
"anthropic/claude-haiku-4.5",
|
|
3075
|
-
|
|
3558
|
+
// $1/$5
|
|
3559
|
+
"zai/glm-5.3",
|
|
3560
|
+
// $1.40/$4.40, 1M ctx
|
|
3561
|
+
"google/gemini-2.5-flash",
|
|
3562
|
+
// $0.30/$2.50
|
|
3563
|
+
"google/gemini-3.5-flash-lite",
|
|
3564
|
+
// $0.30/$2.50
|
|
3076
3565
|
"deepseek/deepseek-chat"
|
|
3566
|
+
// $0.14/$0.28
|
|
3077
3567
|
]
|
|
3078
3568
|
},
|
|
3079
3569
|
MEDIUM: {
|
|
3080
3570
|
primary: "openai/gpt-5.3-codex",
|
|
3081
|
-
// $1.75/$14 - 400K context, 128K output
|
|
3571
|
+
// $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts)
|
|
3082
3572
|
fallback: [
|
|
3083
|
-
"moonshot/kimi-k2.7",
|
|
3084
|
-
// Moonshot flagship
|
|
3085
|
-
"moonshot/kimi-k2.6",
|
|
3086
|
-
"moonshot/kimi-k2.5",
|
|
3087
|
-
"google/gemini-2.5-flash",
|
|
3088
|
-
// 60% retention, good coding capability
|
|
3089
|
-
"google/gemini-2.5-pro",
|
|
3090
|
-
"xai/grok-4-0709",
|
|
3091
3573
|
"anthropic/claude-sonnet-5",
|
|
3092
|
-
|
|
3574
|
+
// $3/$15 — code_agent band 0.98
|
|
3575
|
+
"moonshot/kimi-k3",
|
|
3576
|
+
// $3/$15, 1M ctx — Moonshot flagship
|
|
3577
|
+
"zai/glm-5.3",
|
|
3578
|
+
// $1.40/$4.40 — long-horizon coding
|
|
3579
|
+
"google/gemini-3.6-flash",
|
|
3580
|
+
// $1.50/$7.50
|
|
3581
|
+
"google/gemini-3.5-flash",
|
|
3582
|
+
// $1.50/$9
|
|
3583
|
+
"google/gemini-2.5-pro",
|
|
3584
|
+
// $1.25/$10
|
|
3585
|
+
"xai/grok-4.5",
|
|
3586
|
+
// $2.50/$9
|
|
3587
|
+
"anthropic/claude-sonnet-4.6",
|
|
3588
|
+
// $3/$15
|
|
3589
|
+
"openai/gpt-5.6-terra"
|
|
3590
|
+
// $2/$12
|
|
3093
3591
|
]
|
|
3094
3592
|
},
|
|
3095
3593
|
COMPLEX: {
|
|
@@ -3099,8 +3597,8 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3099
3597
|
// Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking)
|
|
3100
3598
|
// Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is
|
|
3101
3599
|
// also prone to "high demand" 503s (correlated failure — everyone falls
|
|
3102
|
-
// back to Google at the same time). Prefer xAI
|
|
3103
|
-
// flagship → DeepSeek → NVIDIA free instead.
|
|
3600
|
+
// back to Google at the same time). Prefer in-family → xAI → Moonshot →
|
|
3601
|
+
// OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead.
|
|
3104
3602
|
fallback: [
|
|
3105
3603
|
"anthropic/claude-opus-5",
|
|
3106
3604
|
// in-family hot swap first (half the price, 1M ctx + adaptive thinking)
|
|
@@ -3108,52 +3606,54 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3108
3606
|
// in-family hot swap (identical cost to 5)
|
|
3109
3607
|
"anthropic/claude-opus-4.7",
|
|
3110
3608
|
// in-family hot swap (identical cost to 4.8)
|
|
3111
|
-
"anthropic/claude-opus-4.6",
|
|
3112
|
-
// in-family hot swap
|
|
3113
3609
|
"anthropic/claude-sonnet-5",
|
|
3114
3610
|
// Sonnet-tier drop-down, near-Opus quality
|
|
3115
3611
|
"anthropic/claude-sonnet-4.6",
|
|
3116
3612
|
"xai/grok-4.5",
|
|
3117
|
-
// xAI flagship — 503-resistant, direct-xAI SKU
|
|
3118
|
-
"
|
|
3119
|
-
// 503-resistant flagship
|
|
3120
|
-
"moonshot/kimi-k2.7",
|
|
3613
|
+
// xAI flagship — 503-resistant, direct-xAI SKU
|
|
3614
|
+
"moonshot/kimi-k3",
|
|
3121
3615
|
// Moonshot flagship, independent infra
|
|
3122
|
-
"moonshot/kimi-k2.6",
|
|
3123
|
-
"moonshot/kimi-k2.5",
|
|
3124
3616
|
"openai/gpt-5.6-terra",
|
|
3125
|
-
// GPT-5.6 balanced tier —
|
|
3617
|
+
// GPT-5.6 balanced tier — stable (Sol excluded: #202)
|
|
3126
3618
|
"openai/gpt-5.5",
|
|
3127
3619
|
// Prior OpenAI flagship — 1M+ ctx, native agent + computer use
|
|
3128
3620
|
"openai/gpt-5.4",
|
|
3129
3621
|
// Previous flagship (slow but stable, benchmarked at 6,213ms)
|
|
3130
3622
|
"openai/gpt-5.3-codex",
|
|
3623
|
+
"zai/glm-5.3",
|
|
3624
|
+
// Z.AI flagship, 1M ctx
|
|
3625
|
+
"deepseek/deepseek-v4-pro",
|
|
3626
|
+
// strongest open-weight reasoner
|
|
3131
3627
|
"deepseek/deepseek-chat",
|
|
3132
3628
|
// Cheap, reliable
|
|
3133
|
-
"
|
|
3134
|
-
// NVIDIA free ultimate backstop
|
|
3629
|
+
"nvidia/step-3.7-flash"
|
|
3630
|
+
// NVIDIA free ultimate backstop
|
|
3135
3631
|
]
|
|
3136
3632
|
},
|
|
3137
3633
|
REASONING: {
|
|
3138
|
-
|
|
3139
|
-
//
|
|
3634
|
+
// Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both,
|
|
3635
|
+
// plus Sonnet 5's tau2/BrowseComp trajectory evidence).
|
|
3636
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3637
|
+
// $3/$15, 1M ctx, adaptive thinking
|
|
3140
3638
|
fallback: [
|
|
3141
|
-
"anthropic/claude-sonnet-
|
|
3142
|
-
// in-family hot swap — same cost
|
|
3639
|
+
"anthropic/claude-sonnet-4.6",
|
|
3640
|
+
// in-family hot swap — same cost
|
|
3143
3641
|
"anthropic/claude-opus-5",
|
|
3144
3642
|
// Newest flagship Opus w/ adaptive thinking
|
|
3145
3643
|
"anthropic/claude-opus-4.8",
|
|
3146
3644
|
// Prior flagship Opus — identical cost to 5
|
|
3147
3645
|
"anthropic/claude-opus-4.7",
|
|
3148
3646
|
// Flagship Opus w/ adaptive thinking
|
|
3149
|
-
"
|
|
3150
|
-
//
|
|
3151
|
-
"
|
|
3152
|
-
//
|
|
3647
|
+
"xai/grok-4.5",
|
|
3648
|
+
// reasoning band 0.94
|
|
3649
|
+
"deepseek/deepseek-v4-pro",
|
|
3650
|
+
// reasoning band 0.95
|
|
3651
|
+
"xai/grok-4.3",
|
|
3652
|
+
// $1.50/$4 — xAI reasoning model
|
|
3153
3653
|
"openai/o4-mini",
|
|
3154
|
-
//
|
|
3654
|
+
// $1.10/$4.40
|
|
3155
3655
|
"openai/o3"
|
|
3156
|
-
// 2
|
|
3656
|
+
// $2/$8
|
|
3157
3657
|
]
|
|
3158
3658
|
}
|
|
3159
3659
|
},
|
|
@@ -3163,101 +3663,102 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3163
3663
|
primary: "openai/gpt-4o-mini",
|
|
3164
3664
|
// $0.15/$0.60 - best tool compliance at lowest cost
|
|
3165
3665
|
fallback: [
|
|
3166
|
-
"
|
|
3167
|
-
// 1
|
|
3666
|
+
"openai/gpt-5.6-luna",
|
|
3667
|
+
// $0.20/$1.20 — lightweight agentic tier of GPT-5.6
|
|
3668
|
+
"zai/glm-5.3-flash",
|
|
3669
|
+
// $0.15/$0.50 — tool calls verified live 2026-08-27
|
|
3168
3670
|
"anthropic/claude-haiku-4.5",
|
|
3169
|
-
//
|
|
3170
|
-
"
|
|
3171
|
-
//
|
|
3671
|
+
// $1/$5
|
|
3672
|
+
"google/gemini-2.5-flash"
|
|
3673
|
+
// $0.30/$2.50
|
|
3172
3674
|
]
|
|
3173
3675
|
},
|
|
3174
3676
|
MEDIUM: {
|
|
3175
|
-
|
|
3176
|
-
//
|
|
3677
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the
|
|
3678
|
+
// Terminal-Bench and tau2 trajectory evidence in portfolio.ts.
|
|
3679
|
+
primary: "openai/gpt-5-mini",
|
|
3680
|
+
// $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline
|
|
3177
3681
|
fallback: [
|
|
3178
|
-
"
|
|
3179
|
-
//
|
|
3180
|
-
"
|
|
3181
|
-
// $0.
|
|
3182
|
-
"
|
|
3183
|
-
//
|
|
3682
|
+
"google/gemini-3.5-flash",
|
|
3683
|
+
// $1.50/$9 — tool_agent band 0.88
|
|
3684
|
+
"zai/glm-5.3-flash",
|
|
3685
|
+
// $0.15/$0.50 — tools verified
|
|
3686
|
+
"openai/gpt-5.6-terra",
|
|
3687
|
+
// $2/$12
|
|
3184
3688
|
"openai/gpt-4o-mini",
|
|
3185
|
-
//
|
|
3689
|
+
// $0.15/$0.60 — reliable tool calling
|
|
3186
3690
|
"anthropic/claude-haiku-4.5",
|
|
3187
|
-
//
|
|
3188
|
-
"deepseek/deepseek-chat"
|
|
3189
|
-
//
|
|
3691
|
+
// $1/$5
|
|
3692
|
+
"deepseek/deepseek-chat",
|
|
3693
|
+
// $0.14/$0.28
|
|
3694
|
+
"moonshot/kimi-k3"
|
|
3695
|
+
// $3/$15 — tool_agent band 0.85
|
|
3190
3696
|
]
|
|
3191
3697
|
},
|
|
3192
3698
|
COMPLEX: {
|
|
3193
|
-
|
|
3194
|
-
//
|
|
3699
|
+
// Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0,
|
|
3700
|
+
// Terminal-Bench safety band lead (portfolio.ts).
|
|
3701
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3702
|
+
// $3/$15 — best agentic quality per trajectory evidence
|
|
3195
3703
|
// Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s
|
|
3196
3704
|
// correlate with Anthropic outages (everyone falls back together).
|
|
3197
3705
|
// Prefer 503-resistant providers first.
|
|
3198
3706
|
fallback: [
|
|
3199
|
-
"anthropic/claude-sonnet-
|
|
3200
|
-
// in-family hot swap — same cost
|
|
3707
|
+
"anthropic/claude-sonnet-4.6",
|
|
3708
|
+
// in-family hot swap — same cost
|
|
3201
3709
|
"anthropic/claude-opus-5",
|
|
3202
3710
|
// Newest flagship Opus — in-family hot swap
|
|
3203
3711
|
"anthropic/claude-opus-4.8",
|
|
3204
3712
|
// Prior flagship Opus — identical cost to 5
|
|
3205
3713
|
"anthropic/claude-opus-4.7",
|
|
3206
3714
|
// Flagship Opus — in-family hot swap
|
|
3207
|
-
"
|
|
3208
|
-
//
|
|
3209
|
-
"
|
|
3210
|
-
//
|
|
3211
|
-
"moonshot/kimi-k2.7",
|
|
3212
|
-
// Moonshot flagship — strong tool use, independent infra
|
|
3213
|
-
"moonshot/kimi-k2.5",
|
|
3214
|
-
// cost-stability backstop
|
|
3715
|
+
"xai/grok-4.5",
|
|
3716
|
+
// xAI flagship — strong tool use, independent infra
|
|
3717
|
+
"moonshot/kimi-k3",
|
|
3718
|
+
// Moonshot flagship — independent infra
|
|
3215
3719
|
"openai/gpt-5.6-terra",
|
|
3216
|
-
// GPT-5.6 balanced tier —
|
|
3720
|
+
// GPT-5.6 balanced tier — stable (Sol excluded: #202)
|
|
3217
3721
|
"openai/gpt-5.5",
|
|
3218
3722
|
// Prior flagship — native agent + computer use (exactly the agentic-tier use case)
|
|
3219
3723
|
"openai/gpt-5.4",
|
|
3220
|
-
// Previous flagship —
|
|
3724
|
+
// Previous flagship — reliable
|
|
3725
|
+
"openai/gpt-5.3-codex",
|
|
3726
|
+
// code_agent lead
|
|
3727
|
+
"zai/glm-5.3",
|
|
3728
|
+
// long-horizon coding
|
|
3729
|
+
"deepseek/deepseek-v4-pro",
|
|
3730
|
+
// retail high-risk 3/3
|
|
3221
3731
|
"deepseek/deepseek-chat",
|
|
3222
|
-
//
|
|
3223
|
-
"
|
|
3224
|
-
// NVIDIA free ultimate backstop
|
|
3732
|
+
// cheap, reliable
|
|
3733
|
+
"nvidia/step-3.7-flash"
|
|
3734
|
+
// NVIDIA free ultimate backstop
|
|
3225
3735
|
]
|
|
3226
3736
|
},
|
|
3227
3737
|
REASONING: {
|
|
3228
|
-
primary: "anthropic/claude-sonnet-
|
|
3229
|
-
//
|
|
3738
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3739
|
+
// $3/$15 — strong tool use + adaptive thinking
|
|
3230
3740
|
fallback: [
|
|
3231
|
-
"anthropic/claude-sonnet-
|
|
3232
|
-
// in-family hot swap — same cost
|
|
3741
|
+
"anthropic/claude-sonnet-4.6",
|
|
3742
|
+
// in-family hot swap — same cost
|
|
3233
3743
|
"anthropic/claude-opus-5",
|
|
3234
3744
|
// Newest flagship Opus w/ adaptive thinking
|
|
3235
3745
|
"anthropic/claude-opus-4.8",
|
|
3236
3746
|
// Prior flagship Opus — identical cost to 5
|
|
3237
3747
|
"anthropic/claude-opus-4.7",
|
|
3238
3748
|
// Flagship Opus w/ adaptive thinking
|
|
3239
|
-
"
|
|
3240
|
-
//
|
|
3241
|
-
"
|
|
3242
|
-
//
|
|
3749
|
+
"xai/grok-4.5",
|
|
3750
|
+
// reasoning band 0.94
|
|
3751
|
+
"deepseek/deepseek-v4-pro",
|
|
3752
|
+
// reasoning band 0.95
|
|
3243
3753
|
"deepseek/deepseek-reasoner"
|
|
3244
|
-
//
|
|
3754
|
+
// $0.14/$0.28
|
|
3245
3755
|
]
|
|
3246
3756
|
}
|
|
3247
3757
|
},
|
|
3248
|
-
// Time-windowed promotions — auto-applied when active, ignored when expired
|
|
3249
|
-
|
|
3250
|
-
|
|
3251
|
-
|
|
3252
|
-
startDate: "2026-04-01",
|
|
3253
|
-
endDate: "2026-05-01",
|
|
3254
|
-
tierOverrides: {
|
|
3255
|
-
SIMPLE: { primary: "zai/glm-5.1" }
|
|
3256
|
-
},
|
|
3257
|
-
profiles: ["auto"]
|
|
3258
|
-
// only auto profile — eco stays free, premium stays premium
|
|
3259
|
-
}
|
|
3260
|
-
],
|
|
3758
|
+
// Time-windowed promotions — auto-applied when active, ignored when expired.
|
|
3759
|
+
// The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and
|
|
3760
|
+
// has expired; the list is kept empty so the mechanism stays wired.
|
|
3761
|
+
promotions: [],
|
|
3261
3762
|
overrides: {
|
|
3262
3763
|
maxTokensForceComplex: 1e5,
|
|
3263
3764
|
structuredOutputMinTier: "MEDIUM",
|
|
@@ -3599,7 +4100,7 @@ async function createSolanaPaymentPayload(secretKey, fromAddress, recipient, amo
|
|
|
3599
4100
|
}
|
|
3600
4101
|
return null;
|
|
3601
4102
|
};
|
|
3602
|
-
let entry = await getBlockhashEntry(connection, rpcUrl, false);
|
|
4103
|
+
let entry = await getBlockhashEntry(connection, rpcUrl, options.forceFreshBlockhash ?? false);
|
|
3603
4104
|
let serializedTx = findDistinctTx(entry);
|
|
3604
4105
|
if (serializedTx === null) {
|
|
3605
4106
|
entry = await getBlockhashEntry(connection, rpcUrl, true);
|
|
@@ -3850,7 +4351,7 @@ function getCostSummary() {
|
|
|
3850
4351
|
}
|
|
3851
4352
|
|
|
3852
4353
|
// src/version.ts
|
|
3853
|
-
var SDK_VERSION = "3.13.
|
|
4354
|
+
var SDK_VERSION = "3.13.4";
|
|
3854
4355
|
var USER_AGENT = `blockrun-ts/${SDK_VERSION}`;
|
|
3855
4356
|
|
|
3856
4357
|
// src/client.ts
|
|
@@ -8504,6 +9005,68 @@ async function getOrCreateSolanaWallet() {
|
|
|
8504
9005
|
var SOLANA_API_URL = "https://sol.blockrun.ai/api";
|
|
8505
9006
|
var DEFAULT_MAX_TOKENS2 = 1024;
|
|
8506
9007
|
var DEFAULT_TIMEOUT14 = 6e4;
|
|
9008
|
+
var STALE_BLOCKHASH_RETRY_BACKOFFS_MS = [500, 2e3];
|
|
9009
|
+
var MAX_PAYMENT_FAILURE_BYTES = 64 * 1024;
|
|
9010
|
+
var SafeStaleBlockhashError = class extends PaymentError {
|
|
9011
|
+
constructor() {
|
|
9012
|
+
super("Payment verification used an expired Solana blockhash; retrying with a fresh quote.");
|
|
9013
|
+
this.name = "SafeStaleBlockhashError";
|
|
9014
|
+
}
|
|
9015
|
+
};
|
|
9016
|
+
function normalizePaymentSignal(value) {
|
|
9017
|
+
return typeof value === "string" ? value.toLowerCase().replace(/[_\-\s:]/g, "") : "";
|
|
9018
|
+
}
|
|
9019
|
+
async function readPaymentFailureBody(response) {
|
|
9020
|
+
const reader = response.body?.getReader();
|
|
9021
|
+
if (!reader) return "";
|
|
9022
|
+
const decoder = new TextDecoder();
|
|
9023
|
+
let total = 0;
|
|
9024
|
+
let text = "";
|
|
9025
|
+
try {
|
|
9026
|
+
for (; ; ) {
|
|
9027
|
+
const { done, value } = await reader.read();
|
|
9028
|
+
if (done) return text + decoder.decode();
|
|
9029
|
+
total += value.byteLength;
|
|
9030
|
+
if (total > MAX_PAYMENT_FAILURE_BYTES) {
|
|
9031
|
+
void reader.cancel();
|
|
9032
|
+
return null;
|
|
9033
|
+
}
|
|
9034
|
+
text += decoder.decode(value, { stream: true });
|
|
9035
|
+
}
|
|
9036
|
+
} catch {
|
|
9037
|
+
return null;
|
|
9038
|
+
}
|
|
9039
|
+
}
|
|
9040
|
+
async function isSafeStaleBlockhashResponse(response) {
|
|
9041
|
+
const length = Number(response.headers.get("content-length") || "0");
|
|
9042
|
+
if (Number.isFinite(length) && length > MAX_PAYMENT_FAILURE_BYTES) return false;
|
|
9043
|
+
const text = await readPaymentFailureBody(response);
|
|
9044
|
+
if (text === null) return false;
|
|
9045
|
+
let body;
|
|
9046
|
+
try {
|
|
9047
|
+
const parsed = JSON.parse(text);
|
|
9048
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
|
|
9049
|
+
body = parsed;
|
|
9050
|
+
} catch {
|
|
9051
|
+
return false;
|
|
9052
|
+
}
|
|
9053
|
+
const nested = body.error && typeof body.error === "object" && !Array.isArray(body.error) ? body.error : void 0;
|
|
9054
|
+
const errorLabel = typeof body.error === "string" ? body.error : "";
|
|
9055
|
+
const code = normalizePaymentSignal(body.code ?? nested?.code);
|
|
9056
|
+
const reason = normalizePaymentSignal(body.reason);
|
|
9057
|
+
const detail = normalizePaymentSignal(body.invalidMessage);
|
|
9058
|
+
const message = normalizePaymentSignal(nested?.message ?? body.message);
|
|
9059
|
+
const label = normalizePaymentSignal(errorLabel);
|
|
9060
|
+
if (code.includes("settlementfailed") || label.includes("settlementfailed") || message.includes("settlementfailed")) return false;
|
|
9061
|
+
const verifyPhase = code === "paymentinvalid" || label.includes("verificationfailed") || message.includes("verificationfailed");
|
|
9062
|
+
if (!verifyPhase) return false;
|
|
9063
|
+
return code === "paymentblockhashstale" || detail.includes("blockhashnotfound") || detail.includes("blockheightexceeded") || reason === "expiredsignature" || message.includes("expiredsignature");
|
|
9064
|
+
}
|
|
9065
|
+
async function waitForStaleRetry(attempt) {
|
|
9066
|
+
await new Promise(
|
|
9067
|
+
(resolve) => setTimeout(resolve, STALE_BLOCKHASH_RETRY_BACKOFFS_MS[attempt])
|
|
9068
|
+
);
|
|
9069
|
+
}
|
|
8507
9070
|
var DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc";
|
|
8508
9071
|
function resolveRpcConfig(rpcUrl, rpcHeaders) {
|
|
8509
9072
|
const env = typeof process !== "undefined" && process.env ? process.env : {};
|
|
@@ -8950,26 +9513,34 @@ var SolanaLLMClient = class {
|
|
|
8950
9513
|
}
|
|
8951
9514
|
async requestWithPayment(endpoint, body) {
|
|
8952
9515
|
const url = `${this.apiUrl}${endpoint}`;
|
|
8953
|
-
|
|
8954
|
-
|
|
8955
|
-
|
|
8956
|
-
|
|
8957
|
-
|
|
8958
|
-
|
|
8959
|
-
|
|
8960
|
-
|
|
8961
|
-
|
|
8962
|
-
|
|
8963
|
-
|
|
8964
|
-
|
|
8965
|
-
|
|
8966
|
-
|
|
9516
|
+
for (let staleRetries = 0; ; ) {
|
|
9517
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9518
|
+
method: "POST",
|
|
9519
|
+
headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
|
|
9520
|
+
body: JSON.stringify(body)
|
|
9521
|
+
});
|
|
9522
|
+
if (response.status === 402) {
|
|
9523
|
+
try {
|
|
9524
|
+
return await this.handlePaymentAndRetry(url, body, response, staleRetries > 0);
|
|
9525
|
+
} catch (error) {
|
|
9526
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9527
|
+
await waitForStaleRetry(staleRetries++);
|
|
9528
|
+
continue;
|
|
9529
|
+
}
|
|
8967
9530
|
}
|
|
8968
|
-
|
|
9531
|
+
if (!response.ok) {
|
|
9532
|
+
let errorBody;
|
|
9533
|
+
try {
|
|
9534
|
+
errorBody = await response.json();
|
|
9535
|
+
} catch {
|
|
9536
|
+
errorBody = { error: "Request failed" };
|
|
9537
|
+
}
|
|
9538
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9539
|
+
}
|
|
9540
|
+
return response.json();
|
|
8969
9541
|
}
|
|
8970
|
-
return response.json();
|
|
8971
9542
|
}
|
|
8972
|
-
async handlePaymentAndRetry(url, body, response) {
|
|
9543
|
+
async handlePaymentAndRetry(url, body, response, forceFreshBlockhash = false) {
|
|
8973
9544
|
let paymentHeader = response.headers.get("payment-required");
|
|
8974
9545
|
if (!paymentHeader) {
|
|
8975
9546
|
try {
|
|
@@ -9011,7 +9582,8 @@ var SolanaLLMClient = class {
|
|
|
9011
9582
|
extra: details.extra,
|
|
9012
9583
|
extensions,
|
|
9013
9584
|
rpcUrl: this.rpcUrl,
|
|
9014
|
-
rpcHeaders: this.rpcHeaders
|
|
9585
|
+
rpcHeaders: this.rpcHeaders,
|
|
9586
|
+
forceFreshBlockhash
|
|
9015
9587
|
}
|
|
9016
9588
|
);
|
|
9017
9589
|
const retryResponse = await this.fetchWithTimeout(url, {
|
|
@@ -9024,6 +9596,9 @@ var SolanaLLMClient = class {
|
|
|
9024
9596
|
body: JSON.stringify(body)
|
|
9025
9597
|
});
|
|
9026
9598
|
if (retryResponse.status === 402) {
|
|
9599
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9600
|
+
throw new SafeStaleBlockhashError();
|
|
9601
|
+
}
|
|
9027
9602
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9028
9603
|
}
|
|
9029
9604
|
if (!retryResponse.ok) {
|
|
@@ -9042,26 +9617,34 @@ var SolanaLLMClient = class {
|
|
|
9042
9617
|
}
|
|
9043
9618
|
async requestWithPaymentRaw(endpoint, body) {
|
|
9044
9619
|
const url = `${this.apiUrl}${endpoint}`;
|
|
9045
|
-
|
|
9046
|
-
|
|
9047
|
-
|
|
9048
|
-
|
|
9049
|
-
|
|
9050
|
-
|
|
9051
|
-
|
|
9052
|
-
|
|
9053
|
-
|
|
9054
|
-
|
|
9055
|
-
|
|
9056
|
-
|
|
9057
|
-
|
|
9058
|
-
|
|
9620
|
+
for (let staleRetries = 0; ; ) {
|
|
9621
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9622
|
+
method: "POST",
|
|
9623
|
+
headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
|
|
9624
|
+
body: JSON.stringify(body)
|
|
9625
|
+
});
|
|
9626
|
+
if (response.status === 402) {
|
|
9627
|
+
try {
|
|
9628
|
+
return await this.handlePaymentAndRetryRaw(url, body, response, staleRetries > 0);
|
|
9629
|
+
} catch (error) {
|
|
9630
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9631
|
+
await waitForStaleRetry(staleRetries++);
|
|
9632
|
+
continue;
|
|
9633
|
+
}
|
|
9059
9634
|
}
|
|
9060
|
-
|
|
9635
|
+
if (!response.ok) {
|
|
9636
|
+
let errorBody;
|
|
9637
|
+
try {
|
|
9638
|
+
errorBody = await response.json();
|
|
9639
|
+
} catch {
|
|
9640
|
+
errorBody = { error: "Request failed" };
|
|
9641
|
+
}
|
|
9642
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9643
|
+
}
|
|
9644
|
+
return response.json();
|
|
9061
9645
|
}
|
|
9062
|
-
return response.json();
|
|
9063
9646
|
}
|
|
9064
|
-
async handlePaymentAndRetryRaw(url, body, response) {
|
|
9647
|
+
async handlePaymentAndRetryRaw(url, body, response, forceFreshBlockhash = false) {
|
|
9065
9648
|
let paymentHeader = response.headers.get("payment-required");
|
|
9066
9649
|
if (!paymentHeader) {
|
|
9067
9650
|
try {
|
|
@@ -9103,7 +9686,8 @@ var SolanaLLMClient = class {
|
|
|
9103
9686
|
extra: details.extra,
|
|
9104
9687
|
extensions,
|
|
9105
9688
|
rpcUrl: this.rpcUrl,
|
|
9106
|
-
rpcHeaders: this.rpcHeaders
|
|
9689
|
+
rpcHeaders: this.rpcHeaders,
|
|
9690
|
+
forceFreshBlockhash
|
|
9107
9691
|
}
|
|
9108
9692
|
);
|
|
9109
9693
|
const retryResponse = await this.fetchWithTimeout(url, {
|
|
@@ -9116,6 +9700,9 @@ var SolanaLLMClient = class {
|
|
|
9116
9700
|
body: JSON.stringify(body)
|
|
9117
9701
|
});
|
|
9118
9702
|
if (retryResponse.status === 402) {
|
|
9703
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9704
|
+
throw new SafeStaleBlockhashError();
|
|
9705
|
+
}
|
|
9119
9706
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9120
9707
|
}
|
|
9121
9708
|
if (!retryResponse.ok) {
|
|
@@ -9135,25 +9722,33 @@ var SolanaLLMClient = class {
|
|
|
9135
9722
|
async getWithPaymentRaw(endpoint, params) {
|
|
9136
9723
|
const query = params ? "?" + new URLSearchParams(params).toString() : "";
|
|
9137
9724
|
const url = `${this.apiUrl}${endpoint}${query}`;
|
|
9138
|
-
|
|
9139
|
-
|
|
9140
|
-
|
|
9141
|
-
|
|
9142
|
-
|
|
9143
|
-
|
|
9144
|
-
|
|
9145
|
-
|
|
9146
|
-
|
|
9147
|
-
|
|
9148
|
-
|
|
9149
|
-
|
|
9150
|
-
|
|
9725
|
+
for (let staleRetries = 0; ; ) {
|
|
9726
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9727
|
+
method: "GET",
|
|
9728
|
+
headers: { "User-Agent": USER_AGENT }
|
|
9729
|
+
});
|
|
9730
|
+
if (response.status === 402) {
|
|
9731
|
+
try {
|
|
9732
|
+
return await this.handleGetPaymentAndRetryRaw(url, endpoint, params, response, staleRetries > 0);
|
|
9733
|
+
} catch (error) {
|
|
9734
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9735
|
+
await waitForStaleRetry(staleRetries++);
|
|
9736
|
+
continue;
|
|
9737
|
+
}
|
|
9151
9738
|
}
|
|
9152
|
-
|
|
9739
|
+
if (!response.ok) {
|
|
9740
|
+
let errorBody;
|
|
9741
|
+
try {
|
|
9742
|
+
errorBody = await response.json();
|
|
9743
|
+
} catch {
|
|
9744
|
+
errorBody = { error: "Request failed" };
|
|
9745
|
+
}
|
|
9746
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9747
|
+
}
|
|
9748
|
+
return response.json();
|
|
9153
9749
|
}
|
|
9154
|
-
return response.json();
|
|
9155
9750
|
}
|
|
9156
|
-
async handleGetPaymentAndRetryRaw(url, endpoint, params, response) {
|
|
9751
|
+
async handleGetPaymentAndRetryRaw(url, endpoint, params, response, forceFreshBlockhash = false) {
|
|
9157
9752
|
let paymentHeader = response.headers.get("payment-required");
|
|
9158
9753
|
if (!paymentHeader) {
|
|
9159
9754
|
try {
|
|
@@ -9195,7 +9790,8 @@ var SolanaLLMClient = class {
|
|
|
9195
9790
|
extra: details.extra,
|
|
9196
9791
|
extensions,
|
|
9197
9792
|
rpcUrl: this.rpcUrl,
|
|
9198
|
-
rpcHeaders: this.rpcHeaders
|
|
9793
|
+
rpcHeaders: this.rpcHeaders,
|
|
9794
|
+
forceFreshBlockhash
|
|
9199
9795
|
}
|
|
9200
9796
|
);
|
|
9201
9797
|
const query = params ? "?" + new URLSearchParams(params).toString() : "";
|
|
@@ -9208,6 +9804,9 @@ var SolanaLLMClient = class {
|
|
|
9208
9804
|
}
|
|
9209
9805
|
});
|
|
9210
9806
|
if (retryResponse.status === 402) {
|
|
9807
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9808
|
+
throw new SafeStaleBlockhashError();
|
|
9809
|
+
}
|
|
9211
9810
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9212
9811
|
}
|
|
9213
9812
|
if (!retryResponse.ok) {
|