@blockrun/llm 3.13.1 → 3.13.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +219 -187
- package/dist/index.cjs +1121 -522
- package/dist/index.d.cts +16 -0
- package/dist/index.d.ts +16 -0
- package/dist/index.js +1121 -522
- package/package.json +2 -2
package/dist/index.cjs
CHANGED
|
@@ -145,7 +145,7 @@ var APIError = class extends BlockrunError {
|
|
|
145
145
|
}
|
|
146
146
|
};
|
|
147
147
|
|
|
148
|
-
// node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+
|
|
148
|
+
// node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+5d911879d7f1e_fbldkf3f3jmlwtwu53ftq2mj24/node_modules/@blockrun/router-core/dist/index.js
|
|
149
149
|
function scoreTokenCount(estimatedTokens, thresholds) {
|
|
150
150
|
if (estimatedTokens < thresholds.simple) {
|
|
151
151
|
return { name: "tokenCount", score: -1, signal: `short (${estimatedTokens} tokens)` };
|
|
@@ -479,6 +479,21 @@ function applyPromotions(tierConfigs, promotions, profile, now = /* @__PURE__ */
|
|
|
479
479
|
}
|
|
480
480
|
return result;
|
|
481
481
|
}
|
|
482
|
+
function applyUnavailableModels(tierConfigs, unavailableModels) {
|
|
483
|
+
if (!unavailableModels || unavailableModels.length === 0) return tierConfigs;
|
|
484
|
+
const dead = new Set(unavailableModels);
|
|
485
|
+
let result = tierConfigs;
|
|
486
|
+
for (const tier of Object.keys(tierConfigs)) {
|
|
487
|
+
const config = tierConfigs[tier];
|
|
488
|
+
const alive = [config.primary, ...config.fallback].filter((model) => !dead.has(model));
|
|
489
|
+
if (alive.length === 0 || alive[0] === config.primary && alive.length === config.fallback.length + 1) {
|
|
490
|
+
continue;
|
|
491
|
+
}
|
|
492
|
+
if (result === tierConfigs) result = { ...tierConfigs };
|
|
493
|
+
result[tier] = { primary: alive[0], fallback: alive.slice(1) };
|
|
494
|
+
}
|
|
495
|
+
return result;
|
|
496
|
+
}
|
|
482
497
|
var RulesStrategy = class {
|
|
483
498
|
name = "rules";
|
|
484
499
|
route(prompt, systemPrompt, maxOutputTokens, options) {
|
|
@@ -530,6 +545,7 @@ ${value.slice(-(scanLimit - prefixLength))}`;
|
|
|
530
545
|
profile = useAgenticTiers ? "agentic" : "auto";
|
|
531
546
|
}
|
|
532
547
|
tierConfigs = applyPromotions(tierConfigs, config.promotions, profile, options.now);
|
|
548
|
+
tierConfigs = applyUnavailableModels(tierConfigs, options.unavailableModels);
|
|
533
549
|
const agenticScoreValue = ruleResult.agenticScore;
|
|
534
550
|
if (estimatedTokens > config.overrides.maxTokensForceComplex) {
|
|
535
551
|
const decision2 = selectModel(
|
|
@@ -603,14 +619,15 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
603
619
|
supportsVision: true
|
|
604
620
|
},
|
|
605
621
|
"anthropic/claude-haiku-4.5": {
|
|
622
|
+
// override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
|
|
606
623
|
contextWindow: 2e5,
|
|
607
|
-
maxOutputTokens:
|
|
624
|
+
maxOutputTokens: 64e3,
|
|
608
625
|
supportsTools: true,
|
|
609
626
|
supportsVision: true
|
|
610
627
|
},
|
|
611
|
-
"anthropic/claude-opus-4.
|
|
612
|
-
contextWindow:
|
|
613
|
-
maxOutputTokens:
|
|
628
|
+
"anthropic/claude-opus-4.5": {
|
|
629
|
+
contextWindow: 2e5,
|
|
630
|
+
maxOutputTokens: 64e3,
|
|
614
631
|
supportsTools: true,
|
|
615
632
|
supportsVision: true
|
|
616
633
|
},
|
|
@@ -632,12 +649,19 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
632
649
|
supportsTools: true,
|
|
633
650
|
supportsVision: true
|
|
634
651
|
},
|
|
635
|
-
"anthropic/claude-sonnet-4.
|
|
652
|
+
"anthropic/claude-sonnet-4.5": {
|
|
636
653
|
contextWindow: 2e5,
|
|
637
654
|
maxOutputTokens: 64e3,
|
|
638
655
|
supportsTools: true,
|
|
639
656
|
supportsVision: true
|
|
640
657
|
},
|
|
658
|
+
"anthropic/claude-sonnet-4.6": {
|
|
659
|
+
// override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
|
|
660
|
+
contextWindow: 1e6,
|
|
661
|
+
maxOutputTokens: 128e3,
|
|
662
|
+
supportsTools: true,
|
|
663
|
+
supportsVision: true
|
|
664
|
+
},
|
|
641
665
|
"anthropic/claude-sonnet-5": {
|
|
642
666
|
contextWindow: 1e6,
|
|
643
667
|
maxOutputTokens: 128e3,
|
|
@@ -645,14 +669,14 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
645
669
|
supportsVision: true
|
|
646
670
|
},
|
|
647
671
|
"deepseek/deepseek-chat": {
|
|
648
|
-
contextWindow:
|
|
649
|
-
maxOutputTokens:
|
|
672
|
+
contextWindow: 1048576,
|
|
673
|
+
maxOutputTokens: 65536,
|
|
650
674
|
supportsTools: true,
|
|
651
675
|
supportsVision: false
|
|
652
676
|
},
|
|
653
677
|
"deepseek/deepseek-reasoner": {
|
|
654
|
-
contextWindow:
|
|
655
|
-
maxOutputTokens:
|
|
678
|
+
contextWindow: 1048576,
|
|
679
|
+
maxOutputTokens: 65536,
|
|
656
680
|
supportsTools: true,
|
|
657
681
|
supportsVision: false
|
|
658
682
|
},
|
|
@@ -662,62 +686,38 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
662
686
|
supportsTools: true,
|
|
663
687
|
supportsVision: false
|
|
664
688
|
},
|
|
665
|
-
"free/deepseek-v4-flash": {
|
|
666
|
-
contextWindow: 1e6,
|
|
667
|
-
maxOutputTokens: 16384,
|
|
668
|
-
supportsTools: false,
|
|
669
|
-
supportsVision: false
|
|
670
|
-
},
|
|
671
|
-
"free/gpt-oss-120b": {
|
|
672
|
-
contextWindow: 128e3,
|
|
673
|
-
maxOutputTokens: 16384,
|
|
674
|
-
supportsTools: false,
|
|
675
|
-
supportsVision: false
|
|
676
|
-
},
|
|
677
|
-
"free/gpt-oss-20b": {
|
|
678
|
-
contextWindow: 128e3,
|
|
679
|
-
maxOutputTokens: 16384,
|
|
680
|
-
supportsTools: false,
|
|
681
|
-
supportsVision: false
|
|
682
|
-
},
|
|
683
|
-
"free/seed-oss-36b": {
|
|
684
|
-
contextWindow: 131072,
|
|
685
|
-
maxOutputTokens: 16384,
|
|
686
|
-
supportsTools: false,
|
|
687
|
-
supportsVision: false
|
|
688
|
-
},
|
|
689
689
|
"google/gemini-2.5-flash": {
|
|
690
|
-
contextWindow:
|
|
690
|
+
contextWindow: 1048576,
|
|
691
691
|
maxOutputTokens: 65536,
|
|
692
692
|
supportsTools: true,
|
|
693
693
|
supportsVision: true
|
|
694
694
|
},
|
|
695
695
|
"google/gemini-2.5-flash-lite": {
|
|
696
|
-
contextWindow:
|
|
696
|
+
contextWindow: 1048576,
|
|
697
697
|
maxOutputTokens: 65536,
|
|
698
698
|
supportsTools: true,
|
|
699
699
|
supportsVision: false
|
|
700
700
|
},
|
|
701
701
|
"google/gemini-2.5-pro": {
|
|
702
|
-
contextWindow:
|
|
702
|
+
contextWindow: 1048576,
|
|
703
703
|
maxOutputTokens: 65536,
|
|
704
704
|
supportsTools: true,
|
|
705
705
|
supportsVision: true
|
|
706
706
|
},
|
|
707
707
|
"google/gemini-3-flash-preview": {
|
|
708
|
-
contextWindow:
|
|
708
|
+
contextWindow: 1048576,
|
|
709
709
|
maxOutputTokens: 65536,
|
|
710
|
-
supportsTools:
|
|
710
|
+
supportsTools: true,
|
|
711
711
|
supportsVision: true
|
|
712
712
|
},
|
|
713
713
|
"google/gemini-3.1-flash-lite": {
|
|
714
|
-
contextWindow:
|
|
715
|
-
maxOutputTokens:
|
|
714
|
+
contextWindow: 1048576,
|
|
715
|
+
maxOutputTokens: 65536,
|
|
716
716
|
supportsTools: true,
|
|
717
717
|
supportsVision: false
|
|
718
718
|
},
|
|
719
719
|
"google/gemini-3.1-pro": {
|
|
720
|
-
contextWindow:
|
|
720
|
+
contextWindow: 1048576,
|
|
721
721
|
maxOutputTokens: 65536,
|
|
722
722
|
supportsTools: true,
|
|
723
723
|
supportsVision: true
|
|
@@ -728,23 +728,29 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
728
728
|
supportsTools: true,
|
|
729
729
|
supportsVision: true
|
|
730
730
|
},
|
|
731
|
-
"
|
|
732
|
-
contextWindow:
|
|
733
|
-
maxOutputTokens:
|
|
731
|
+
"google/gemini-3.5-flash-lite": {
|
|
732
|
+
contextWindow: 1048576,
|
|
733
|
+
maxOutputTokens: 65536,
|
|
734
734
|
supportsTools: true,
|
|
735
|
-
supportsVision:
|
|
735
|
+
supportsVision: false
|
|
736
736
|
},
|
|
737
|
-
"
|
|
738
|
-
contextWindow:
|
|
737
|
+
"google/gemini-3.6-flash": {
|
|
738
|
+
contextWindow: 1048576,
|
|
739
739
|
maxOutputTokens: 65536,
|
|
740
740
|
supportsTools: true,
|
|
741
741
|
supportsVision: true
|
|
742
742
|
},
|
|
743
|
-
"
|
|
744
|
-
contextWindow:
|
|
743
|
+
"minimax/minimax-m2.7": {
|
|
744
|
+
contextWindow: 204800,
|
|
745
|
+
maxOutputTokens: 16384,
|
|
746
|
+
supportsTools: true,
|
|
747
|
+
supportsVision: false
|
|
748
|
+
},
|
|
749
|
+
"minimax/minimax-m3": {
|
|
750
|
+
contextWindow: 1048576,
|
|
745
751
|
maxOutputTokens: 65536,
|
|
746
752
|
supportsTools: true,
|
|
747
|
-
supportsVision:
|
|
753
|
+
supportsVision: false
|
|
748
754
|
},
|
|
749
755
|
"moonshot/kimi-k3": {
|
|
750
756
|
contextWindow: 1048576,
|
|
@@ -752,7 +758,63 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
752
758
|
supportsTools: true,
|
|
753
759
|
supportsVision: true
|
|
754
760
|
},
|
|
761
|
+
"nvidia/mistral-nemotron": {
|
|
762
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
763
|
+
contextWindow: 131072,
|
|
764
|
+
maxOutputTokens: 16384,
|
|
765
|
+
supportsTools: false,
|
|
766
|
+
supportsVision: false
|
|
767
|
+
},
|
|
768
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
|
|
769
|
+
contextWindow: 256e3,
|
|
770
|
+
maxOutputTokens: 16384,
|
|
771
|
+
supportsTools: false,
|
|
772
|
+
supportsVision: true
|
|
773
|
+
},
|
|
774
|
+
"nvidia/nemotron-nano-12b-v2-vl": {
|
|
775
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
776
|
+
contextWindow: 131072,
|
|
777
|
+
maxOutputTokens: 16384,
|
|
778
|
+
supportsTools: false,
|
|
779
|
+
supportsVision: true
|
|
780
|
+
},
|
|
781
|
+
"nvidia/nemotron-nano-9b-v2": {
|
|
782
|
+
contextWindow: 131072,
|
|
783
|
+
maxOutputTokens: 16384,
|
|
784
|
+
supportsTools: false,
|
|
785
|
+
supportsVision: false
|
|
786
|
+
},
|
|
787
|
+
"nvidia/step-3.7-flash": {
|
|
788
|
+
contextWindow: 131072,
|
|
789
|
+
maxOutputTokens: 16384,
|
|
790
|
+
supportsTools: false,
|
|
791
|
+
supportsVision: false
|
|
792
|
+
},
|
|
793
|
+
"openai/chat-latest": {
|
|
794
|
+
contextWindow: 128e3,
|
|
795
|
+
maxOutputTokens: 128e3,
|
|
796
|
+
supportsTools: true,
|
|
797
|
+
supportsVision: true
|
|
798
|
+
},
|
|
755
799
|
"openai/gpt-4.1": {
|
|
800
|
+
contextWindow: 128e3,
|
|
801
|
+
maxOutputTokens: 32768,
|
|
802
|
+
supportsTools: true,
|
|
803
|
+
supportsVision: true
|
|
804
|
+
},
|
|
805
|
+
"openai/gpt-4.1-mini": {
|
|
806
|
+
contextWindow: 128e3,
|
|
807
|
+
maxOutputTokens: 32768,
|
|
808
|
+
supportsTools: true,
|
|
809
|
+
supportsVision: false
|
|
810
|
+
},
|
|
811
|
+
"openai/gpt-4.1-nano": {
|
|
812
|
+
contextWindow: 128e3,
|
|
813
|
+
maxOutputTokens: 32768,
|
|
814
|
+
supportsTools: true,
|
|
815
|
+
supportsVision: false
|
|
816
|
+
},
|
|
817
|
+
"openai/gpt-4o": {
|
|
756
818
|
contextWindow: 128e3,
|
|
757
819
|
maxOutputTokens: 16384,
|
|
758
820
|
supportsTools: true,
|
|
@@ -766,17 +828,44 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
766
828
|
},
|
|
767
829
|
"openai/gpt-5-mini": {
|
|
768
830
|
contextWindow: 2e5,
|
|
769
|
-
maxOutputTokens:
|
|
831
|
+
maxOutputTokens: 128e3,
|
|
770
832
|
supportsTools: true,
|
|
771
833
|
supportsVision: false
|
|
772
834
|
},
|
|
835
|
+
"openai/gpt-5.2": {
|
|
836
|
+
contextWindow: 4e5,
|
|
837
|
+
maxOutputTokens: 128e3,
|
|
838
|
+
supportsTools: true,
|
|
839
|
+
supportsVision: true
|
|
840
|
+
},
|
|
841
|
+
"openai/gpt-5.2-pro": {
|
|
842
|
+
// supportsTools: not probed — fails closed
|
|
843
|
+
contextWindow: 4e5,
|
|
844
|
+
maxOutputTokens: 128e3,
|
|
845
|
+
supportsTools: false,
|
|
846
|
+
supportsVision: true
|
|
847
|
+
},
|
|
848
|
+
"openai/gpt-5.3": {
|
|
849
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
850
|
+
contextWindow: 128e3,
|
|
851
|
+
maxOutputTokens: 128e3,
|
|
852
|
+
supportsTools: false,
|
|
853
|
+
supportsVision: true
|
|
854
|
+
},
|
|
773
855
|
"openai/gpt-5.3-codex": {
|
|
856
|
+
// supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe measured an incident, not the model. Codex's function calling is established by the 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot claiming the model cannot call tools.
|
|
774
857
|
contextWindow: 4e5,
|
|
775
858
|
maxOutputTokens: 128e3,
|
|
776
859
|
supportsTools: true,
|
|
777
860
|
supportsVision: false
|
|
778
861
|
},
|
|
779
862
|
"openai/gpt-5.4": {
|
|
863
|
+
contextWindow: 105e4,
|
|
864
|
+
maxOutputTokens: 128e3,
|
|
865
|
+
supportsTools: true,
|
|
866
|
+
supportsVision: true
|
|
867
|
+
},
|
|
868
|
+
"openai/gpt-5.4-mini": {
|
|
780
869
|
contextWindow: 4e5,
|
|
781
870
|
maxOutputTokens: 128e3,
|
|
782
871
|
supportsTools: true,
|
|
@@ -784,30 +873,92 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
784
873
|
},
|
|
785
874
|
"openai/gpt-5.4-nano": {
|
|
786
875
|
contextWindow: 105e4,
|
|
787
|
-
maxOutputTokens:
|
|
876
|
+
maxOutputTokens: 128e3,
|
|
788
877
|
supportsTools: true,
|
|
789
878
|
supportsVision: false
|
|
790
879
|
},
|
|
880
|
+
"openai/gpt-5.4-pro": {
|
|
881
|
+
// supportsTools: not probed — fails closed
|
|
882
|
+
contextWindow: 105e4,
|
|
883
|
+
maxOutputTokens: 128e3,
|
|
884
|
+
supportsTools: false,
|
|
885
|
+
supportsVision: true
|
|
886
|
+
},
|
|
791
887
|
"openai/gpt-5.5": {
|
|
792
888
|
contextWindow: 105e4,
|
|
793
889
|
maxOutputTokens: 128e3,
|
|
794
890
|
supportsTools: true,
|
|
795
891
|
supportsVision: true
|
|
796
892
|
},
|
|
893
|
+
"openai/gpt-5.5-pro": {
|
|
894
|
+
// supportsTools: not probed — fails closed
|
|
895
|
+
contextWindow: 105e4,
|
|
896
|
+
maxOutputTokens: 128e3,
|
|
897
|
+
supportsTools: false,
|
|
898
|
+
supportsVision: true
|
|
899
|
+
},
|
|
900
|
+
"openai/gpt-5.6-luna": {
|
|
901
|
+
contextWindow: 105e4,
|
|
902
|
+
maxOutputTokens: 128e3,
|
|
903
|
+
supportsTools: true,
|
|
904
|
+
supportsVision: true
|
|
905
|
+
},
|
|
906
|
+
"openai/gpt-5.6-luna-pro": {
|
|
907
|
+
contextWindow: 105e4,
|
|
908
|
+
maxOutputTokens: 128e3,
|
|
909
|
+
supportsTools: false,
|
|
910
|
+
supportsVision: true
|
|
911
|
+
},
|
|
912
|
+
"openai/gpt-5.6-sol": {
|
|
913
|
+
contextWindow: 105e4,
|
|
914
|
+
maxOutputTokens: 128e3,
|
|
915
|
+
supportsTools: true,
|
|
916
|
+
supportsVision: true
|
|
917
|
+
},
|
|
918
|
+
"openai/gpt-5.6-sol-pro": {
|
|
919
|
+
contextWindow: 105e4,
|
|
920
|
+
maxOutputTokens: 128e3,
|
|
921
|
+
supportsTools: true,
|
|
922
|
+
supportsVision: true
|
|
923
|
+
},
|
|
797
924
|
"openai/gpt-5.6-terra": {
|
|
798
925
|
contextWindow: 105e4,
|
|
799
926
|
maxOutputTokens: 128e3,
|
|
800
927
|
supportsTools: true,
|
|
801
928
|
supportsVision: true
|
|
802
929
|
},
|
|
930
|
+
"openai/gpt-5.6-terra-pro": {
|
|
931
|
+
contextWindow: 105e4,
|
|
932
|
+
maxOutputTokens: 128e3,
|
|
933
|
+
supportsTools: true,
|
|
934
|
+
supportsVision: true
|
|
935
|
+
},
|
|
936
|
+
"openai/o1": {
|
|
937
|
+
contextWindow: 2e5,
|
|
938
|
+
maxOutputTokens: 1e5,
|
|
939
|
+
supportsTools: true,
|
|
940
|
+
supportsVision: false
|
|
941
|
+
},
|
|
803
942
|
"openai/o3": {
|
|
804
943
|
contextWindow: 2e5,
|
|
805
944
|
maxOutputTokens: 1e5,
|
|
806
945
|
supportsTools: true,
|
|
807
946
|
supportsVision: false
|
|
808
947
|
},
|
|
948
|
+
"openai/o3-mini": {
|
|
949
|
+
contextWindow: 128e3,
|
|
950
|
+
maxOutputTokens: 1e5,
|
|
951
|
+
supportsTools: true,
|
|
952
|
+
supportsVision: false
|
|
953
|
+
},
|
|
809
954
|
"openai/o4-mini": {
|
|
810
955
|
contextWindow: 128e3,
|
|
956
|
+
maxOutputTokens: 1e5,
|
|
957
|
+
supportsTools: true,
|
|
958
|
+
supportsVision: false
|
|
959
|
+
},
|
|
960
|
+
"qwen/qwen3.7-flash": {
|
|
961
|
+
contextWindow: 1e6,
|
|
811
962
|
maxOutputTokens: 65536,
|
|
812
963
|
supportsTools: true,
|
|
813
964
|
supportsVision: false
|
|
@@ -818,299 +969,605 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
818
969
|
supportsTools: true,
|
|
819
970
|
supportsVision: false
|
|
820
971
|
},
|
|
821
|
-
"
|
|
822
|
-
contextWindow:
|
|
823
|
-
maxOutputTokens:
|
|
972
|
+
"qwen/qwen3.7-plus": {
|
|
973
|
+
contextWindow: 1e6,
|
|
974
|
+
maxOutputTokens: 131072,
|
|
824
975
|
supportsTools: true,
|
|
825
976
|
supportsVision: false
|
|
826
977
|
},
|
|
827
|
-
"
|
|
828
|
-
contextWindow:
|
|
829
|
-
maxOutputTokens:
|
|
978
|
+
"tencent/hy3": {
|
|
979
|
+
contextWindow: 262144,
|
|
980
|
+
maxOutputTokens: 128e3,
|
|
830
981
|
supportsTools: true,
|
|
831
982
|
supportsVision: false
|
|
832
983
|
},
|
|
833
|
-
"xai/grok-4
|
|
834
|
-
contextWindow:
|
|
984
|
+
"xai/grok-4.3": {
|
|
985
|
+
contextWindow: 1e6,
|
|
835
986
|
maxOutputTokens: 16384,
|
|
836
987
|
supportsTools: true,
|
|
837
|
-
supportsVision:
|
|
988
|
+
supportsVision: true
|
|
838
989
|
},
|
|
839
|
-
"xai/grok-4
|
|
840
|
-
contextWindow:
|
|
990
|
+
"xai/grok-4.5": {
|
|
991
|
+
contextWindow: 5e5,
|
|
841
992
|
maxOutputTokens: 16384,
|
|
842
993
|
supportsTools: true,
|
|
843
|
-
supportsVision:
|
|
994
|
+
supportsVision: true
|
|
844
995
|
},
|
|
845
|
-
"xai/grok-
|
|
846
|
-
contextWindow:
|
|
996
|
+
"xai/grok-build-0.1": {
|
|
997
|
+
contextWindow: 256e3,
|
|
847
998
|
maxOutputTokens: 16384,
|
|
848
999
|
supportsTools: true,
|
|
849
1000
|
supportsVision: false
|
|
850
1001
|
},
|
|
851
|
-
"
|
|
852
|
-
contextWindow:
|
|
853
|
-
maxOutputTokens:
|
|
854
|
-
supportsTools: true,
|
|
855
|
-
supportsVision: false
|
|
1002
|
+
"xiaomi/mimo-v2.5-pro": {
|
|
1003
|
+
contextWindow: 1048576,
|
|
1004
|
+
maxOutputTokens: 131072,
|
|
1005
|
+
supportsTools: true,
|
|
1006
|
+
supportsVision: false
|
|
1007
|
+
},
|
|
1008
|
+
"zai/glm-5": {
|
|
1009
|
+
contextWindow: 2e5,
|
|
1010
|
+
maxOutputTokens: 128e3,
|
|
1011
|
+
supportsTools: true,
|
|
1012
|
+
supportsVision: false
|
|
1013
|
+
},
|
|
1014
|
+
"zai/glm-5-turbo": {
|
|
1015
|
+
contextWindow: 2e5,
|
|
1016
|
+
maxOutputTokens: 128e3,
|
|
1017
|
+
supportsTools: true,
|
|
1018
|
+
supportsVision: false
|
|
1019
|
+
},
|
|
1020
|
+
"zai/glm-5.1": {
|
|
1021
|
+
contextWindow: 2e5,
|
|
1022
|
+
maxOutputTokens: 128e3,
|
|
1023
|
+
supportsTools: true,
|
|
1024
|
+
supportsVision: false
|
|
1025
|
+
},
|
|
1026
|
+
"zai/glm-5.2": {
|
|
1027
|
+
contextWindow: 1e6,
|
|
1028
|
+
maxOutputTokens: 131072,
|
|
1029
|
+
supportsTools: true,
|
|
1030
|
+
supportsVision: false
|
|
1031
|
+
},
|
|
1032
|
+
"zai/glm-5.3": {
|
|
1033
|
+
contextWindow: 1e6,
|
|
1034
|
+
maxOutputTokens: 131072,
|
|
1035
|
+
supportsTools: true,
|
|
1036
|
+
supportsVision: false
|
|
1037
|
+
},
|
|
1038
|
+
"zai/glm-5.3-flash": {
|
|
1039
|
+
contextWindow: 1e6,
|
|
1040
|
+
maxOutputTokens: 131072,
|
|
1041
|
+
supportsTools: true,
|
|
1042
|
+
supportsVision: true
|
|
1043
|
+
}
|
|
1044
|
+
});
|
|
1045
|
+
var model_profiles_generated_default = {
|
|
1046
|
+
"anthropic/claude-fable-5": {
|
|
1047
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1048
|
+
latencyMs: 9298.5,
|
|
1049
|
+
p95LatencyMs: 9873.4,
|
|
1050
|
+
outputTokensPerSecond: 55.17,
|
|
1051
|
+
errorRate: 0,
|
|
1052
|
+
samples: 3
|
|
1053
|
+
},
|
|
1054
|
+
"anthropic/claude-haiku-4.5": {
|
|
1055
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1056
|
+
latencyMs: 3157.4,
|
|
1057
|
+
p95LatencyMs: 3170.7,
|
|
1058
|
+
outputTokensPerSecond: 162.16,
|
|
1059
|
+
errorRate: 0,
|
|
1060
|
+
samples: 3
|
|
1061
|
+
},
|
|
1062
|
+
"anthropic/claude-opus-4.5": {
|
|
1063
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1064
|
+
latencyMs: 6497.7,
|
|
1065
|
+
p95LatencyMs: 6953.7,
|
|
1066
|
+
outputTokensPerSecond: 78.99,
|
|
1067
|
+
errorRate: 0,
|
|
1068
|
+
samples: 3
|
|
1069
|
+
},
|
|
1070
|
+
"anthropic/claude-opus-4.7": {
|
|
1071
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1072
|
+
latencyMs: 5316.5,
|
|
1073
|
+
p95LatencyMs: 6121.5,
|
|
1074
|
+
outputTokensPerSecond: 97.34,
|
|
1075
|
+
errorRate: 0,
|
|
1076
|
+
samples: 3
|
|
1077
|
+
},
|
|
1078
|
+
"anthropic/claude-opus-4.8": {
|
|
1079
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1080
|
+
latencyMs: 6216.1,
|
|
1081
|
+
p95LatencyMs: 6847.7,
|
|
1082
|
+
outputTokensPerSecond: 82.81,
|
|
1083
|
+
errorRate: 0,
|
|
1084
|
+
samples: 3
|
|
1085
|
+
},
|
|
1086
|
+
"anthropic/claude-opus-5": {
|
|
1087
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1088
|
+
latencyMs: 7309,
|
|
1089
|
+
p95LatencyMs: 7745.2,
|
|
1090
|
+
outputTokensPerSecond: 70.17,
|
|
1091
|
+
errorRate: 0,
|
|
1092
|
+
samples: 3
|
|
1093
|
+
},
|
|
1094
|
+
"anthropic/claude-sonnet-4.5": {
|
|
1095
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1096
|
+
latencyMs: 6330.4,
|
|
1097
|
+
p95LatencyMs: 6631.6,
|
|
1098
|
+
outputTokensPerSecond: 81.03,
|
|
1099
|
+
errorRate: 0,
|
|
1100
|
+
samples: 3
|
|
1101
|
+
},
|
|
1102
|
+
"anthropic/claude-sonnet-4.6": {
|
|
1103
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1104
|
+
latencyMs: 6508,
|
|
1105
|
+
p95LatencyMs: 6698.3,
|
|
1106
|
+
outputTokensPerSecond: 78.6,
|
|
1107
|
+
errorRate: 0,
|
|
1108
|
+
samples: 3
|
|
1109
|
+
},
|
|
1110
|
+
"anthropic/claude-sonnet-5": {
|
|
1111
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1112
|
+
latencyMs: 6165.4,
|
|
1113
|
+
p95LatencyMs: 6582.9,
|
|
1114
|
+
outputTokensPerSecond: 83.62,
|
|
1115
|
+
errorRate: 0,
|
|
1116
|
+
samples: 3
|
|
1117
|
+
},
|
|
1118
|
+
"deepseek/deepseek-chat": {
|
|
1119
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1120
|
+
latencyMs: 4351.4,
|
|
1121
|
+
p95LatencyMs: 4543.7,
|
|
1122
|
+
outputTokensPerSecond: 117.78,
|
|
1123
|
+
errorRate: 0,
|
|
1124
|
+
samples: 3
|
|
1125
|
+
},
|
|
1126
|
+
"deepseek/deepseek-reasoner": {
|
|
1127
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1128
|
+
latencyMs: 5201.2,
|
|
1129
|
+
p95LatencyMs: 6079.6,
|
|
1130
|
+
outputTokensPerSecond: 99.77,
|
|
1131
|
+
errorRate: 0,
|
|
1132
|
+
samples: 3
|
|
1133
|
+
},
|
|
1134
|
+
"deepseek/deepseek-v4-pro": {
|
|
1135
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1136
|
+
latencyMs: 8781.2,
|
|
1137
|
+
p95LatencyMs: 9881.1,
|
|
1138
|
+
outputTokensPerSecond: 58.98,
|
|
1139
|
+
errorRate: 0,
|
|
1140
|
+
samples: 3
|
|
1141
|
+
},
|
|
1142
|
+
"google/gemini-2.5-flash": {
|
|
1143
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1144
|
+
latencyMs: 5416.4,
|
|
1145
|
+
p95LatencyMs: 6442.8,
|
|
1146
|
+
outputTokensPerSecond: 213.07,
|
|
1147
|
+
errorRate: 0,
|
|
1148
|
+
samples: 3
|
|
1149
|
+
},
|
|
1150
|
+
"google/gemini-2.5-flash-lite": {
|
|
1151
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1152
|
+
latencyMs: 5002.6,
|
|
1153
|
+
p95LatencyMs: 5780.3,
|
|
1154
|
+
outputTokensPerSecond: 408.43,
|
|
1155
|
+
errorRate: 0,
|
|
1156
|
+
samples: 3
|
|
1157
|
+
},
|
|
1158
|
+
"google/gemini-2.5-pro": {
|
|
1159
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1160
|
+
latencyMs: 28169.5,
|
|
1161
|
+
p95LatencyMs: 29491.4,
|
|
1162
|
+
outputTokensPerSecond: 147.3,
|
|
1163
|
+
errorRate: 0,
|
|
1164
|
+
samples: 3
|
|
1165
|
+
},
|
|
1166
|
+
"google/gemini-3-flash-preview": {
|
|
1167
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1168
|
+
latencyMs: 4717.1,
|
|
1169
|
+
p95LatencyMs: 5037.1,
|
|
1170
|
+
outputTokensPerSecond: 198.71,
|
|
1171
|
+
errorRate: 0,
|
|
1172
|
+
samples: 3
|
|
1173
|
+
},
|
|
1174
|
+
"google/gemini-3.1-flash-lite": {
|
|
1175
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1176
|
+
latencyMs: 2855.8,
|
|
1177
|
+
p95LatencyMs: 3172.7,
|
|
1178
|
+
outputTokensPerSecond: 286.91,
|
|
1179
|
+
errorRate: 0,
|
|
1180
|
+
samples: 3
|
|
1181
|
+
},
|
|
1182
|
+
"google/gemini-3.1-pro": {
|
|
1183
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1184
|
+
latencyMs: 24194.1,
|
|
1185
|
+
p95LatencyMs: 27269.6,
|
|
1186
|
+
outputTokensPerSecond: 109.47,
|
|
1187
|
+
errorRate: 0,
|
|
1188
|
+
samples: 3
|
|
1189
|
+
},
|
|
1190
|
+
"google/gemini-3.5-flash": {
|
|
1191
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1192
|
+
latencyMs: 5320.6,
|
|
1193
|
+
p95LatencyMs: 5429.8,
|
|
1194
|
+
outputTokensPerSecond: 226.21,
|
|
1195
|
+
errorRate: 0,
|
|
1196
|
+
samples: 3
|
|
1197
|
+
},
|
|
1198
|
+
"google/gemini-3.5-flash-lite": {
|
|
1199
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1200
|
+
latencyMs: 3515.8,
|
|
1201
|
+
p95LatencyMs: 4363.4,
|
|
1202
|
+
outputTokensPerSecond: 248.9,
|
|
1203
|
+
errorRate: 0,
|
|
1204
|
+
samples: 3
|
|
1205
|
+
},
|
|
1206
|
+
"google/gemini-3.6-flash": {
|
|
1207
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1208
|
+
latencyMs: 13020,
|
|
1209
|
+
p95LatencyMs: 15383.1,
|
|
1210
|
+
outputTokensPerSecond: 187.87,
|
|
1211
|
+
errorRate: 0,
|
|
1212
|
+
samples: 3
|
|
1213
|
+
},
|
|
1214
|
+
"minimax/minimax-m2.7": {
|
|
1215
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1216
|
+
latencyMs: 8761.1,
|
|
1217
|
+
p95LatencyMs: 10199.3,
|
|
1218
|
+
outputTokensPerSecond: 59.18,
|
|
1219
|
+
errorRate: 0,
|
|
1220
|
+
samples: 3
|
|
1221
|
+
},
|
|
1222
|
+
"minimax/minimax-m3": {
|
|
1223
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1224
|
+
latencyMs: 11101.9,
|
|
1225
|
+
p95LatencyMs: 26087.1,
|
|
1226
|
+
outputTokensPerSecond: 101.12,
|
|
1227
|
+
errorRate: 0,
|
|
1228
|
+
samples: 3
|
|
1229
|
+
},
|
|
1230
|
+
"moonshot/kimi-k3": {
|
|
1231
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1232
|
+
latencyMs: 24498.9,
|
|
1233
|
+
p95LatencyMs: 40365.3,
|
|
1234
|
+
outputTokensPerSecond: 25.11,
|
|
1235
|
+
errorRate: 0,
|
|
1236
|
+
samples: 3
|
|
1237
|
+
},
|
|
1238
|
+
"nvidia/mistral-nemotron": {
|
|
1239
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1240
|
+
latencyMs: 7349.6,
|
|
1241
|
+
p95LatencyMs: 9932.3,
|
|
1242
|
+
outputTokensPerSecond: 79.48,
|
|
1243
|
+
errorRate: 0.3333,
|
|
1244
|
+
samples: 3
|
|
1245
|
+
},
|
|
1246
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
|
|
1247
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1248
|
+
latencyMs: 9324.6,
|
|
1249
|
+
p95LatencyMs: 12992,
|
|
1250
|
+
outputTokensPerSecond: 64.96,
|
|
1251
|
+
errorRate: 0.3333,
|
|
1252
|
+
samples: 3
|
|
1253
|
+
},
|
|
1254
|
+
"nvidia/nemotron-nano-12b-v2-vl": {
|
|
1255
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1256
|
+
latencyMs: 5846.9,
|
|
1257
|
+
p95LatencyMs: 5846.9,
|
|
1258
|
+
outputTokensPerSecond: 87.57,
|
|
1259
|
+
errorRate: 0.6667,
|
|
1260
|
+
samples: 3
|
|
1261
|
+
},
|
|
1262
|
+
"nvidia/nemotron-nano-9b-v2": {
|
|
1263
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1264
|
+
latencyMs: 5282.5,
|
|
1265
|
+
p95LatencyMs: 5282.5,
|
|
1266
|
+
outputTokensPerSecond: 96.92,
|
|
1267
|
+
errorRate: 0.6667,
|
|
1268
|
+
samples: 3
|
|
1269
|
+
},
|
|
1270
|
+
"nvidia/step-3.7-flash": {
|
|
1271
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1272
|
+
latencyMs: 4617.4,
|
|
1273
|
+
p95LatencyMs: 5237.4,
|
|
1274
|
+
outputTokensPerSecond: 112.92,
|
|
1275
|
+
errorRate: 0.3333,
|
|
1276
|
+
samples: 3
|
|
1277
|
+
},
|
|
1278
|
+
"openai/chat-latest": {
|
|
1279
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1280
|
+
latencyMs: 3690.9,
|
|
1281
|
+
p95LatencyMs: 4344,
|
|
1282
|
+
outputTokensPerSecond: 111.85,
|
|
1283
|
+
errorRate: 0,
|
|
1284
|
+
samples: 3
|
|
1285
|
+
},
|
|
1286
|
+
"openai/gpt-4.1": {
|
|
1287
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1288
|
+
latencyMs: 3527.9,
|
|
1289
|
+
p95LatencyMs: 3831.7,
|
|
1290
|
+
outputTokensPerSecond: 141.27,
|
|
1291
|
+
errorRate: 0,
|
|
1292
|
+
samples: 3
|
|
1293
|
+
},
|
|
1294
|
+
"openai/gpt-4.1-mini": {
|
|
1295
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1296
|
+
latencyMs: 4268.2,
|
|
1297
|
+
p95LatencyMs: 5101.5,
|
|
1298
|
+
outputTokensPerSecond: 103.42,
|
|
1299
|
+
errorRate: 0,
|
|
1300
|
+
samples: 3
|
|
856
1301
|
},
|
|
857
|
-
"
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
1302
|
+
"openai/gpt-4.1-nano": {
|
|
1303
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1304
|
+
latencyMs: 3088.3,
|
|
1305
|
+
p95LatencyMs: 3369.2,
|
|
1306
|
+
outputTokensPerSecond: 150.31,
|
|
1307
|
+
errorRate: 0,
|
|
1308
|
+
samples: 3
|
|
862
1309
|
},
|
|
863
|
-
"
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
1310
|
+
"openai/gpt-4o": {
|
|
1311
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1312
|
+
latencyMs: 2995.2,
|
|
1313
|
+
p95LatencyMs: 3174.2,
|
|
1314
|
+
outputTokensPerSecond: 171.32,
|
|
1315
|
+
errorRate: 0,
|
|
1316
|
+
samples: 3
|
|
868
1317
|
},
|
|
869
|
-
"
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
}
|
|
875
|
-
});
|
|
876
|
-
var model_profiles_generated_default = {
|
|
877
|
-
"openai/gpt-5.5": {
|
|
878
|
-
measuredAt: "2026-07-21T10:21:31Z",
|
|
879
|
-
latencyMs: 6243.1,
|
|
880
|
-
p95LatencyMs: 9865,
|
|
881
|
-
outputTokensPerSecond: 12.53,
|
|
1318
|
+
"openai/gpt-4o-mini": {
|
|
1319
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1320
|
+
latencyMs: 4751.5,
|
|
1321
|
+
p95LatencyMs: 4930.4,
|
|
1322
|
+
outputTokensPerSecond: 107.84,
|
|
882
1323
|
errorRate: 0,
|
|
883
1324
|
samples: 3
|
|
884
1325
|
},
|
|
885
|
-
"openai/gpt-5
|
|
886
|
-
measuredAt: "2026-
|
|
887
|
-
latencyMs:
|
|
888
|
-
p95LatencyMs:
|
|
889
|
-
outputTokensPerSecond:
|
|
1326
|
+
"openai/gpt-5-mini": {
|
|
1327
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1328
|
+
latencyMs: 4558.1,
|
|
1329
|
+
p95LatencyMs: 5081.9,
|
|
1330
|
+
outputTokensPerSecond: 113.25,
|
|
890
1331
|
errorRate: 0,
|
|
891
1332
|
samples: 3
|
|
892
1333
|
},
|
|
893
|
-
"openai/gpt-5.
|
|
894
|
-
measuredAt: "2026-
|
|
895
|
-
latencyMs:
|
|
896
|
-
p95LatencyMs:
|
|
897
|
-
outputTokensPerSecond:
|
|
898
|
-
errorRate: 0
|
|
1334
|
+
"openai/gpt-5.2": {
|
|
1335
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1336
|
+
latencyMs: 5436.6,
|
|
1337
|
+
p95LatencyMs: 5928.8,
|
|
1338
|
+
outputTokensPerSecond: 95.47,
|
|
1339
|
+
errorRate: 0,
|
|
899
1340
|
samples: 3
|
|
900
1341
|
},
|
|
901
1342
|
"openai/gpt-5.3-codex": {
|
|
902
|
-
measuredAt: "2026-
|
|
903
|
-
latencyMs:
|
|
904
|
-
p95LatencyMs:
|
|
905
|
-
outputTokensPerSecond:
|
|
906
|
-
errorRate: 0,
|
|
1343
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1344
|
+
latencyMs: 15290.4,
|
|
1345
|
+
p95LatencyMs: 15290.4,
|
|
1346
|
+
outputTokensPerSecond: 33.49,
|
|
1347
|
+
errorRate: 0.6667,
|
|
907
1348
|
samples: 3
|
|
908
1349
|
},
|
|
909
|
-
"
|
|
910
|
-
measuredAt: "2026-
|
|
911
|
-
latencyMs:
|
|
912
|
-
p95LatencyMs:
|
|
913
|
-
outputTokensPerSecond:
|
|
1350
|
+
"openai/gpt-5.4": {
|
|
1351
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1352
|
+
latencyMs: 5596,
|
|
1353
|
+
p95LatencyMs: 5919.4,
|
|
1354
|
+
outputTokensPerSecond: 91.67,
|
|
914
1355
|
errorRate: 0,
|
|
915
1356
|
samples: 3
|
|
916
1357
|
},
|
|
917
|
-
"
|
|
918
|
-
measuredAt: "2026-
|
|
919
|
-
latencyMs:
|
|
920
|
-
p95LatencyMs:
|
|
921
|
-
outputTokensPerSecond:
|
|
1358
|
+
"openai/gpt-5.4-mini": {
|
|
1359
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1360
|
+
latencyMs: 3377.8,
|
|
1361
|
+
p95LatencyMs: 3646.8,
|
|
1362
|
+
outputTokensPerSecond: 138.08,
|
|
922
1363
|
errorRate: 0,
|
|
923
1364
|
samples: 3
|
|
924
1365
|
},
|
|
925
|
-
"
|
|
926
|
-
measuredAt: "2026-
|
|
927
|
-
latencyMs:
|
|
928
|
-
p95LatencyMs:
|
|
929
|
-
outputTokensPerSecond:
|
|
1366
|
+
"openai/gpt-5.4-nano": {
|
|
1367
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1368
|
+
latencyMs: 4040.4,
|
|
1369
|
+
p95LatencyMs: 4205.9,
|
|
1370
|
+
outputTokensPerSecond: 118.52,
|
|
930
1371
|
errorRate: 0,
|
|
931
1372
|
samples: 3
|
|
932
1373
|
},
|
|
933
|
-
"
|
|
934
|
-
measuredAt: "2026-
|
|
935
|
-
latencyMs:
|
|
936
|
-
p95LatencyMs:
|
|
937
|
-
outputTokensPerSecond:
|
|
1374
|
+
"openai/gpt-5.5": {
|
|
1375
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1376
|
+
latencyMs: 6367.8,
|
|
1377
|
+
p95LatencyMs: 7330.7,
|
|
1378
|
+
outputTokensPerSecond: 81.29,
|
|
938
1379
|
errorRate: 0,
|
|
939
1380
|
samples: 3
|
|
940
1381
|
},
|
|
941
|
-
"
|
|
942
|
-
measuredAt: "2026-
|
|
943
|
-
latencyMs:
|
|
944
|
-
p95LatencyMs:
|
|
945
|
-
outputTokensPerSecond:
|
|
1382
|
+
"openai/gpt-5.6-luna": {
|
|
1383
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1384
|
+
latencyMs: 6064.5,
|
|
1385
|
+
p95LatencyMs: 7347.3,
|
|
1386
|
+
outputTokensPerSecond: 87.93,
|
|
946
1387
|
errorRate: 0,
|
|
947
1388
|
samples: 3
|
|
948
1389
|
},
|
|
949
|
-
"
|
|
950
|
-
measuredAt: "2026-
|
|
951
|
-
latencyMs:
|
|
952
|
-
p95LatencyMs:
|
|
953
|
-
outputTokensPerSecond:
|
|
1390
|
+
"openai/gpt-5.6-luna-pro": {
|
|
1391
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1392
|
+
latencyMs: 13914.9,
|
|
1393
|
+
p95LatencyMs: 13914.9,
|
|
1394
|
+
outputTokensPerSecond: 36.79,
|
|
1395
|
+
errorRate: 0.6667,
|
|
1396
|
+
samples: 3
|
|
1397
|
+
},
|
|
1398
|
+
"openai/gpt-5.6-sol": {
|
|
1399
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1400
|
+
latencyMs: 7720.2,
|
|
1401
|
+
p95LatencyMs: 9108.2,
|
|
1402
|
+
outputTokensPerSecond: 67.47,
|
|
954
1403
|
errorRate: 0,
|
|
955
1404
|
samples: 3
|
|
956
1405
|
},
|
|
957
|
-
"
|
|
958
|
-
measuredAt: "2026-
|
|
959
|
-
latencyMs:
|
|
960
|
-
p95LatencyMs:
|
|
961
|
-
outputTokensPerSecond:
|
|
1406
|
+
"openai/gpt-5.6-sol-pro": {
|
|
1407
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1408
|
+
latencyMs: 11442.7,
|
|
1409
|
+
p95LatencyMs: 13363.1,
|
|
1410
|
+
outputTokensPerSecond: 148.75,
|
|
962
1411
|
errorRate: 0,
|
|
963
1412
|
samples: 3
|
|
964
1413
|
},
|
|
965
|
-
"
|
|
966
|
-
measuredAt: "2026-
|
|
967
|
-
latencyMs:
|
|
968
|
-
p95LatencyMs:
|
|
969
|
-
outputTokensPerSecond:
|
|
1414
|
+
"openai/gpt-5.6-terra": {
|
|
1415
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1416
|
+
latencyMs: 4941,
|
|
1417
|
+
p95LatencyMs: 5095.3,
|
|
1418
|
+
outputTokensPerSecond: 103.69,
|
|
970
1419
|
errorRate: 0,
|
|
971
1420
|
samples: 3
|
|
972
1421
|
},
|
|
973
|
-
"
|
|
974
|
-
measuredAt: "2026-
|
|
975
|
-
latencyMs:
|
|
976
|
-
p95LatencyMs:
|
|
977
|
-
outputTokensPerSecond:
|
|
1422
|
+
"openai/gpt-5.6-terra-pro": {
|
|
1423
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1424
|
+
latencyMs: 3574.1,
|
|
1425
|
+
p95LatencyMs: 4126.3,
|
|
1426
|
+
outputTokensPerSecond: 133.59,
|
|
978
1427
|
errorRate: 0,
|
|
979
1428
|
samples: 3
|
|
980
1429
|
},
|
|
981
|
-
"
|
|
982
|
-
measuredAt: "2026-
|
|
983
|
-
latencyMs:
|
|
984
|
-
p95LatencyMs:
|
|
985
|
-
outputTokensPerSecond:
|
|
1430
|
+
"openai/o1": {
|
|
1431
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1432
|
+
latencyMs: 4324.9,
|
|
1433
|
+
p95LatencyMs: 5838.1,
|
|
1434
|
+
outputTokensPerSecond: 125.86,
|
|
986
1435
|
errorRate: 0,
|
|
987
1436
|
samples: 3
|
|
988
1437
|
},
|
|
989
|
-
"
|
|
990
|
-
measuredAt: "2026-
|
|
991
|
-
latencyMs:
|
|
992
|
-
p95LatencyMs:
|
|
993
|
-
outputTokensPerSecond:
|
|
1438
|
+
"openai/o3": {
|
|
1439
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1440
|
+
latencyMs: 5463.4,
|
|
1441
|
+
p95LatencyMs: 5613.1,
|
|
1442
|
+
outputTokensPerSecond: 93.8,
|
|
994
1443
|
errorRate: 0,
|
|
995
1444
|
samples: 3
|
|
996
1445
|
},
|
|
997
|
-
"
|
|
998
|
-
measuredAt: "2026-
|
|
999
|
-
latencyMs:
|
|
1000
|
-
p95LatencyMs:
|
|
1001
|
-
outputTokensPerSecond:
|
|
1446
|
+
"openai/o3-mini": {
|
|
1447
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1448
|
+
latencyMs: 2912.7,
|
|
1449
|
+
p95LatencyMs: 3092.1,
|
|
1450
|
+
outputTokensPerSecond: 176.49,
|
|
1002
1451
|
errorRate: 0,
|
|
1003
1452
|
samples: 3
|
|
1004
1453
|
},
|
|
1005
|
-
"
|
|
1006
|
-
measuredAt: "2026-
|
|
1007
|
-
latencyMs:
|
|
1008
|
-
p95LatencyMs:
|
|
1009
|
-
outputTokensPerSecond:
|
|
1010
|
-
errorRate: 0
|
|
1454
|
+
"openai/o4-mini": {
|
|
1455
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1456
|
+
latencyMs: 4958.7,
|
|
1457
|
+
p95LatencyMs: 5313,
|
|
1458
|
+
outputTokensPerSecond: 103.81,
|
|
1459
|
+
errorRate: 0,
|
|
1011
1460
|
samples: 3
|
|
1012
1461
|
},
|
|
1013
|
-
"
|
|
1014
|
-
measuredAt: "2026-
|
|
1015
|
-
latencyMs:
|
|
1016
|
-
p95LatencyMs:
|
|
1017
|
-
outputTokensPerSecond:
|
|
1462
|
+
"qwen/qwen3.7-flash": {
|
|
1463
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1464
|
+
latencyMs: 3385.5,
|
|
1465
|
+
p95LatencyMs: 4042.7,
|
|
1466
|
+
outputTokensPerSecond: 153.94,
|
|
1018
1467
|
errorRate: 0,
|
|
1019
1468
|
samples: 3
|
|
1020
1469
|
},
|
|
1021
|
-
"
|
|
1022
|
-
measuredAt: "2026-
|
|
1023
|
-
latencyMs:
|
|
1024
|
-
p95LatencyMs:
|
|
1025
|
-
outputTokensPerSecond:
|
|
1470
|
+
"qwen/qwen3.7-max": {
|
|
1471
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1472
|
+
latencyMs: 9387.1,
|
|
1473
|
+
p95LatencyMs: 10490.2,
|
|
1474
|
+
outputTokensPerSecond: 54.92,
|
|
1026
1475
|
errorRate: 0,
|
|
1027
1476
|
samples: 3
|
|
1028
1477
|
},
|
|
1029
|
-
"
|
|
1030
|
-
measuredAt: "2026-
|
|
1031
|
-
latencyMs:
|
|
1032
|
-
p95LatencyMs:
|
|
1033
|
-
outputTokensPerSecond:
|
|
1034
|
-
errorRate: 0
|
|
1478
|
+
"qwen/qwen3.7-plus": {
|
|
1479
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1480
|
+
latencyMs: 9766.6,
|
|
1481
|
+
p95LatencyMs: 9798.2,
|
|
1482
|
+
outputTokensPerSecond: 52.42,
|
|
1483
|
+
errorRate: 0,
|
|
1035
1484
|
samples: 3
|
|
1036
1485
|
},
|
|
1037
|
-
"
|
|
1038
|
-
measuredAt: "2026-
|
|
1039
|
-
latencyMs:
|
|
1040
|
-
p95LatencyMs:
|
|
1041
|
-
outputTokensPerSecond:
|
|
1486
|
+
"tencent/hy3": {
|
|
1487
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1488
|
+
latencyMs: 6062.3,
|
|
1489
|
+
p95LatencyMs: 7070.2,
|
|
1490
|
+
outputTokensPerSecond: 87.3,
|
|
1042
1491
|
errorRate: 0,
|
|
1043
1492
|
samples: 3
|
|
1044
1493
|
},
|
|
1045
|
-
"
|
|
1046
|
-
measuredAt: "2026-
|
|
1047
|
-
latencyMs:
|
|
1048
|
-
p95LatencyMs:
|
|
1049
|
-
outputTokensPerSecond:
|
|
1494
|
+
"xai/grok-4.3": {
|
|
1495
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1496
|
+
latencyMs: 9467.7,
|
|
1497
|
+
p95LatencyMs: 10087.9,
|
|
1498
|
+
outputTokensPerSecond: 48.36,
|
|
1050
1499
|
errorRate: 0,
|
|
1051
1500
|
samples: 3
|
|
1052
1501
|
},
|
|
1053
|
-
"
|
|
1054
|
-
measuredAt: "2026-
|
|
1055
|
-
latencyMs:
|
|
1056
|
-
p95LatencyMs:
|
|
1057
|
-
outputTokensPerSecond:
|
|
1502
|
+
"xai/grok-4.5": {
|
|
1503
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1504
|
+
latencyMs: 13564.8,
|
|
1505
|
+
p95LatencyMs: 17351.9,
|
|
1506
|
+
outputTokensPerSecond: 60.71,
|
|
1058
1507
|
errorRate: 0,
|
|
1059
1508
|
samples: 3
|
|
1060
1509
|
},
|
|
1061
|
-
"
|
|
1062
|
-
measuredAt: "2026-
|
|
1063
|
-
latencyMs:
|
|
1064
|
-
p95LatencyMs:
|
|
1065
|
-
outputTokensPerSecond:
|
|
1510
|
+
"xai/grok-build-0.1": {
|
|
1511
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1512
|
+
latencyMs: 16394.8,
|
|
1513
|
+
p95LatencyMs: 18035.4,
|
|
1514
|
+
outputTokensPerSecond: 96.86,
|
|
1066
1515
|
errorRate: 0,
|
|
1067
1516
|
samples: 3
|
|
1068
1517
|
},
|
|
1069
|
-
"
|
|
1070
|
-
measuredAt: "2026-
|
|
1071
|
-
latencyMs:
|
|
1072
|
-
p95LatencyMs:
|
|
1073
|
-
outputTokensPerSecond:
|
|
1518
|
+
"xiaomi/mimo-v2.5-pro": {
|
|
1519
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1520
|
+
latencyMs: 12070.7,
|
|
1521
|
+
p95LatencyMs: 12386.8,
|
|
1522
|
+
outputTokensPerSecond: 42.44,
|
|
1074
1523
|
errorRate: 0,
|
|
1075
1524
|
samples: 3
|
|
1076
1525
|
},
|
|
1077
1526
|
"zai/glm-5": {
|
|
1078
|
-
measuredAt: "2026-
|
|
1079
|
-
latencyMs:
|
|
1080
|
-
p95LatencyMs:
|
|
1081
|
-
outputTokensPerSecond:
|
|
1527
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1528
|
+
latencyMs: 6839.7,
|
|
1529
|
+
p95LatencyMs: 7261.4,
|
|
1530
|
+
outputTokensPerSecond: 75.16,
|
|
1531
|
+
errorRate: 0,
|
|
1532
|
+
samples: 3
|
|
1533
|
+
},
|
|
1534
|
+
"zai/glm-5-turbo": {
|
|
1535
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1536
|
+
latencyMs: 55348.5,
|
|
1537
|
+
p95LatencyMs: 114086.6,
|
|
1538
|
+
outputTokensPerSecond: 14.64,
|
|
1082
1539
|
errorRate: 0,
|
|
1083
1540
|
samples: 3
|
|
1084
1541
|
},
|
|
1085
|
-
"
|
|
1086
|
-
measuredAt: "2026-
|
|
1087
|
-
latencyMs:
|
|
1088
|
-
p95LatencyMs:
|
|
1089
|
-
outputTokensPerSecond:
|
|
1542
|
+
"zai/glm-5.1": {
|
|
1543
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1544
|
+
latencyMs: 15658.4,
|
|
1545
|
+
p95LatencyMs: 17307.1,
|
|
1546
|
+
outputTokensPerSecond: 32.9,
|
|
1090
1547
|
errorRate: 0,
|
|
1091
1548
|
samples: 3
|
|
1092
1549
|
},
|
|
1093
|
-
"
|
|
1094
|
-
measuredAt: "2026-
|
|
1095
|
-
latencyMs:
|
|
1096
|
-
p95LatencyMs:
|
|
1097
|
-
outputTokensPerSecond:
|
|
1550
|
+
"zai/glm-5.2": {
|
|
1551
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1552
|
+
latencyMs: 10308.5,
|
|
1553
|
+
p95LatencyMs: 15127.6,
|
|
1554
|
+
outputTokensPerSecond: 54.87,
|
|
1098
1555
|
errorRate: 0,
|
|
1099
1556
|
samples: 3
|
|
1100
1557
|
},
|
|
1101
|
-
"
|
|
1102
|
-
measuredAt: "2026-
|
|
1103
|
-
latencyMs:
|
|
1104
|
-
p95LatencyMs:
|
|
1105
|
-
outputTokensPerSecond:
|
|
1558
|
+
"zai/glm-5.3": {
|
|
1559
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1560
|
+
latencyMs: 7272.4,
|
|
1561
|
+
p95LatencyMs: 7998.1,
|
|
1562
|
+
outputTokensPerSecond: 71.09,
|
|
1106
1563
|
errorRate: 0,
|
|
1107
1564
|
samples: 3
|
|
1108
1565
|
},
|
|
1109
|
-
"
|
|
1110
|
-
measuredAt: "2026-
|
|
1111
|
-
latencyMs:
|
|
1112
|
-
p95LatencyMs:
|
|
1113
|
-
outputTokensPerSecond:
|
|
1566
|
+
"zai/glm-5.3-flash": {
|
|
1567
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1568
|
+
latencyMs: 10545.3,
|
|
1569
|
+
p95LatencyMs: 11672.4,
|
|
1570
|
+
outputTokensPerSecond: 49.01,
|
|
1114
1571
|
errorRate: 0,
|
|
1115
1572
|
samples: 3
|
|
1116
1573
|
}
|
|
@@ -1124,11 +1581,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1124
1581
|
latencyMs: 2305,
|
|
1125
1582
|
outputTokensPerSecond: 140.6
|
|
1126
1583
|
},
|
|
1127
|
-
"anthropic/claude-opus-4.6": {
|
|
1128
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1129
|
-
latencyMs: 2139,
|
|
1130
|
-
outputTokensPerSecond: 119.7
|
|
1131
|
-
},
|
|
1132
1584
|
"anthropic/claude-sonnet-4.6": {
|
|
1133
1585
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1134
1586
|
latencyMs: 2110,
|
|
@@ -1162,11 +1614,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1162
1614
|
latencyMs: 1609,
|
|
1163
1615
|
outputTokensPerSecond: 167.2
|
|
1164
1616
|
},
|
|
1165
|
-
"moonshot/kimi-k2.5": {
|
|
1166
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1167
|
-
latencyMs: 1646,
|
|
1168
|
-
outputTokensPerSecond: 155.7
|
|
1169
|
-
},
|
|
1170
1617
|
"openai/gpt-4o-mini": {
|
|
1171
1618
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1172
1619
|
latencyMs: 2764,
|
|
@@ -1176,18 +1623,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1176
1623
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1177
1624
|
latencyMs: 7935,
|
|
1178
1625
|
outputTokensPerSecond: 32.3
|
|
1179
|
-
},
|
|
1180
|
-
"xai/grok-4-1-fast-non-reasoning": {
|
|
1181
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1182
|
-
latencyMs: 1244,
|
|
1183
|
-
outputTokensPerSecond: 205.8,
|
|
1184
|
-
intelligenceIndex: 41
|
|
1185
|
-
},
|
|
1186
|
-
"xai/grok-4-1-fast-reasoning": {
|
|
1187
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1188
|
-
latencyMs: 1454,
|
|
1189
|
-
outputTokensPerSecond: 176.2,
|
|
1190
|
-
intelligenceIndex: 41
|
|
1191
1626
|
}
|
|
1192
1627
|
});
|
|
1193
1628
|
function inferToolRequirement(prompt, _systemPrompt, toolChoice) {
|
|
@@ -1680,7 +2115,7 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1680
2115
|
match(["gpt-5.3-codex"], 1),
|
|
1681
2116
|
match(["claude-sonnet-4.6"], 0.94),
|
|
1682
2117
|
match(["glm-5.2"], 0.9),
|
|
1683
|
-
match(["
|
|
2118
|
+
match(["deepseek-v4-pro"], 0.86)
|
|
1684
2119
|
);
|
|
1685
2120
|
case "reasoning":
|
|
1686
2121
|
return Math.max(
|
|
@@ -1704,14 +2139,13 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1704
2139
|
base,
|
|
1705
2140
|
match(["gemini-3.5-flash"], 1),
|
|
1706
2141
|
match(["grok-4.5"], 0.93),
|
|
1707
|
-
match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
|
|
1708
|
-
match(["kimi-k2.7"], 0.84)
|
|
2142
|
+
match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
|
|
1709
2143
|
);
|
|
1710
2144
|
case "vision":
|
|
1711
2145
|
return Math.max(
|
|
1712
2146
|
base,
|
|
1713
2147
|
match(["gemini-3.1-pro"], 0.96),
|
|
1714
|
-
match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-
|
|
2148
|
+
match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9)
|
|
1715
2149
|
);
|
|
1716
2150
|
case "long_context":
|
|
1717
2151
|
return Math.max(
|
|
@@ -1723,17 +2157,18 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1723
2157
|
);
|
|
1724
2158
|
case "extraction": {
|
|
1725
2159
|
const kimiExtractionAffinity = language === "zh" ? 1 : 0.9;
|
|
2160
|
+
const otherExtractionAffinity = language === "zh" ? 0.88 : 0.9;
|
|
1726
2161
|
return Math.max(
|
|
1727
2162
|
base,
|
|
1728
|
-
match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"],
|
|
1729
|
-
match(["claude-sonnet-5", "claude-sonnet-4.6"],
|
|
1730
|
-
match(["kimi-k3"
|
|
2163
|
+
match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], otherExtractionAffinity),
|
|
2164
|
+
match(["claude-sonnet-5", "claude-sonnet-4.6"], otherExtractionAffinity),
|
|
2165
|
+
match(["kimi-k3"], kimiExtractionAffinity)
|
|
1731
2166
|
);
|
|
1732
2167
|
}
|
|
1733
2168
|
default:
|
|
1734
2169
|
return Math.max(
|
|
1735
2170
|
base,
|
|
1736
|
-
match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"
|
|
2171
|
+
match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)
|
|
1737
2172
|
);
|
|
1738
2173
|
}
|
|
1739
2174
|
}
|
|
@@ -1792,6 +2227,9 @@ function evidenceCandidates(task) {
|
|
|
1792
2227
|
"deepseek/deepseek-v4-pro"
|
|
1793
2228
|
];
|
|
1794
2229
|
}
|
|
2230
|
+
if (task === "extraction") {
|
|
2231
|
+
return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"];
|
|
2232
|
+
}
|
|
1795
2233
|
if (task === "reasoning_math") {
|
|
1796
2234
|
return [
|
|
1797
2235
|
"google/gemini-3.5-flash",
|
|
@@ -1848,9 +2286,12 @@ var PortfolioStrategy = class {
|
|
|
1848
2286
|
const targetTier = (features.taskType === "reasoning_mcq" || features.taskType === "reasoning_math") && (base.tier === "SIMPLE" || base.tier === "MEDIUM") ? "REASONING" : base.tier;
|
|
1849
2287
|
const tierConfig = tierConfigs[targetTier];
|
|
1850
2288
|
const configuredCandidates = tierConfig ? getFallbackChain(targetTier, tierConfigs) : [];
|
|
2289
|
+
const unavailable = new Set(options.unavailableModels ?? []);
|
|
1851
2290
|
const chain = [
|
|
1852
2291
|
.../* @__PURE__ */ new Set([...configuredCandidates, ...evidenceCandidates(features.taskType)])
|
|
1853
|
-
].filter(
|
|
2292
|
+
].filter(
|
|
2293
|
+
(model2) => typeof model2 === "string" && model2.length > 0 && !unavailable.has(model2)
|
|
2294
|
+
);
|
|
1854
2295
|
const eligible = chain.filter(
|
|
1855
2296
|
(model2) => isEligible(model2, features, maxOutputTokens, options)
|
|
1856
2297
|
);
|
|
@@ -1927,13 +2368,7 @@ var PortfolioStrategy = class {
|
|
|
1927
2368
|
...eligibleCandidates.filter(
|
|
1928
2369
|
(model2) => !scoredModels.includes(model2) && !webResearchFallbackOrder.includes(model2)
|
|
1929
2370
|
)
|
|
1930
|
-
] :
|
|
1931
|
-
...scoredModels,
|
|
1932
|
-
...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
|
|
1933
|
-
] : [
|
|
1934
|
-
...scoredModels,
|
|
1935
|
-
...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
|
|
1936
|
-
];
|
|
2371
|
+
] : [...scoredModels, ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))];
|
|
1937
2372
|
const model = ranked[0] ?? base.model;
|
|
1938
2373
|
const selectedTierConfigs = {
|
|
1939
2374
|
...tierConfigs,
|
|
@@ -1970,7 +2405,7 @@ var PortfolioStrategy = class {
|
|
|
1970
2405
|
}
|
|
1971
2406
|
};
|
|
1972
2407
|
var DEFAULT_ROUTING_CONFIG = {
|
|
1973
|
-
version: "3.
|
|
2408
|
+
version: "3.5",
|
|
1974
2409
|
strategy: "portfolio",
|
|
1975
2410
|
portfolio: {
|
|
1976
2411
|
auto: {
|
|
@@ -3024,186 +3459,249 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3024
3459
|
// Below this confidence → ambiguous (null tier)
|
|
3025
3460
|
confidenceThreshold: 0.7
|
|
3026
3461
|
},
|
|
3462
|
+
// ─── Tier chains ───
|
|
3463
|
+
//
|
|
3464
|
+
// Catalog refresh 2026-08-29 (V3.5). Every chain below names only models
|
|
3465
|
+
// the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the
|
|
3466
|
+
// gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast
|
|
3467
|
+
// and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview,
|
|
3468
|
+
// the whole `free/*` namespace — were removed everywhere, including fallback
|
|
3469
|
+
// rungs, so a routed model is always one a user can find on blockrun.ai/models.
|
|
3470
|
+
//
|
|
3471
|
+
// Primaries moved only where portfolio.ts already carries calibration
|
|
3472
|
+
// evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for
|
|
3473
|
+
// agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no
|
|
3474
|
+
// trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash,
|
|
3475
|
+
// gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs;
|
|
3476
|
+
// promotion waits for a calibration run, because version recency is not a
|
|
3477
|
+
// quality signal.
|
|
3478
|
+
//
|
|
3479
|
+
// Latency figures in comments are the 2026-08-29 gateway probe
|
|
3480
|
+
// (model-profiles.generated.json); prices are the catalog list.
|
|
3027
3481
|
// Auto (balanced) tier configs - current default smart routing
|
|
3028
|
-
// Benchmark-tuned 2026-03-16: balancing quality (retention) + latency
|
|
3029
3482
|
tiers: {
|
|
3030
3483
|
SIMPLE: {
|
|
3031
3484
|
primary: "google/gemini-2.5-flash",
|
|
3032
|
-
//
|
|
3485
|
+
// $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer
|
|
3033
3486
|
fallback: [
|
|
3034
3487
|
"google/gemini-3-flash-preview",
|
|
3035
|
-
//
|
|
3488
|
+
// $0.50/$3 — GPQA 5/6 in the 2026-07 calibration
|
|
3489
|
+
"google/gemini-3.5-flash-lite",
|
|
3490
|
+
// $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation
|
|
3036
3491
|
"deepseek/deepseek-chat",
|
|
3037
|
-
//
|
|
3038
|
-
"moonshot/kimi-k2.5",
|
|
3039
|
-
// 1,646ms, IQ 47, strong quality
|
|
3492
|
+
// $0.14/$0.28, 1M ctx
|
|
3040
3493
|
"google/gemini-3.1-flash-lite",
|
|
3041
|
-
// $0.25/$1.50, 1M
|
|
3042
|
-
"
|
|
3043
|
-
//
|
|
3494
|
+
// $0.25/$1.50, 1M ctx
|
|
3495
|
+
"openai/gpt-5.6-luna",
|
|
3496
|
+
// $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30)
|
|
3044
3497
|
"openai/gpt-5.4-nano",
|
|
3045
|
-
// $0.20/$1.25, 1M
|
|
3046
|
-
"
|
|
3047
|
-
//
|
|
3048
|
-
"
|
|
3049
|
-
//
|
|
3498
|
+
// $0.20/$1.25, 1M ctx
|
|
3499
|
+
"google/gemini-2.5-flash-lite",
|
|
3500
|
+
// $0.10/$0.40
|
|
3501
|
+
"nvidia/step-3.7-flash"
|
|
3502
|
+
// FREE backstop — NVIDIA free tier (probed 2026-08-21)
|
|
3050
3503
|
]
|
|
3051
3504
|
},
|
|
3052
3505
|
MEDIUM: {
|
|
3053
|
-
|
|
3054
|
-
//
|
|
3506
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the
|
|
3507
|
+
// calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts).
|
|
3508
|
+
primary: "google/gemini-3.5-flash",
|
|
3509
|
+
// $1.50/$9, 1M ctx, vision + tools
|
|
3055
3510
|
fallback: [
|
|
3056
|
-
"
|
|
3057
|
-
//
|
|
3058
|
-
"
|
|
3059
|
-
// $0.
|
|
3511
|
+
"google/gemini-3.6-flash",
|
|
3512
|
+
// $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration
|
|
3513
|
+
"zai/glm-5.3-flash",
|
|
3514
|
+
// $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27
|
|
3515
|
+
"openai/gpt-5.6-terra",
|
|
3516
|
+
// $2/$12, 1M ctx — GPT-5.6 balanced tier
|
|
3060
3517
|
"google/gemini-3-flash-preview",
|
|
3061
|
-
//
|
|
3518
|
+
// $0.50/$3
|
|
3062
3519
|
"deepseek/deepseek-chat",
|
|
3063
|
-
//
|
|
3520
|
+
// $0.14/$0.28
|
|
3064
3521
|
"google/gemini-2.5-flash",
|
|
3065
|
-
//
|
|
3522
|
+
// $0.30/$2.50
|
|
3523
|
+
"minimax/minimax-m3",
|
|
3524
|
+
// $0.30/$1.20, 1M ctx
|
|
3066
3525
|
"google/gemini-3.1-flash-lite",
|
|
3067
|
-
// $0.25/$1.50
|
|
3068
|
-
"
|
|
3069
|
-
//
|
|
3070
|
-
"
|
|
3071
|
-
//
|
|
3072
|
-
"xai/grok-3-mini"
|
|
3073
|
-
// 1,202ms, $0.30/$0.50
|
|
3526
|
+
// $0.25/$1.50
|
|
3527
|
+
"openai/gpt-5.6-luna",
|
|
3528
|
+
// $0.20/$1.20
|
|
3529
|
+
"google/gemini-2.5-flash-lite"
|
|
3530
|
+
// $0.10/$0.40
|
|
3074
3531
|
]
|
|
3075
3532
|
},
|
|
3076
3533
|
COMPLEX: {
|
|
3077
3534
|
primary: "google/gemini-3.1-pro",
|
|
3078
|
-
//
|
|
3535
|
+
// $2/$12 — proven long-context flagship (portfolio.ts long_context lead)
|
|
3079
3536
|
fallback: [
|
|
3080
|
-
"google/gemini-3-flash
|
|
3081
|
-
// 1
|
|
3082
|
-
"
|
|
3083
|
-
// 1
|
|
3084
|
-
"google/gemini-2.5-pro",
|
|
3085
|
-
// 1,294ms
|
|
3537
|
+
"google/gemini-3.6-flash",
|
|
3538
|
+
// $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here)
|
|
3539
|
+
"google/gemini-3.5-flash",
|
|
3540
|
+
// $1.50/$9 — calibrated
|
|
3086
3541
|
"anthropic/claude-sonnet-5",
|
|
3087
|
-
// near-Opus quality
|
|
3542
|
+
// $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated
|
|
3543
|
+
"xai/grok-4.5",
|
|
3544
|
+
// $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden)
|
|
3545
|
+
"google/gemini-2.5-pro",
|
|
3546
|
+
// $1.25/$10
|
|
3088
3547
|
"anthropic/claude-sonnet-4.6",
|
|
3089
|
-
//
|
|
3090
|
-
"deepseek/deepseek-chat",
|
|
3091
|
-
// 1,431ms, IQ 32
|
|
3092
|
-
"google/gemini-2.5-flash",
|
|
3093
|
-
// 1,238ms, IQ 20 — cheap last resort
|
|
3548
|
+
// $3/$15
|
|
3094
3549
|
"openai/gpt-5.6-terra",
|
|
3095
|
-
// GPT-5.6 balanced tier
|
|
3550
|
+
// $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202)
|
|
3096
3551
|
"openai/gpt-5.5",
|
|
3097
|
-
//
|
|
3098
|
-
"openai/gpt-5.4"
|
|
3099
|
-
//
|
|
3552
|
+
// $5/$30 — prior OpenAI flagship
|
|
3553
|
+
"openai/gpt-5.4",
|
|
3554
|
+
// $2.50/$15 — previous flagship, benchmarked
|
|
3555
|
+
"zai/glm-5.3",
|
|
3556
|
+
// $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19
|
|
3557
|
+
"moonshot/kimi-k3",
|
|
3558
|
+
// $3/$15, 1M ctx — Moonshot flagship (K2.7 successor)
|
|
3559
|
+
"deepseek/deepseek-v4-pro",
|
|
3560
|
+
// $0.435/$0.87 — strongest open-weight reasoner
|
|
3561
|
+
"deepseek/deepseek-chat",
|
|
3562
|
+
// $0.14/$0.28 — cheap last resort
|
|
3563
|
+
"google/gemini-2.5-flash"
|
|
3564
|
+
// $0.30/$2.50
|
|
3100
3565
|
]
|
|
3101
3566
|
},
|
|
3102
3567
|
REASONING: {
|
|
3103
|
-
|
|
3104
|
-
//
|
|
3568
|
+
// Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek
|
|
3569
|
+
// Reasoner is the cheapest listed reasoner at the same 1M context.
|
|
3570
|
+
primary: "deepseek/deepseek-reasoner",
|
|
3571
|
+
// $0.14/$0.28, 1M ctx
|
|
3105
3572
|
fallback: [
|
|
3106
|
-
"xai/grok-4-fast-reasoning",
|
|
3107
|
-
// 1,298ms, $0.20/$0.50
|
|
3108
|
-
"deepseek/deepseek-reasoner",
|
|
3109
|
-
// V4 Flash thinking ($0.20/$0.40, 1M ctx)
|
|
3110
3573
|
"deepseek/deepseek-v4-pro",
|
|
3111
|
-
//
|
|
3574
|
+
// $0.435/$0.87 — calibrated reasoning band 0.95
|
|
3575
|
+
"xai/grok-4.3",
|
|
3576
|
+
// $1.50/$4, 1M ctx — xAI reasoning model, vision
|
|
3577
|
+
"qwen/qwen3.7-plus",
|
|
3578
|
+
// $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed)
|
|
3579
|
+
"google/gemini-3.5-flash",
|
|
3580
|
+
// $1.50/$9 — MGSM 5/5
|
|
3112
3581
|
"openai/o4-mini",
|
|
3113
|
-
//
|
|
3582
|
+
// $1.10/$4.40
|
|
3114
3583
|
"openai/o3"
|
|
3115
|
-
// 2
|
|
3584
|
+
// $2/$8
|
|
3116
3585
|
]
|
|
3117
3586
|
}
|
|
3118
3587
|
},
|
|
3119
3588
|
// Eco tier configs - absolute cheapest (blockrun/eco)
|
|
3120
3589
|
ecoTiers: {
|
|
3121
3590
|
SIMPLE: {
|
|
3122
|
-
primary: "
|
|
3123
|
-
// FREE
|
|
3591
|
+
primary: "nvidia/step-3.7-flash",
|
|
3592
|
+
// FREE — NVIDIA free tier flagship
|
|
3124
3593
|
fallback: [
|
|
3125
|
-
"
|
|
3126
|
-
// FREE —
|
|
3127
|
-
|
|
3128
|
-
//
|
|
3129
|
-
//
|
|
3130
|
-
//
|
|
3131
|
-
"google/gemini-3.1-flash-lite",
|
|
3132
|
-
// $0.25/$1.50 — newest flash-lite
|
|
3133
|
-
"openai/gpt-5.4-nano",
|
|
3134
|
-
// $0.20/$1.25 — fast nano
|
|
3594
|
+
"nvidia/nemotron-nano-9b-v2",
|
|
3595
|
+
// FREE — compact + fast, high-volume light tasks
|
|
3596
|
+
// The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash
|
|
3597
|
+
// 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400
|
|
3598
|
+
// 2026-08-21). Each retirement retargets the two free rungs to the
|
|
3599
|
+
// current free tier; the paid rungs below never move.
|
|
3135
3600
|
"google/gemini-2.5-flash-lite",
|
|
3136
|
-
// $0.10/$0.40
|
|
3137
|
-
"
|
|
3138
|
-
// $0.
|
|
3601
|
+
// $0.10/$0.40 — cheapest paid rung
|
|
3602
|
+
"zai/glm-5.3-flash",
|
|
3603
|
+
// $0.15/$0.50, 1M ctx, vision + tools
|
|
3604
|
+
"openai/gpt-5.6-luna",
|
|
3605
|
+
// $0.20/$1.20, 1M ctx
|
|
3606
|
+
"openai/gpt-5.4-nano",
|
|
3607
|
+
// $0.20/$1.25
|
|
3608
|
+
"google/gemini-3.1-flash-lite"
|
|
3609
|
+
// $0.25/$1.50
|
|
3139
3610
|
]
|
|
3140
3611
|
},
|
|
3141
3612
|
MEDIUM: {
|
|
3142
|
-
primary: "
|
|
3143
|
-
// $0.
|
|
3613
|
+
primary: "zai/glm-5.3-flash",
|
|
3614
|
+
// $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model
|
|
3144
3615
|
fallback: [
|
|
3616
|
+
"deepseek/deepseek-chat",
|
|
3617
|
+
// $0.14/$0.28
|
|
3618
|
+
"google/gemini-3.1-flash-lite",
|
|
3619
|
+
// $0.25/$1.50
|
|
3620
|
+
"openai/gpt-5.6-luna",
|
|
3621
|
+
// $0.20/$1.20
|
|
3145
3622
|
"openai/gpt-5.4-nano",
|
|
3146
3623
|
// $0.20/$1.25
|
|
3147
3624
|
"google/gemini-2.5-flash-lite",
|
|
3148
3625
|
// $0.10/$0.40
|
|
3149
|
-
"xai/grok-4-fast-non-reasoning",
|
|
3150
3626
|
"google/gemini-2.5-flash"
|
|
3627
|
+
// $0.30/$2.50
|
|
3151
3628
|
]
|
|
3152
3629
|
},
|
|
3153
3630
|
COMPLEX: {
|
|
3154
|
-
primary: "
|
|
3155
|
-
// $0.
|
|
3631
|
+
primary: "zai/glm-5.3-flash",
|
|
3632
|
+
// $0.15/$0.50, 1M ctx
|
|
3156
3633
|
fallback: [
|
|
3157
|
-
"
|
|
3158
|
-
|
|
3159
|
-
"
|
|
3160
|
-
|
|
3634
|
+
"deepseek/deepseek-chat",
|
|
3635
|
+
// $0.14/$0.28, 1M ctx
|
|
3636
|
+
"minimax/minimax-m3",
|
|
3637
|
+
// $0.30/$1.20, 1M ctx
|
|
3638
|
+
"deepseek/deepseek-v4-pro",
|
|
3639
|
+
// $0.435/$0.87
|
|
3640
|
+
"google/gemini-3.1-flash-lite",
|
|
3641
|
+
// $0.25/$1.50
|
|
3642
|
+
"google/gemini-2.5-flash"
|
|
3643
|
+
// $0.30/$2.50
|
|
3161
3644
|
]
|
|
3162
3645
|
},
|
|
3163
3646
|
REASONING: {
|
|
3164
|
-
primary: "
|
|
3165
|
-
// $0.
|
|
3647
|
+
primary: "deepseek/deepseek-reasoner",
|
|
3648
|
+
// $0.14/$0.28, 1M ctx — cheapest listed reasoner
|
|
3166
3649
|
fallback: [
|
|
3167
|
-
"
|
|
3168
|
-
|
|
3169
|
-
|
|
3170
|
-
|
|
3171
|
-
|
|
3650
|
+
"deepseek/deepseek-v4-pro",
|
|
3651
|
+
// $0.435/$0.87
|
|
3652
|
+
"qwen/qwen3.7-plus",
|
|
3653
|
+
// $0.32/$1.28 — reasoning
|
|
3654
|
+
"minimax/minimax-m3",
|
|
3655
|
+
// $0.30/$1.20 — reasoning + coding
|
|
3656
|
+
"zai/glm-5.3-flash"
|
|
3657
|
+
// $0.15/$0.50 — reasoning tokens alongside content
|
|
3172
3658
|
]
|
|
3173
3659
|
}
|
|
3174
3660
|
},
|
|
3175
3661
|
// Premium tier configs - best quality (blockrun/premium)
|
|
3176
|
-
// codex=complex coding,
|
|
3662
|
+
// codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits
|
|
3177
3663
|
premiumTiers: {
|
|
3178
3664
|
SIMPLE: {
|
|
3179
|
-
|
|
3180
|
-
|
|
3665
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08).
|
|
3666
|
+
primary: "google/gemini-3.5-flash",
|
|
3667
|
+
// $1.50/$9, 1M ctx, vision + tools — calibrated
|
|
3181
3668
|
fallback: [
|
|
3182
|
-
"
|
|
3183
|
-
//
|
|
3184
|
-
"moonshot/kimi-k2.5",
|
|
3185
|
-
// $0.60/$3.00 - proven reliable backstop when Moonshot direct API falters
|
|
3186
|
-
"google/gemini-2.5-flash",
|
|
3187
|
-
// 60% retention, fast growth
|
|
3669
|
+
"google/gemini-3.6-flash",
|
|
3670
|
+
// $1.50/$7.50 — newest Flash
|
|
3188
3671
|
"anthropic/claude-haiku-4.5",
|
|
3189
|
-
|
|
3672
|
+
// $1/$5
|
|
3673
|
+
"zai/glm-5.3",
|
|
3674
|
+
// $1.40/$4.40, 1M ctx
|
|
3675
|
+
"google/gemini-2.5-flash",
|
|
3676
|
+
// $0.30/$2.50
|
|
3677
|
+
"google/gemini-3.5-flash-lite",
|
|
3678
|
+
// $0.30/$2.50
|
|
3190
3679
|
"deepseek/deepseek-chat"
|
|
3680
|
+
// $0.14/$0.28
|
|
3191
3681
|
]
|
|
3192
3682
|
},
|
|
3193
3683
|
MEDIUM: {
|
|
3194
3684
|
primary: "openai/gpt-5.3-codex",
|
|
3195
|
-
// $1.75/$14 - 400K context, 128K output
|
|
3685
|
+
// $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts)
|
|
3196
3686
|
fallback: [
|
|
3197
|
-
"moonshot/kimi-k2.7",
|
|
3198
|
-
// Moonshot flagship
|
|
3199
|
-
"moonshot/kimi-k2.6",
|
|
3200
|
-
"moonshot/kimi-k2.5",
|
|
3201
|
-
"google/gemini-2.5-flash",
|
|
3202
|
-
// 60% retention, good coding capability
|
|
3203
|
-
"google/gemini-2.5-pro",
|
|
3204
|
-
"xai/grok-4-0709",
|
|
3205
3687
|
"anthropic/claude-sonnet-5",
|
|
3206
|
-
|
|
3688
|
+
// $3/$15 — code_agent band 0.98
|
|
3689
|
+
"moonshot/kimi-k3",
|
|
3690
|
+
// $3/$15, 1M ctx — Moonshot flagship
|
|
3691
|
+
"zai/glm-5.3",
|
|
3692
|
+
// $1.40/$4.40 — long-horizon coding
|
|
3693
|
+
"google/gemini-3.6-flash",
|
|
3694
|
+
// $1.50/$7.50
|
|
3695
|
+
"google/gemini-3.5-flash",
|
|
3696
|
+
// $1.50/$9
|
|
3697
|
+
"google/gemini-2.5-pro",
|
|
3698
|
+
// $1.25/$10
|
|
3699
|
+
"xai/grok-4.5",
|
|
3700
|
+
// $2.50/$9
|
|
3701
|
+
"anthropic/claude-sonnet-4.6",
|
|
3702
|
+
// $3/$15
|
|
3703
|
+
"openai/gpt-5.6-terra"
|
|
3704
|
+
// $2/$12
|
|
3207
3705
|
]
|
|
3208
3706
|
},
|
|
3209
3707
|
COMPLEX: {
|
|
@@ -3213,8 +3711,8 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3213
3711
|
// Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking)
|
|
3214
3712
|
// Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is
|
|
3215
3713
|
// also prone to "high demand" 503s (correlated failure — everyone falls
|
|
3216
|
-
// back to Google at the same time). Prefer xAI
|
|
3217
|
-
// flagship → DeepSeek → NVIDIA free instead.
|
|
3714
|
+
// back to Google at the same time). Prefer in-family → xAI → Moonshot →
|
|
3715
|
+
// OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead.
|
|
3218
3716
|
fallback: [
|
|
3219
3717
|
"anthropic/claude-opus-5",
|
|
3220
3718
|
// in-family hot swap first (half the price, 1M ctx + adaptive thinking)
|
|
@@ -3222,52 +3720,54 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3222
3720
|
// in-family hot swap (identical cost to 5)
|
|
3223
3721
|
"anthropic/claude-opus-4.7",
|
|
3224
3722
|
// in-family hot swap (identical cost to 4.8)
|
|
3225
|
-
"anthropic/claude-opus-4.6",
|
|
3226
|
-
// in-family hot swap
|
|
3227
3723
|
"anthropic/claude-sonnet-5",
|
|
3228
3724
|
// Sonnet-tier drop-down, near-Opus quality
|
|
3229
3725
|
"anthropic/claude-sonnet-4.6",
|
|
3230
3726
|
"xai/grok-4.5",
|
|
3231
|
-
// xAI flagship — 503-resistant, direct-xAI SKU
|
|
3232
|
-
"
|
|
3233
|
-
// 503-resistant flagship
|
|
3234
|
-
"moonshot/kimi-k2.7",
|
|
3727
|
+
// xAI flagship — 503-resistant, direct-xAI SKU
|
|
3728
|
+
"moonshot/kimi-k3",
|
|
3235
3729
|
// Moonshot flagship, independent infra
|
|
3236
|
-
"moonshot/kimi-k2.6",
|
|
3237
|
-
"moonshot/kimi-k2.5",
|
|
3238
3730
|
"openai/gpt-5.6-terra",
|
|
3239
|
-
// GPT-5.6 balanced tier —
|
|
3731
|
+
// GPT-5.6 balanced tier — stable (Sol excluded: #202)
|
|
3240
3732
|
"openai/gpt-5.5",
|
|
3241
3733
|
// Prior OpenAI flagship — 1M+ ctx, native agent + computer use
|
|
3242
3734
|
"openai/gpt-5.4",
|
|
3243
3735
|
// Previous flagship (slow but stable, benchmarked at 6,213ms)
|
|
3244
3736
|
"openai/gpt-5.3-codex",
|
|
3737
|
+
"zai/glm-5.3",
|
|
3738
|
+
// Z.AI flagship, 1M ctx
|
|
3739
|
+
"deepseek/deepseek-v4-pro",
|
|
3740
|
+
// strongest open-weight reasoner
|
|
3245
3741
|
"deepseek/deepseek-chat",
|
|
3246
3742
|
// Cheap, reliable
|
|
3247
|
-
"
|
|
3248
|
-
// NVIDIA free ultimate backstop
|
|
3743
|
+
"nvidia/step-3.7-flash"
|
|
3744
|
+
// NVIDIA free ultimate backstop
|
|
3249
3745
|
]
|
|
3250
3746
|
},
|
|
3251
3747
|
REASONING: {
|
|
3252
|
-
|
|
3253
|
-
//
|
|
3748
|
+
// Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both,
|
|
3749
|
+
// plus Sonnet 5's tau2/BrowseComp trajectory evidence).
|
|
3750
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3751
|
+
// $3/$15, 1M ctx, adaptive thinking
|
|
3254
3752
|
fallback: [
|
|
3255
|
-
"anthropic/claude-sonnet-
|
|
3256
|
-
// in-family hot swap — same cost
|
|
3753
|
+
"anthropic/claude-sonnet-4.6",
|
|
3754
|
+
// in-family hot swap — same cost
|
|
3257
3755
|
"anthropic/claude-opus-5",
|
|
3258
3756
|
// Newest flagship Opus w/ adaptive thinking
|
|
3259
3757
|
"anthropic/claude-opus-4.8",
|
|
3260
3758
|
// Prior flagship Opus — identical cost to 5
|
|
3261
3759
|
"anthropic/claude-opus-4.7",
|
|
3262
3760
|
// Flagship Opus w/ adaptive thinking
|
|
3263
|
-
"
|
|
3264
|
-
//
|
|
3265
|
-
"
|
|
3266
|
-
//
|
|
3761
|
+
"xai/grok-4.5",
|
|
3762
|
+
// reasoning band 0.94
|
|
3763
|
+
"deepseek/deepseek-v4-pro",
|
|
3764
|
+
// reasoning band 0.95
|
|
3765
|
+
"xai/grok-4.3",
|
|
3766
|
+
// $1.50/$4 — xAI reasoning model
|
|
3267
3767
|
"openai/o4-mini",
|
|
3268
|
-
//
|
|
3768
|
+
// $1.10/$4.40
|
|
3269
3769
|
"openai/o3"
|
|
3270
|
-
// 2
|
|
3770
|
+
// $2/$8
|
|
3271
3771
|
]
|
|
3272
3772
|
}
|
|
3273
3773
|
},
|
|
@@ -3277,101 +3777,102 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3277
3777
|
primary: "openai/gpt-4o-mini",
|
|
3278
3778
|
// $0.15/$0.60 - best tool compliance at lowest cost
|
|
3279
3779
|
fallback: [
|
|
3280
|
-
"
|
|
3281
|
-
// 1
|
|
3780
|
+
"openai/gpt-5.6-luna",
|
|
3781
|
+
// $0.20/$1.20 — lightweight agentic tier of GPT-5.6
|
|
3782
|
+
"zai/glm-5.3-flash",
|
|
3783
|
+
// $0.15/$0.50 — tool calls verified live 2026-08-27
|
|
3282
3784
|
"anthropic/claude-haiku-4.5",
|
|
3283
|
-
//
|
|
3284
|
-
"
|
|
3285
|
-
//
|
|
3785
|
+
// $1/$5
|
|
3786
|
+
"google/gemini-2.5-flash"
|
|
3787
|
+
// $0.30/$2.50
|
|
3286
3788
|
]
|
|
3287
3789
|
},
|
|
3288
3790
|
MEDIUM: {
|
|
3289
|
-
|
|
3290
|
-
//
|
|
3791
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the
|
|
3792
|
+
// Terminal-Bench and tau2 trajectory evidence in portfolio.ts.
|
|
3793
|
+
primary: "openai/gpt-5-mini",
|
|
3794
|
+
// $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline
|
|
3291
3795
|
fallback: [
|
|
3292
|
-
"
|
|
3293
|
-
//
|
|
3294
|
-
"
|
|
3295
|
-
// $0.
|
|
3296
|
-
"
|
|
3297
|
-
//
|
|
3796
|
+
"google/gemini-3.5-flash",
|
|
3797
|
+
// $1.50/$9 — tool_agent band 0.88
|
|
3798
|
+
"zai/glm-5.3-flash",
|
|
3799
|
+
// $0.15/$0.50 — tools verified
|
|
3800
|
+
"openai/gpt-5.6-terra",
|
|
3801
|
+
// $2/$12
|
|
3298
3802
|
"openai/gpt-4o-mini",
|
|
3299
|
-
//
|
|
3803
|
+
// $0.15/$0.60 — reliable tool calling
|
|
3300
3804
|
"anthropic/claude-haiku-4.5",
|
|
3301
|
-
//
|
|
3302
|
-
"deepseek/deepseek-chat"
|
|
3303
|
-
//
|
|
3805
|
+
// $1/$5
|
|
3806
|
+
"deepseek/deepseek-chat",
|
|
3807
|
+
// $0.14/$0.28
|
|
3808
|
+
"moonshot/kimi-k3"
|
|
3809
|
+
// $3/$15 — tool_agent band 0.85
|
|
3304
3810
|
]
|
|
3305
3811
|
},
|
|
3306
3812
|
COMPLEX: {
|
|
3307
|
-
|
|
3308
|
-
//
|
|
3813
|
+
// Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0,
|
|
3814
|
+
// Terminal-Bench safety band lead (portfolio.ts).
|
|
3815
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3816
|
+
// $3/$15 — best agentic quality per trajectory evidence
|
|
3309
3817
|
// Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s
|
|
3310
3818
|
// correlate with Anthropic outages (everyone falls back together).
|
|
3311
3819
|
// Prefer 503-resistant providers first.
|
|
3312
3820
|
fallback: [
|
|
3313
|
-
"anthropic/claude-sonnet-
|
|
3314
|
-
// in-family hot swap — same cost
|
|
3821
|
+
"anthropic/claude-sonnet-4.6",
|
|
3822
|
+
// in-family hot swap — same cost
|
|
3315
3823
|
"anthropic/claude-opus-5",
|
|
3316
3824
|
// Newest flagship Opus — in-family hot swap
|
|
3317
3825
|
"anthropic/claude-opus-4.8",
|
|
3318
3826
|
// Prior flagship Opus — identical cost to 5
|
|
3319
3827
|
"anthropic/claude-opus-4.7",
|
|
3320
3828
|
// Flagship Opus — in-family hot swap
|
|
3321
|
-
"
|
|
3322
|
-
//
|
|
3323
|
-
"
|
|
3324
|
-
//
|
|
3325
|
-
"moonshot/kimi-k2.7",
|
|
3326
|
-
// Moonshot flagship — strong tool use, independent infra
|
|
3327
|
-
"moonshot/kimi-k2.5",
|
|
3328
|
-
// cost-stability backstop
|
|
3829
|
+
"xai/grok-4.5",
|
|
3830
|
+
// xAI flagship — strong tool use, independent infra
|
|
3831
|
+
"moonshot/kimi-k3",
|
|
3832
|
+
// Moonshot flagship — independent infra
|
|
3329
3833
|
"openai/gpt-5.6-terra",
|
|
3330
|
-
// GPT-5.6 balanced tier —
|
|
3834
|
+
// GPT-5.6 balanced tier — stable (Sol excluded: #202)
|
|
3331
3835
|
"openai/gpt-5.5",
|
|
3332
3836
|
// Prior flagship — native agent + computer use (exactly the agentic-tier use case)
|
|
3333
3837
|
"openai/gpt-5.4",
|
|
3334
|
-
// Previous flagship —
|
|
3838
|
+
// Previous flagship — reliable
|
|
3839
|
+
"openai/gpt-5.3-codex",
|
|
3840
|
+
// code_agent lead
|
|
3841
|
+
"zai/glm-5.3",
|
|
3842
|
+
// long-horizon coding
|
|
3843
|
+
"deepseek/deepseek-v4-pro",
|
|
3844
|
+
// retail high-risk 3/3
|
|
3335
3845
|
"deepseek/deepseek-chat",
|
|
3336
|
-
//
|
|
3337
|
-
"
|
|
3338
|
-
// NVIDIA free ultimate backstop
|
|
3846
|
+
// cheap, reliable
|
|
3847
|
+
"nvidia/step-3.7-flash"
|
|
3848
|
+
// NVIDIA free ultimate backstop
|
|
3339
3849
|
]
|
|
3340
3850
|
},
|
|
3341
3851
|
REASONING: {
|
|
3342
|
-
primary: "anthropic/claude-sonnet-
|
|
3343
|
-
//
|
|
3852
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3853
|
+
// $3/$15 — strong tool use + adaptive thinking
|
|
3344
3854
|
fallback: [
|
|
3345
|
-
"anthropic/claude-sonnet-
|
|
3346
|
-
// in-family hot swap — same cost
|
|
3855
|
+
"anthropic/claude-sonnet-4.6",
|
|
3856
|
+
// in-family hot swap — same cost
|
|
3347
3857
|
"anthropic/claude-opus-5",
|
|
3348
3858
|
// Newest flagship Opus w/ adaptive thinking
|
|
3349
3859
|
"anthropic/claude-opus-4.8",
|
|
3350
3860
|
// Prior flagship Opus — identical cost to 5
|
|
3351
3861
|
"anthropic/claude-opus-4.7",
|
|
3352
3862
|
// Flagship Opus w/ adaptive thinking
|
|
3353
|
-
"
|
|
3354
|
-
//
|
|
3355
|
-
"
|
|
3356
|
-
//
|
|
3863
|
+
"xai/grok-4.5",
|
|
3864
|
+
// reasoning band 0.94
|
|
3865
|
+
"deepseek/deepseek-v4-pro",
|
|
3866
|
+
// reasoning band 0.95
|
|
3357
3867
|
"deepseek/deepseek-reasoner"
|
|
3358
|
-
//
|
|
3868
|
+
// $0.14/$0.28
|
|
3359
3869
|
]
|
|
3360
3870
|
}
|
|
3361
3871
|
},
|
|
3362
|
-
// Time-windowed promotions — auto-applied when active, ignored when expired
|
|
3363
|
-
|
|
3364
|
-
|
|
3365
|
-
|
|
3366
|
-
startDate: "2026-04-01",
|
|
3367
|
-
endDate: "2026-05-01",
|
|
3368
|
-
tierOverrides: {
|
|
3369
|
-
SIMPLE: { primary: "zai/glm-5.1" }
|
|
3370
|
-
},
|
|
3371
|
-
profiles: ["auto"]
|
|
3372
|
-
// only auto profile — eco stays free, premium stays premium
|
|
3373
|
-
}
|
|
3374
|
-
],
|
|
3872
|
+
// Time-windowed promotions — auto-applied when active, ignored when expired.
|
|
3873
|
+
// The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and
|
|
3874
|
+
// has expired; the list is kept empty so the mechanism stays wired.
|
|
3875
|
+
promotions: [],
|
|
3375
3876
|
overrides: {
|
|
3376
3877
|
maxTokensForceComplex: 1e5,
|
|
3377
3878
|
structuredOutputMinTier: "MEDIUM",
|
|
@@ -3713,7 +4214,7 @@ async function createSolanaPaymentPayload(secretKey, fromAddress, recipient, amo
|
|
|
3713
4214
|
}
|
|
3714
4215
|
return null;
|
|
3715
4216
|
};
|
|
3716
|
-
let entry = await getBlockhashEntry(connection, rpcUrl, false);
|
|
4217
|
+
let entry = await getBlockhashEntry(connection, rpcUrl, options.forceFreshBlockhash ?? false);
|
|
3717
4218
|
let serializedTx = findDistinctTx(entry);
|
|
3718
4219
|
if (serializedTx === null) {
|
|
3719
4220
|
entry = await getBlockhashEntry(connection, rpcUrl, true);
|
|
@@ -3964,7 +4465,7 @@ function getCostSummary() {
|
|
|
3964
4465
|
}
|
|
3965
4466
|
|
|
3966
4467
|
// src/version.ts
|
|
3967
|
-
var SDK_VERSION = "3.13.
|
|
4468
|
+
var SDK_VERSION = "3.13.4";
|
|
3968
4469
|
var USER_AGENT = `blockrun-ts/${SDK_VERSION}`;
|
|
3969
4470
|
|
|
3970
4471
|
// src/client.ts
|
|
@@ -8618,6 +9119,68 @@ async function getOrCreateSolanaWallet() {
|
|
|
8618
9119
|
var SOLANA_API_URL = "https://sol.blockrun.ai/api";
|
|
8619
9120
|
var DEFAULT_MAX_TOKENS2 = 1024;
|
|
8620
9121
|
var DEFAULT_TIMEOUT14 = 6e4;
|
|
9122
|
+
var STALE_BLOCKHASH_RETRY_BACKOFFS_MS = [500, 2e3];
|
|
9123
|
+
var MAX_PAYMENT_FAILURE_BYTES = 64 * 1024;
|
|
9124
|
+
var SafeStaleBlockhashError = class extends PaymentError {
|
|
9125
|
+
constructor() {
|
|
9126
|
+
super("Payment verification used an expired Solana blockhash; retrying with a fresh quote.");
|
|
9127
|
+
this.name = "SafeStaleBlockhashError";
|
|
9128
|
+
}
|
|
9129
|
+
};
|
|
9130
|
+
function normalizePaymentSignal(value) {
|
|
9131
|
+
return typeof value === "string" ? value.toLowerCase().replace(/[_\-\s:]/g, "") : "";
|
|
9132
|
+
}
|
|
9133
|
+
async function readPaymentFailureBody(response) {
|
|
9134
|
+
const reader = response.body?.getReader();
|
|
9135
|
+
if (!reader) return "";
|
|
9136
|
+
const decoder = new TextDecoder();
|
|
9137
|
+
let total = 0;
|
|
9138
|
+
let text = "";
|
|
9139
|
+
try {
|
|
9140
|
+
for (; ; ) {
|
|
9141
|
+
const { done, value } = await reader.read();
|
|
9142
|
+
if (done) return text + decoder.decode();
|
|
9143
|
+
total += value.byteLength;
|
|
9144
|
+
if (total > MAX_PAYMENT_FAILURE_BYTES) {
|
|
9145
|
+
void reader.cancel();
|
|
9146
|
+
return null;
|
|
9147
|
+
}
|
|
9148
|
+
text += decoder.decode(value, { stream: true });
|
|
9149
|
+
}
|
|
9150
|
+
} catch {
|
|
9151
|
+
return null;
|
|
9152
|
+
}
|
|
9153
|
+
}
|
|
9154
|
+
async function isSafeStaleBlockhashResponse(response) {
|
|
9155
|
+
const length = Number(response.headers.get("content-length") || "0");
|
|
9156
|
+
if (Number.isFinite(length) && length > MAX_PAYMENT_FAILURE_BYTES) return false;
|
|
9157
|
+
const text = await readPaymentFailureBody(response);
|
|
9158
|
+
if (text === null) return false;
|
|
9159
|
+
let body;
|
|
9160
|
+
try {
|
|
9161
|
+
const parsed = JSON.parse(text);
|
|
9162
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
|
|
9163
|
+
body = parsed;
|
|
9164
|
+
} catch {
|
|
9165
|
+
return false;
|
|
9166
|
+
}
|
|
9167
|
+
const nested = body.error && typeof body.error === "object" && !Array.isArray(body.error) ? body.error : void 0;
|
|
9168
|
+
const errorLabel = typeof body.error === "string" ? body.error : "";
|
|
9169
|
+
const code = normalizePaymentSignal(body.code ?? nested?.code);
|
|
9170
|
+
const reason = normalizePaymentSignal(body.reason);
|
|
9171
|
+
const detail = normalizePaymentSignal(body.invalidMessage);
|
|
9172
|
+
const message = normalizePaymentSignal(nested?.message ?? body.message);
|
|
9173
|
+
const label = normalizePaymentSignal(errorLabel);
|
|
9174
|
+
if (code.includes("settlementfailed") || label.includes("settlementfailed") || message.includes("settlementfailed")) return false;
|
|
9175
|
+
const verifyPhase = code === "paymentinvalid" || label.includes("verificationfailed") || message.includes("verificationfailed");
|
|
9176
|
+
if (!verifyPhase) return false;
|
|
9177
|
+
return code === "paymentblockhashstale" || detail.includes("blockhashnotfound") || detail.includes("blockheightexceeded") || reason === "expiredsignature" || message.includes("expiredsignature");
|
|
9178
|
+
}
|
|
9179
|
+
async function waitForStaleRetry(attempt) {
|
|
9180
|
+
await new Promise(
|
|
9181
|
+
(resolve) => setTimeout(resolve, STALE_BLOCKHASH_RETRY_BACKOFFS_MS[attempt])
|
|
9182
|
+
);
|
|
9183
|
+
}
|
|
8621
9184
|
var DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc";
|
|
8622
9185
|
function resolveRpcConfig(rpcUrl, rpcHeaders) {
|
|
8623
9186
|
const env = typeof process !== "undefined" && process.env ? process.env : {};
|
|
@@ -9064,26 +9627,34 @@ var SolanaLLMClient = class {
|
|
|
9064
9627
|
}
|
|
9065
9628
|
async requestWithPayment(endpoint, body) {
|
|
9066
9629
|
const url = `${this.apiUrl}${endpoint}`;
|
|
9067
|
-
|
|
9068
|
-
|
|
9069
|
-
|
|
9070
|
-
|
|
9071
|
-
|
|
9072
|
-
|
|
9073
|
-
|
|
9074
|
-
|
|
9075
|
-
|
|
9076
|
-
|
|
9077
|
-
|
|
9078
|
-
|
|
9079
|
-
|
|
9080
|
-
|
|
9630
|
+
for (let staleRetries = 0; ; ) {
|
|
9631
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9632
|
+
method: "POST",
|
|
9633
|
+
headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
|
|
9634
|
+
body: JSON.stringify(body)
|
|
9635
|
+
});
|
|
9636
|
+
if (response.status === 402) {
|
|
9637
|
+
try {
|
|
9638
|
+
return await this.handlePaymentAndRetry(url, body, response, staleRetries > 0);
|
|
9639
|
+
} catch (error) {
|
|
9640
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9641
|
+
await waitForStaleRetry(staleRetries++);
|
|
9642
|
+
continue;
|
|
9643
|
+
}
|
|
9081
9644
|
}
|
|
9082
|
-
|
|
9645
|
+
if (!response.ok) {
|
|
9646
|
+
let errorBody;
|
|
9647
|
+
try {
|
|
9648
|
+
errorBody = await response.json();
|
|
9649
|
+
} catch {
|
|
9650
|
+
errorBody = { error: "Request failed" };
|
|
9651
|
+
}
|
|
9652
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9653
|
+
}
|
|
9654
|
+
return response.json();
|
|
9083
9655
|
}
|
|
9084
|
-
return response.json();
|
|
9085
9656
|
}
|
|
9086
|
-
async handlePaymentAndRetry(url, body, response) {
|
|
9657
|
+
async handlePaymentAndRetry(url, body, response, forceFreshBlockhash = false) {
|
|
9087
9658
|
let paymentHeader = response.headers.get("payment-required");
|
|
9088
9659
|
if (!paymentHeader) {
|
|
9089
9660
|
try {
|
|
@@ -9125,7 +9696,8 @@ var SolanaLLMClient = class {
|
|
|
9125
9696
|
extra: details.extra,
|
|
9126
9697
|
extensions,
|
|
9127
9698
|
rpcUrl: this.rpcUrl,
|
|
9128
|
-
rpcHeaders: this.rpcHeaders
|
|
9699
|
+
rpcHeaders: this.rpcHeaders,
|
|
9700
|
+
forceFreshBlockhash
|
|
9129
9701
|
}
|
|
9130
9702
|
);
|
|
9131
9703
|
const retryResponse = await this.fetchWithTimeout(url, {
|
|
@@ -9138,6 +9710,9 @@ var SolanaLLMClient = class {
|
|
|
9138
9710
|
body: JSON.stringify(body)
|
|
9139
9711
|
});
|
|
9140
9712
|
if (retryResponse.status === 402) {
|
|
9713
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9714
|
+
throw new SafeStaleBlockhashError();
|
|
9715
|
+
}
|
|
9141
9716
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9142
9717
|
}
|
|
9143
9718
|
if (!retryResponse.ok) {
|
|
@@ -9156,26 +9731,34 @@ var SolanaLLMClient = class {
|
|
|
9156
9731
|
}
|
|
9157
9732
|
async requestWithPaymentRaw(endpoint, body) {
|
|
9158
9733
|
const url = `${this.apiUrl}${endpoint}`;
|
|
9159
|
-
|
|
9160
|
-
|
|
9161
|
-
|
|
9162
|
-
|
|
9163
|
-
|
|
9164
|
-
|
|
9165
|
-
|
|
9166
|
-
|
|
9167
|
-
|
|
9168
|
-
|
|
9169
|
-
|
|
9170
|
-
|
|
9171
|
-
|
|
9172
|
-
|
|
9734
|
+
for (let staleRetries = 0; ; ) {
|
|
9735
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9736
|
+
method: "POST",
|
|
9737
|
+
headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
|
|
9738
|
+
body: JSON.stringify(body)
|
|
9739
|
+
});
|
|
9740
|
+
if (response.status === 402) {
|
|
9741
|
+
try {
|
|
9742
|
+
return await this.handlePaymentAndRetryRaw(url, body, response, staleRetries > 0);
|
|
9743
|
+
} catch (error) {
|
|
9744
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9745
|
+
await waitForStaleRetry(staleRetries++);
|
|
9746
|
+
continue;
|
|
9747
|
+
}
|
|
9173
9748
|
}
|
|
9174
|
-
|
|
9749
|
+
if (!response.ok) {
|
|
9750
|
+
let errorBody;
|
|
9751
|
+
try {
|
|
9752
|
+
errorBody = await response.json();
|
|
9753
|
+
} catch {
|
|
9754
|
+
errorBody = { error: "Request failed" };
|
|
9755
|
+
}
|
|
9756
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9757
|
+
}
|
|
9758
|
+
return response.json();
|
|
9175
9759
|
}
|
|
9176
|
-
return response.json();
|
|
9177
9760
|
}
|
|
9178
|
-
async handlePaymentAndRetryRaw(url, body, response) {
|
|
9761
|
+
async handlePaymentAndRetryRaw(url, body, response, forceFreshBlockhash = false) {
|
|
9179
9762
|
let paymentHeader = response.headers.get("payment-required");
|
|
9180
9763
|
if (!paymentHeader) {
|
|
9181
9764
|
try {
|
|
@@ -9217,7 +9800,8 @@ var SolanaLLMClient = class {
|
|
|
9217
9800
|
extra: details.extra,
|
|
9218
9801
|
extensions,
|
|
9219
9802
|
rpcUrl: this.rpcUrl,
|
|
9220
|
-
rpcHeaders: this.rpcHeaders
|
|
9803
|
+
rpcHeaders: this.rpcHeaders,
|
|
9804
|
+
forceFreshBlockhash
|
|
9221
9805
|
}
|
|
9222
9806
|
);
|
|
9223
9807
|
const retryResponse = await this.fetchWithTimeout(url, {
|
|
@@ -9230,6 +9814,9 @@ var SolanaLLMClient = class {
|
|
|
9230
9814
|
body: JSON.stringify(body)
|
|
9231
9815
|
});
|
|
9232
9816
|
if (retryResponse.status === 402) {
|
|
9817
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9818
|
+
throw new SafeStaleBlockhashError();
|
|
9819
|
+
}
|
|
9233
9820
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9234
9821
|
}
|
|
9235
9822
|
if (!retryResponse.ok) {
|
|
@@ -9249,25 +9836,33 @@ var SolanaLLMClient = class {
|
|
|
9249
9836
|
async getWithPaymentRaw(endpoint, params) {
|
|
9250
9837
|
const query = params ? "?" + new URLSearchParams(params).toString() : "";
|
|
9251
9838
|
const url = `${this.apiUrl}${endpoint}${query}`;
|
|
9252
|
-
|
|
9253
|
-
|
|
9254
|
-
|
|
9255
|
-
|
|
9256
|
-
|
|
9257
|
-
|
|
9258
|
-
|
|
9259
|
-
|
|
9260
|
-
|
|
9261
|
-
|
|
9262
|
-
|
|
9263
|
-
|
|
9264
|
-
|
|
9839
|
+
for (let staleRetries = 0; ; ) {
|
|
9840
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9841
|
+
method: "GET",
|
|
9842
|
+
headers: { "User-Agent": USER_AGENT }
|
|
9843
|
+
});
|
|
9844
|
+
if (response.status === 402) {
|
|
9845
|
+
try {
|
|
9846
|
+
return await this.handleGetPaymentAndRetryRaw(url, endpoint, params, response, staleRetries > 0);
|
|
9847
|
+
} catch (error) {
|
|
9848
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9849
|
+
await waitForStaleRetry(staleRetries++);
|
|
9850
|
+
continue;
|
|
9851
|
+
}
|
|
9265
9852
|
}
|
|
9266
|
-
|
|
9853
|
+
if (!response.ok) {
|
|
9854
|
+
let errorBody;
|
|
9855
|
+
try {
|
|
9856
|
+
errorBody = await response.json();
|
|
9857
|
+
} catch {
|
|
9858
|
+
errorBody = { error: "Request failed" };
|
|
9859
|
+
}
|
|
9860
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9861
|
+
}
|
|
9862
|
+
return response.json();
|
|
9267
9863
|
}
|
|
9268
|
-
return response.json();
|
|
9269
9864
|
}
|
|
9270
|
-
async handleGetPaymentAndRetryRaw(url, endpoint, params, response) {
|
|
9865
|
+
async handleGetPaymentAndRetryRaw(url, endpoint, params, response, forceFreshBlockhash = false) {
|
|
9271
9866
|
let paymentHeader = response.headers.get("payment-required");
|
|
9272
9867
|
if (!paymentHeader) {
|
|
9273
9868
|
try {
|
|
@@ -9309,7 +9904,8 @@ var SolanaLLMClient = class {
|
|
|
9309
9904
|
extra: details.extra,
|
|
9310
9905
|
extensions,
|
|
9311
9906
|
rpcUrl: this.rpcUrl,
|
|
9312
|
-
rpcHeaders: this.rpcHeaders
|
|
9907
|
+
rpcHeaders: this.rpcHeaders,
|
|
9908
|
+
forceFreshBlockhash
|
|
9313
9909
|
}
|
|
9314
9910
|
);
|
|
9315
9911
|
const query = params ? "?" + new URLSearchParams(params).toString() : "";
|
|
@@ -9322,6 +9918,9 @@ var SolanaLLMClient = class {
|
|
|
9322
9918
|
}
|
|
9323
9919
|
});
|
|
9324
9920
|
if (retryResponse.status === 402) {
|
|
9921
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9922
|
+
throw new SafeStaleBlockhashError();
|
|
9923
|
+
}
|
|
9325
9924
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9326
9925
|
}
|
|
9327
9926
|
if (!retryResponse.ok) {
|