@blockrun/llm 3.13.2 → 3.13.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +219 -187
- package/dist/index.cjs +1121 -521
- package/dist/index.d.cts +16 -0
- package/dist/index.d.ts +16 -0
- package/dist/index.js +1121 -521
- package/package.json +12 -10
package/dist/index.cjs
CHANGED
|
@@ -145,7 +145,7 @@ var APIError = class extends BlockrunError {
|
|
|
145
145
|
}
|
|
146
146
|
};
|
|
147
147
|
|
|
148
|
-
// node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+
|
|
148
|
+
// node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+5d911879d7f1e_fbldkf3f3jmlwtwu53ftq2mj24/node_modules/@blockrun/router-core/dist/index.js
|
|
149
149
|
function scoreTokenCount(estimatedTokens, thresholds) {
|
|
150
150
|
if (estimatedTokens < thresholds.simple) {
|
|
151
151
|
return { name: "tokenCount", score: -1, signal: `short (${estimatedTokens} tokens)` };
|
|
@@ -479,6 +479,21 @@ function applyPromotions(tierConfigs, promotions, profile, now = /* @__PURE__ */
|
|
|
479
479
|
}
|
|
480
480
|
return result;
|
|
481
481
|
}
|
|
482
|
+
function applyUnavailableModels(tierConfigs, unavailableModels) {
|
|
483
|
+
if (!unavailableModels || unavailableModels.length === 0) return tierConfigs;
|
|
484
|
+
const dead = new Set(unavailableModels);
|
|
485
|
+
let result = tierConfigs;
|
|
486
|
+
for (const tier of Object.keys(tierConfigs)) {
|
|
487
|
+
const config = tierConfigs[tier];
|
|
488
|
+
const alive = [config.primary, ...config.fallback].filter((model) => !dead.has(model));
|
|
489
|
+
if (alive.length === 0 || alive[0] === config.primary && alive.length === config.fallback.length + 1) {
|
|
490
|
+
continue;
|
|
491
|
+
}
|
|
492
|
+
if (result === tierConfigs) result = { ...tierConfigs };
|
|
493
|
+
result[tier] = { primary: alive[0], fallback: alive.slice(1) };
|
|
494
|
+
}
|
|
495
|
+
return result;
|
|
496
|
+
}
|
|
482
497
|
var RulesStrategy = class {
|
|
483
498
|
name = "rules";
|
|
484
499
|
route(prompt, systemPrompt, maxOutputTokens, options) {
|
|
@@ -530,6 +545,7 @@ ${value.slice(-(scanLimit - prefixLength))}`;
|
|
|
530
545
|
profile = useAgenticTiers ? "agentic" : "auto";
|
|
531
546
|
}
|
|
532
547
|
tierConfigs = applyPromotions(tierConfigs, config.promotions, profile, options.now);
|
|
548
|
+
tierConfigs = applyUnavailableModels(tierConfigs, options.unavailableModels);
|
|
533
549
|
const agenticScoreValue = ruleResult.agenticScore;
|
|
534
550
|
if (estimatedTokens > config.overrides.maxTokensForceComplex) {
|
|
535
551
|
const decision2 = selectModel(
|
|
@@ -603,14 +619,15 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
603
619
|
supportsVision: true
|
|
604
620
|
},
|
|
605
621
|
"anthropic/claude-haiku-4.5": {
|
|
622
|
+
// override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
|
|
606
623
|
contextWindow: 2e5,
|
|
607
|
-
maxOutputTokens:
|
|
624
|
+
maxOutputTokens: 64e3,
|
|
608
625
|
supportsTools: true,
|
|
609
626
|
supportsVision: true
|
|
610
627
|
},
|
|
611
|
-
"anthropic/claude-opus-4.
|
|
612
|
-
contextWindow:
|
|
613
|
-
maxOutputTokens:
|
|
628
|
+
"anthropic/claude-opus-4.5": {
|
|
629
|
+
contextWindow: 2e5,
|
|
630
|
+
maxOutputTokens: 64e3,
|
|
614
631
|
supportsTools: true,
|
|
615
632
|
supportsVision: true
|
|
616
633
|
},
|
|
@@ -632,12 +649,19 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
632
649
|
supportsTools: true,
|
|
633
650
|
supportsVision: true
|
|
634
651
|
},
|
|
635
|
-
"anthropic/claude-sonnet-4.
|
|
652
|
+
"anthropic/claude-sonnet-4.5": {
|
|
636
653
|
contextWindow: 2e5,
|
|
637
654
|
maxOutputTokens: 64e3,
|
|
638
655
|
supportsTools: true,
|
|
639
656
|
supportsVision: true
|
|
640
657
|
},
|
|
658
|
+
"anthropic/claude-sonnet-4.6": {
|
|
659
|
+
// override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
|
|
660
|
+
contextWindow: 1e6,
|
|
661
|
+
maxOutputTokens: 128e3,
|
|
662
|
+
supportsTools: true,
|
|
663
|
+
supportsVision: true
|
|
664
|
+
},
|
|
641
665
|
"anthropic/claude-sonnet-5": {
|
|
642
666
|
contextWindow: 1e6,
|
|
643
667
|
maxOutputTokens: 128e3,
|
|
@@ -645,14 +669,14 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
645
669
|
supportsVision: true
|
|
646
670
|
},
|
|
647
671
|
"deepseek/deepseek-chat": {
|
|
648
|
-
contextWindow:
|
|
649
|
-
maxOutputTokens:
|
|
672
|
+
contextWindow: 1048576,
|
|
673
|
+
maxOutputTokens: 65536,
|
|
650
674
|
supportsTools: true,
|
|
651
675
|
supportsVision: false
|
|
652
676
|
},
|
|
653
677
|
"deepseek/deepseek-reasoner": {
|
|
654
|
-
contextWindow:
|
|
655
|
-
maxOutputTokens:
|
|
678
|
+
contextWindow: 1048576,
|
|
679
|
+
maxOutputTokens: 65536,
|
|
656
680
|
supportsTools: true,
|
|
657
681
|
supportsVision: false
|
|
658
682
|
},
|
|
@@ -662,62 +686,38 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
662
686
|
supportsTools: true,
|
|
663
687
|
supportsVision: false
|
|
664
688
|
},
|
|
665
|
-
"free/deepseek-v4-flash": {
|
|
666
|
-
contextWindow: 1e6,
|
|
667
|
-
maxOutputTokens: 16384,
|
|
668
|
-
supportsTools: false,
|
|
669
|
-
supportsVision: false
|
|
670
|
-
},
|
|
671
|
-
"free/gpt-oss-120b": {
|
|
672
|
-
contextWindow: 128e3,
|
|
673
|
-
maxOutputTokens: 16384,
|
|
674
|
-
supportsTools: false,
|
|
675
|
-
supportsVision: false
|
|
676
|
-
},
|
|
677
|
-
"free/gpt-oss-20b": {
|
|
678
|
-
contextWindow: 128e3,
|
|
679
|
-
maxOutputTokens: 16384,
|
|
680
|
-
supportsTools: false,
|
|
681
|
-
supportsVision: false
|
|
682
|
-
},
|
|
683
|
-
"free/seed-oss-36b": {
|
|
684
|
-
contextWindow: 131072,
|
|
685
|
-
maxOutputTokens: 16384,
|
|
686
|
-
supportsTools: false,
|
|
687
|
-
supportsVision: false
|
|
688
|
-
},
|
|
689
689
|
"google/gemini-2.5-flash": {
|
|
690
|
-
contextWindow:
|
|
690
|
+
contextWindow: 1048576,
|
|
691
691
|
maxOutputTokens: 65536,
|
|
692
692
|
supportsTools: true,
|
|
693
693
|
supportsVision: true
|
|
694
694
|
},
|
|
695
695
|
"google/gemini-2.5-flash-lite": {
|
|
696
|
-
contextWindow:
|
|
696
|
+
contextWindow: 1048576,
|
|
697
697
|
maxOutputTokens: 65536,
|
|
698
698
|
supportsTools: true,
|
|
699
699
|
supportsVision: false
|
|
700
700
|
},
|
|
701
701
|
"google/gemini-2.5-pro": {
|
|
702
|
-
contextWindow:
|
|
702
|
+
contextWindow: 1048576,
|
|
703
703
|
maxOutputTokens: 65536,
|
|
704
704
|
supportsTools: true,
|
|
705
705
|
supportsVision: true
|
|
706
706
|
},
|
|
707
707
|
"google/gemini-3-flash-preview": {
|
|
708
|
-
contextWindow:
|
|
708
|
+
contextWindow: 1048576,
|
|
709
709
|
maxOutputTokens: 65536,
|
|
710
|
-
supportsTools:
|
|
710
|
+
supportsTools: true,
|
|
711
711
|
supportsVision: true
|
|
712
712
|
},
|
|
713
713
|
"google/gemini-3.1-flash-lite": {
|
|
714
|
-
contextWindow:
|
|
715
|
-
maxOutputTokens:
|
|
714
|
+
contextWindow: 1048576,
|
|
715
|
+
maxOutputTokens: 65536,
|
|
716
716
|
supportsTools: true,
|
|
717
717
|
supportsVision: false
|
|
718
718
|
},
|
|
719
719
|
"google/gemini-3.1-pro": {
|
|
720
|
-
contextWindow:
|
|
720
|
+
contextWindow: 1048576,
|
|
721
721
|
maxOutputTokens: 65536,
|
|
722
722
|
supportsTools: true,
|
|
723
723
|
supportsVision: true
|
|
@@ -728,23 +728,29 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
728
728
|
supportsTools: true,
|
|
729
729
|
supportsVision: true
|
|
730
730
|
},
|
|
731
|
-
"
|
|
732
|
-
contextWindow:
|
|
733
|
-
maxOutputTokens:
|
|
731
|
+
"google/gemini-3.5-flash-lite": {
|
|
732
|
+
contextWindow: 1048576,
|
|
733
|
+
maxOutputTokens: 65536,
|
|
734
734
|
supportsTools: true,
|
|
735
|
-
supportsVision:
|
|
735
|
+
supportsVision: false
|
|
736
736
|
},
|
|
737
|
-
"
|
|
738
|
-
contextWindow:
|
|
737
|
+
"google/gemini-3.6-flash": {
|
|
738
|
+
contextWindow: 1048576,
|
|
739
739
|
maxOutputTokens: 65536,
|
|
740
740
|
supportsTools: true,
|
|
741
741
|
supportsVision: true
|
|
742
742
|
},
|
|
743
|
-
"
|
|
744
|
-
contextWindow:
|
|
743
|
+
"minimax/minimax-m2.7": {
|
|
744
|
+
contextWindow: 204800,
|
|
745
|
+
maxOutputTokens: 16384,
|
|
746
|
+
supportsTools: true,
|
|
747
|
+
supportsVision: false
|
|
748
|
+
},
|
|
749
|
+
"minimax/minimax-m3": {
|
|
750
|
+
contextWindow: 1048576,
|
|
745
751
|
maxOutputTokens: 65536,
|
|
746
752
|
supportsTools: true,
|
|
747
|
-
supportsVision:
|
|
753
|
+
supportsVision: false
|
|
748
754
|
},
|
|
749
755
|
"moonshot/kimi-k3": {
|
|
750
756
|
contextWindow: 1048576,
|
|
@@ -752,7 +758,63 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
752
758
|
supportsTools: true,
|
|
753
759
|
supportsVision: true
|
|
754
760
|
},
|
|
761
|
+
"nvidia/mistral-nemotron": {
|
|
762
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
763
|
+
contextWindow: 131072,
|
|
764
|
+
maxOutputTokens: 16384,
|
|
765
|
+
supportsTools: false,
|
|
766
|
+
supportsVision: false
|
|
767
|
+
},
|
|
768
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
|
|
769
|
+
contextWindow: 256e3,
|
|
770
|
+
maxOutputTokens: 16384,
|
|
771
|
+
supportsTools: false,
|
|
772
|
+
supportsVision: true
|
|
773
|
+
},
|
|
774
|
+
"nvidia/nemotron-nano-12b-v2-vl": {
|
|
775
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
776
|
+
contextWindow: 131072,
|
|
777
|
+
maxOutputTokens: 16384,
|
|
778
|
+
supportsTools: false,
|
|
779
|
+
supportsVision: true
|
|
780
|
+
},
|
|
781
|
+
"nvidia/nemotron-nano-9b-v2": {
|
|
782
|
+
contextWindow: 131072,
|
|
783
|
+
maxOutputTokens: 16384,
|
|
784
|
+
supportsTools: false,
|
|
785
|
+
supportsVision: false
|
|
786
|
+
},
|
|
787
|
+
"nvidia/step-3.7-flash": {
|
|
788
|
+
contextWindow: 131072,
|
|
789
|
+
maxOutputTokens: 16384,
|
|
790
|
+
supportsTools: false,
|
|
791
|
+
supportsVision: false
|
|
792
|
+
},
|
|
793
|
+
"openai/chat-latest": {
|
|
794
|
+
contextWindow: 128e3,
|
|
795
|
+
maxOutputTokens: 128e3,
|
|
796
|
+
supportsTools: true,
|
|
797
|
+
supportsVision: true
|
|
798
|
+
},
|
|
755
799
|
"openai/gpt-4.1": {
|
|
800
|
+
contextWindow: 128e3,
|
|
801
|
+
maxOutputTokens: 32768,
|
|
802
|
+
supportsTools: true,
|
|
803
|
+
supportsVision: true
|
|
804
|
+
},
|
|
805
|
+
"openai/gpt-4.1-mini": {
|
|
806
|
+
contextWindow: 128e3,
|
|
807
|
+
maxOutputTokens: 32768,
|
|
808
|
+
supportsTools: true,
|
|
809
|
+
supportsVision: false
|
|
810
|
+
},
|
|
811
|
+
"openai/gpt-4.1-nano": {
|
|
812
|
+
contextWindow: 128e3,
|
|
813
|
+
maxOutputTokens: 32768,
|
|
814
|
+
supportsTools: true,
|
|
815
|
+
supportsVision: false
|
|
816
|
+
},
|
|
817
|
+
"openai/gpt-4o": {
|
|
756
818
|
contextWindow: 128e3,
|
|
757
819
|
maxOutputTokens: 16384,
|
|
758
820
|
supportsTools: true,
|
|
@@ -766,17 +828,44 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
766
828
|
},
|
|
767
829
|
"openai/gpt-5-mini": {
|
|
768
830
|
contextWindow: 2e5,
|
|
769
|
-
maxOutputTokens:
|
|
831
|
+
maxOutputTokens: 128e3,
|
|
770
832
|
supportsTools: true,
|
|
771
833
|
supportsVision: false
|
|
772
834
|
},
|
|
835
|
+
"openai/gpt-5.2": {
|
|
836
|
+
contextWindow: 4e5,
|
|
837
|
+
maxOutputTokens: 128e3,
|
|
838
|
+
supportsTools: true,
|
|
839
|
+
supportsVision: true
|
|
840
|
+
},
|
|
841
|
+
"openai/gpt-5.2-pro": {
|
|
842
|
+
// supportsTools: not probed — fails closed
|
|
843
|
+
contextWindow: 4e5,
|
|
844
|
+
maxOutputTokens: 128e3,
|
|
845
|
+
supportsTools: false,
|
|
846
|
+
supportsVision: true
|
|
847
|
+
},
|
|
848
|
+
"openai/gpt-5.3": {
|
|
849
|
+
// supportsTools: gateway unavailable at probe time — fails closed
|
|
850
|
+
contextWindow: 128e3,
|
|
851
|
+
maxOutputTokens: 128e3,
|
|
852
|
+
supportsTools: false,
|
|
853
|
+
supportsVision: true
|
|
854
|
+
},
|
|
773
855
|
"openai/gpt-5.3-codex": {
|
|
856
|
+
// supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe measured an incident, not the model. Codex's function calling is established by the 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot claiming the model cannot call tools.
|
|
774
857
|
contextWindow: 4e5,
|
|
775
858
|
maxOutputTokens: 128e3,
|
|
776
859
|
supportsTools: true,
|
|
777
860
|
supportsVision: false
|
|
778
861
|
},
|
|
779
862
|
"openai/gpt-5.4": {
|
|
863
|
+
contextWindow: 105e4,
|
|
864
|
+
maxOutputTokens: 128e3,
|
|
865
|
+
supportsTools: true,
|
|
866
|
+
supportsVision: true
|
|
867
|
+
},
|
|
868
|
+
"openai/gpt-5.4-mini": {
|
|
780
869
|
contextWindow: 4e5,
|
|
781
870
|
maxOutputTokens: 128e3,
|
|
782
871
|
supportsTools: true,
|
|
@@ -784,30 +873,92 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
784
873
|
},
|
|
785
874
|
"openai/gpt-5.4-nano": {
|
|
786
875
|
contextWindow: 105e4,
|
|
787
|
-
maxOutputTokens:
|
|
876
|
+
maxOutputTokens: 128e3,
|
|
788
877
|
supportsTools: true,
|
|
789
878
|
supportsVision: false
|
|
790
879
|
},
|
|
880
|
+
"openai/gpt-5.4-pro": {
|
|
881
|
+
// supportsTools: not probed — fails closed
|
|
882
|
+
contextWindow: 105e4,
|
|
883
|
+
maxOutputTokens: 128e3,
|
|
884
|
+
supportsTools: false,
|
|
885
|
+
supportsVision: true
|
|
886
|
+
},
|
|
791
887
|
"openai/gpt-5.5": {
|
|
792
888
|
contextWindow: 105e4,
|
|
793
889
|
maxOutputTokens: 128e3,
|
|
794
890
|
supportsTools: true,
|
|
795
891
|
supportsVision: true
|
|
796
892
|
},
|
|
893
|
+
"openai/gpt-5.5-pro": {
|
|
894
|
+
// supportsTools: not probed — fails closed
|
|
895
|
+
contextWindow: 105e4,
|
|
896
|
+
maxOutputTokens: 128e3,
|
|
897
|
+
supportsTools: false,
|
|
898
|
+
supportsVision: true
|
|
899
|
+
},
|
|
900
|
+
"openai/gpt-5.6-luna": {
|
|
901
|
+
contextWindow: 105e4,
|
|
902
|
+
maxOutputTokens: 128e3,
|
|
903
|
+
supportsTools: true,
|
|
904
|
+
supportsVision: true
|
|
905
|
+
},
|
|
906
|
+
"openai/gpt-5.6-luna-pro": {
|
|
907
|
+
contextWindow: 105e4,
|
|
908
|
+
maxOutputTokens: 128e3,
|
|
909
|
+
supportsTools: false,
|
|
910
|
+
supportsVision: true
|
|
911
|
+
},
|
|
912
|
+
"openai/gpt-5.6-sol": {
|
|
913
|
+
contextWindow: 105e4,
|
|
914
|
+
maxOutputTokens: 128e3,
|
|
915
|
+
supportsTools: true,
|
|
916
|
+
supportsVision: true
|
|
917
|
+
},
|
|
918
|
+
"openai/gpt-5.6-sol-pro": {
|
|
919
|
+
contextWindow: 105e4,
|
|
920
|
+
maxOutputTokens: 128e3,
|
|
921
|
+
supportsTools: true,
|
|
922
|
+
supportsVision: true
|
|
923
|
+
},
|
|
797
924
|
"openai/gpt-5.6-terra": {
|
|
798
925
|
contextWindow: 105e4,
|
|
799
926
|
maxOutputTokens: 128e3,
|
|
800
927
|
supportsTools: true,
|
|
801
928
|
supportsVision: true
|
|
802
929
|
},
|
|
930
|
+
"openai/gpt-5.6-terra-pro": {
|
|
931
|
+
contextWindow: 105e4,
|
|
932
|
+
maxOutputTokens: 128e3,
|
|
933
|
+
supportsTools: true,
|
|
934
|
+
supportsVision: true
|
|
935
|
+
},
|
|
936
|
+
"openai/o1": {
|
|
937
|
+
contextWindow: 2e5,
|
|
938
|
+
maxOutputTokens: 1e5,
|
|
939
|
+
supportsTools: true,
|
|
940
|
+
supportsVision: false
|
|
941
|
+
},
|
|
803
942
|
"openai/o3": {
|
|
804
943
|
contextWindow: 2e5,
|
|
805
944
|
maxOutputTokens: 1e5,
|
|
806
945
|
supportsTools: true,
|
|
807
946
|
supportsVision: false
|
|
808
947
|
},
|
|
948
|
+
"openai/o3-mini": {
|
|
949
|
+
contextWindow: 128e3,
|
|
950
|
+
maxOutputTokens: 1e5,
|
|
951
|
+
supportsTools: true,
|
|
952
|
+
supportsVision: false
|
|
953
|
+
},
|
|
809
954
|
"openai/o4-mini": {
|
|
810
955
|
contextWindow: 128e3,
|
|
956
|
+
maxOutputTokens: 1e5,
|
|
957
|
+
supportsTools: true,
|
|
958
|
+
supportsVision: false
|
|
959
|
+
},
|
|
960
|
+
"qwen/qwen3.7-flash": {
|
|
961
|
+
contextWindow: 1e6,
|
|
811
962
|
maxOutputTokens: 65536,
|
|
812
963
|
supportsTools: true,
|
|
813
964
|
supportsVision: false
|
|
@@ -818,299 +969,605 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
|
|
|
818
969
|
supportsTools: true,
|
|
819
970
|
supportsVision: false
|
|
820
971
|
},
|
|
821
|
-
"
|
|
822
|
-
contextWindow:
|
|
823
|
-
maxOutputTokens:
|
|
972
|
+
"qwen/qwen3.7-plus": {
|
|
973
|
+
contextWindow: 1e6,
|
|
974
|
+
maxOutputTokens: 131072,
|
|
824
975
|
supportsTools: true,
|
|
825
976
|
supportsVision: false
|
|
826
977
|
},
|
|
827
|
-
"
|
|
828
|
-
contextWindow:
|
|
829
|
-
maxOutputTokens:
|
|
978
|
+
"tencent/hy3": {
|
|
979
|
+
contextWindow: 262144,
|
|
980
|
+
maxOutputTokens: 128e3,
|
|
830
981
|
supportsTools: true,
|
|
831
982
|
supportsVision: false
|
|
832
983
|
},
|
|
833
|
-
"xai/grok-4
|
|
834
|
-
contextWindow:
|
|
984
|
+
"xai/grok-4.3": {
|
|
985
|
+
contextWindow: 1e6,
|
|
835
986
|
maxOutputTokens: 16384,
|
|
836
987
|
supportsTools: true,
|
|
837
|
-
supportsVision:
|
|
988
|
+
supportsVision: true
|
|
838
989
|
},
|
|
839
|
-
"xai/grok-4
|
|
840
|
-
contextWindow:
|
|
990
|
+
"xai/grok-4.5": {
|
|
991
|
+
contextWindow: 5e5,
|
|
841
992
|
maxOutputTokens: 16384,
|
|
842
993
|
supportsTools: true,
|
|
843
|
-
supportsVision:
|
|
994
|
+
supportsVision: true
|
|
844
995
|
},
|
|
845
|
-
"xai/grok-
|
|
846
|
-
contextWindow:
|
|
996
|
+
"xai/grok-build-0.1": {
|
|
997
|
+
contextWindow: 256e3,
|
|
847
998
|
maxOutputTokens: 16384,
|
|
848
999
|
supportsTools: true,
|
|
849
1000
|
supportsVision: false
|
|
850
1001
|
},
|
|
851
|
-
"
|
|
852
|
-
contextWindow:
|
|
853
|
-
maxOutputTokens:
|
|
854
|
-
supportsTools: true,
|
|
855
|
-
supportsVision: false
|
|
1002
|
+
"xiaomi/mimo-v2.5-pro": {
|
|
1003
|
+
contextWindow: 1048576,
|
|
1004
|
+
maxOutputTokens: 131072,
|
|
1005
|
+
supportsTools: true,
|
|
1006
|
+
supportsVision: false
|
|
1007
|
+
},
|
|
1008
|
+
"zai/glm-5": {
|
|
1009
|
+
contextWindow: 2e5,
|
|
1010
|
+
maxOutputTokens: 128e3,
|
|
1011
|
+
supportsTools: true,
|
|
1012
|
+
supportsVision: false
|
|
1013
|
+
},
|
|
1014
|
+
"zai/glm-5-turbo": {
|
|
1015
|
+
contextWindow: 2e5,
|
|
1016
|
+
maxOutputTokens: 128e3,
|
|
1017
|
+
supportsTools: true,
|
|
1018
|
+
supportsVision: false
|
|
1019
|
+
},
|
|
1020
|
+
"zai/glm-5.1": {
|
|
1021
|
+
contextWindow: 2e5,
|
|
1022
|
+
maxOutputTokens: 128e3,
|
|
1023
|
+
supportsTools: true,
|
|
1024
|
+
supportsVision: false
|
|
1025
|
+
},
|
|
1026
|
+
"zai/glm-5.2": {
|
|
1027
|
+
contextWindow: 1e6,
|
|
1028
|
+
maxOutputTokens: 131072,
|
|
1029
|
+
supportsTools: true,
|
|
1030
|
+
supportsVision: false
|
|
1031
|
+
},
|
|
1032
|
+
"zai/glm-5.3": {
|
|
1033
|
+
contextWindow: 1e6,
|
|
1034
|
+
maxOutputTokens: 131072,
|
|
1035
|
+
supportsTools: true,
|
|
1036
|
+
supportsVision: false
|
|
1037
|
+
},
|
|
1038
|
+
"zai/glm-5.3-flash": {
|
|
1039
|
+
contextWindow: 1e6,
|
|
1040
|
+
maxOutputTokens: 131072,
|
|
1041
|
+
supportsTools: true,
|
|
1042
|
+
supportsVision: true
|
|
1043
|
+
}
|
|
1044
|
+
});
|
|
1045
|
+
var model_profiles_generated_default = {
|
|
1046
|
+
"anthropic/claude-fable-5": {
|
|
1047
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1048
|
+
latencyMs: 9298.5,
|
|
1049
|
+
p95LatencyMs: 9873.4,
|
|
1050
|
+
outputTokensPerSecond: 55.17,
|
|
1051
|
+
errorRate: 0,
|
|
1052
|
+
samples: 3
|
|
1053
|
+
},
|
|
1054
|
+
"anthropic/claude-haiku-4.5": {
|
|
1055
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1056
|
+
latencyMs: 3157.4,
|
|
1057
|
+
p95LatencyMs: 3170.7,
|
|
1058
|
+
outputTokensPerSecond: 162.16,
|
|
1059
|
+
errorRate: 0,
|
|
1060
|
+
samples: 3
|
|
1061
|
+
},
|
|
1062
|
+
"anthropic/claude-opus-4.5": {
|
|
1063
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1064
|
+
latencyMs: 6497.7,
|
|
1065
|
+
p95LatencyMs: 6953.7,
|
|
1066
|
+
outputTokensPerSecond: 78.99,
|
|
1067
|
+
errorRate: 0,
|
|
1068
|
+
samples: 3
|
|
1069
|
+
},
|
|
1070
|
+
"anthropic/claude-opus-4.7": {
|
|
1071
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1072
|
+
latencyMs: 5316.5,
|
|
1073
|
+
p95LatencyMs: 6121.5,
|
|
1074
|
+
outputTokensPerSecond: 97.34,
|
|
1075
|
+
errorRate: 0,
|
|
1076
|
+
samples: 3
|
|
1077
|
+
},
|
|
1078
|
+
"anthropic/claude-opus-4.8": {
|
|
1079
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1080
|
+
latencyMs: 6216.1,
|
|
1081
|
+
p95LatencyMs: 6847.7,
|
|
1082
|
+
outputTokensPerSecond: 82.81,
|
|
1083
|
+
errorRate: 0,
|
|
1084
|
+
samples: 3
|
|
1085
|
+
},
|
|
1086
|
+
"anthropic/claude-opus-5": {
|
|
1087
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1088
|
+
latencyMs: 7309,
|
|
1089
|
+
p95LatencyMs: 7745.2,
|
|
1090
|
+
outputTokensPerSecond: 70.17,
|
|
1091
|
+
errorRate: 0,
|
|
1092
|
+
samples: 3
|
|
1093
|
+
},
|
|
1094
|
+
"anthropic/claude-sonnet-4.5": {
|
|
1095
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1096
|
+
latencyMs: 6330.4,
|
|
1097
|
+
p95LatencyMs: 6631.6,
|
|
1098
|
+
outputTokensPerSecond: 81.03,
|
|
1099
|
+
errorRate: 0,
|
|
1100
|
+
samples: 3
|
|
1101
|
+
},
|
|
1102
|
+
"anthropic/claude-sonnet-4.6": {
|
|
1103
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1104
|
+
latencyMs: 6508,
|
|
1105
|
+
p95LatencyMs: 6698.3,
|
|
1106
|
+
outputTokensPerSecond: 78.6,
|
|
1107
|
+
errorRate: 0,
|
|
1108
|
+
samples: 3
|
|
1109
|
+
},
|
|
1110
|
+
"anthropic/claude-sonnet-5": {
|
|
1111
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1112
|
+
latencyMs: 6165.4,
|
|
1113
|
+
p95LatencyMs: 6582.9,
|
|
1114
|
+
outputTokensPerSecond: 83.62,
|
|
1115
|
+
errorRate: 0,
|
|
1116
|
+
samples: 3
|
|
1117
|
+
},
|
|
1118
|
+
"deepseek/deepseek-chat": {
|
|
1119
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1120
|
+
latencyMs: 4351.4,
|
|
1121
|
+
p95LatencyMs: 4543.7,
|
|
1122
|
+
outputTokensPerSecond: 117.78,
|
|
1123
|
+
errorRate: 0,
|
|
1124
|
+
samples: 3
|
|
1125
|
+
},
|
|
1126
|
+
"deepseek/deepseek-reasoner": {
|
|
1127
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1128
|
+
latencyMs: 5201.2,
|
|
1129
|
+
p95LatencyMs: 6079.6,
|
|
1130
|
+
outputTokensPerSecond: 99.77,
|
|
1131
|
+
errorRate: 0,
|
|
1132
|
+
samples: 3
|
|
1133
|
+
},
|
|
1134
|
+
"deepseek/deepseek-v4-pro": {
|
|
1135
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1136
|
+
latencyMs: 8781.2,
|
|
1137
|
+
p95LatencyMs: 9881.1,
|
|
1138
|
+
outputTokensPerSecond: 58.98,
|
|
1139
|
+
errorRate: 0,
|
|
1140
|
+
samples: 3
|
|
1141
|
+
},
|
|
1142
|
+
"google/gemini-2.5-flash": {
|
|
1143
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1144
|
+
latencyMs: 5416.4,
|
|
1145
|
+
p95LatencyMs: 6442.8,
|
|
1146
|
+
outputTokensPerSecond: 213.07,
|
|
1147
|
+
errorRate: 0,
|
|
1148
|
+
samples: 3
|
|
1149
|
+
},
|
|
1150
|
+
"google/gemini-2.5-flash-lite": {
|
|
1151
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1152
|
+
latencyMs: 5002.6,
|
|
1153
|
+
p95LatencyMs: 5780.3,
|
|
1154
|
+
outputTokensPerSecond: 408.43,
|
|
1155
|
+
errorRate: 0,
|
|
1156
|
+
samples: 3
|
|
1157
|
+
},
|
|
1158
|
+
"google/gemini-2.5-pro": {
|
|
1159
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1160
|
+
latencyMs: 28169.5,
|
|
1161
|
+
p95LatencyMs: 29491.4,
|
|
1162
|
+
outputTokensPerSecond: 147.3,
|
|
1163
|
+
errorRate: 0,
|
|
1164
|
+
samples: 3
|
|
1165
|
+
},
|
|
1166
|
+
"google/gemini-3-flash-preview": {
|
|
1167
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1168
|
+
latencyMs: 4717.1,
|
|
1169
|
+
p95LatencyMs: 5037.1,
|
|
1170
|
+
outputTokensPerSecond: 198.71,
|
|
1171
|
+
errorRate: 0,
|
|
1172
|
+
samples: 3
|
|
1173
|
+
},
|
|
1174
|
+
"google/gemini-3.1-flash-lite": {
|
|
1175
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1176
|
+
latencyMs: 2855.8,
|
|
1177
|
+
p95LatencyMs: 3172.7,
|
|
1178
|
+
outputTokensPerSecond: 286.91,
|
|
1179
|
+
errorRate: 0,
|
|
1180
|
+
samples: 3
|
|
1181
|
+
},
|
|
1182
|
+
"google/gemini-3.1-pro": {
|
|
1183
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1184
|
+
latencyMs: 24194.1,
|
|
1185
|
+
p95LatencyMs: 27269.6,
|
|
1186
|
+
outputTokensPerSecond: 109.47,
|
|
1187
|
+
errorRate: 0,
|
|
1188
|
+
samples: 3
|
|
1189
|
+
},
|
|
1190
|
+
"google/gemini-3.5-flash": {
|
|
1191
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1192
|
+
latencyMs: 5320.6,
|
|
1193
|
+
p95LatencyMs: 5429.8,
|
|
1194
|
+
outputTokensPerSecond: 226.21,
|
|
1195
|
+
errorRate: 0,
|
|
1196
|
+
samples: 3
|
|
1197
|
+
},
|
|
1198
|
+
"google/gemini-3.5-flash-lite": {
|
|
1199
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1200
|
+
latencyMs: 3515.8,
|
|
1201
|
+
p95LatencyMs: 4363.4,
|
|
1202
|
+
outputTokensPerSecond: 248.9,
|
|
1203
|
+
errorRate: 0,
|
|
1204
|
+
samples: 3
|
|
1205
|
+
},
|
|
1206
|
+
"google/gemini-3.6-flash": {
|
|
1207
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1208
|
+
latencyMs: 13020,
|
|
1209
|
+
p95LatencyMs: 15383.1,
|
|
1210
|
+
outputTokensPerSecond: 187.87,
|
|
1211
|
+
errorRate: 0,
|
|
1212
|
+
samples: 3
|
|
1213
|
+
},
|
|
1214
|
+
"minimax/minimax-m2.7": {
|
|
1215
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1216
|
+
latencyMs: 8761.1,
|
|
1217
|
+
p95LatencyMs: 10199.3,
|
|
1218
|
+
outputTokensPerSecond: 59.18,
|
|
1219
|
+
errorRate: 0,
|
|
1220
|
+
samples: 3
|
|
1221
|
+
},
|
|
1222
|
+
"minimax/minimax-m3": {
|
|
1223
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1224
|
+
latencyMs: 11101.9,
|
|
1225
|
+
p95LatencyMs: 26087.1,
|
|
1226
|
+
outputTokensPerSecond: 101.12,
|
|
1227
|
+
errorRate: 0,
|
|
1228
|
+
samples: 3
|
|
1229
|
+
},
|
|
1230
|
+
"moonshot/kimi-k3": {
|
|
1231
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1232
|
+
latencyMs: 24498.9,
|
|
1233
|
+
p95LatencyMs: 40365.3,
|
|
1234
|
+
outputTokensPerSecond: 25.11,
|
|
1235
|
+
errorRate: 0,
|
|
1236
|
+
samples: 3
|
|
1237
|
+
},
|
|
1238
|
+
"nvidia/mistral-nemotron": {
|
|
1239
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1240
|
+
latencyMs: 7349.6,
|
|
1241
|
+
p95LatencyMs: 9932.3,
|
|
1242
|
+
outputTokensPerSecond: 79.48,
|
|
1243
|
+
errorRate: 0.3333,
|
|
1244
|
+
samples: 3
|
|
1245
|
+
},
|
|
1246
|
+
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
|
|
1247
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1248
|
+
latencyMs: 9324.6,
|
|
1249
|
+
p95LatencyMs: 12992,
|
|
1250
|
+
outputTokensPerSecond: 64.96,
|
|
1251
|
+
errorRate: 0.3333,
|
|
1252
|
+
samples: 3
|
|
1253
|
+
},
|
|
1254
|
+
"nvidia/nemotron-nano-12b-v2-vl": {
|
|
1255
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1256
|
+
latencyMs: 5846.9,
|
|
1257
|
+
p95LatencyMs: 5846.9,
|
|
1258
|
+
outputTokensPerSecond: 87.57,
|
|
1259
|
+
errorRate: 0.6667,
|
|
1260
|
+
samples: 3
|
|
1261
|
+
},
|
|
1262
|
+
"nvidia/nemotron-nano-9b-v2": {
|
|
1263
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1264
|
+
latencyMs: 5282.5,
|
|
1265
|
+
p95LatencyMs: 5282.5,
|
|
1266
|
+
outputTokensPerSecond: 96.92,
|
|
1267
|
+
errorRate: 0.6667,
|
|
1268
|
+
samples: 3
|
|
1269
|
+
},
|
|
1270
|
+
"nvidia/step-3.7-flash": {
|
|
1271
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1272
|
+
latencyMs: 4617.4,
|
|
1273
|
+
p95LatencyMs: 5237.4,
|
|
1274
|
+
outputTokensPerSecond: 112.92,
|
|
1275
|
+
errorRate: 0.3333,
|
|
1276
|
+
samples: 3
|
|
1277
|
+
},
|
|
1278
|
+
"openai/chat-latest": {
|
|
1279
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1280
|
+
latencyMs: 3690.9,
|
|
1281
|
+
p95LatencyMs: 4344,
|
|
1282
|
+
outputTokensPerSecond: 111.85,
|
|
1283
|
+
errorRate: 0,
|
|
1284
|
+
samples: 3
|
|
1285
|
+
},
|
|
1286
|
+
"openai/gpt-4.1": {
|
|
1287
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1288
|
+
latencyMs: 3527.9,
|
|
1289
|
+
p95LatencyMs: 3831.7,
|
|
1290
|
+
outputTokensPerSecond: 141.27,
|
|
1291
|
+
errorRate: 0,
|
|
1292
|
+
samples: 3
|
|
1293
|
+
},
|
|
1294
|
+
"openai/gpt-4.1-mini": {
|
|
1295
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1296
|
+
latencyMs: 4268.2,
|
|
1297
|
+
p95LatencyMs: 5101.5,
|
|
1298
|
+
outputTokensPerSecond: 103.42,
|
|
1299
|
+
errorRate: 0,
|
|
1300
|
+
samples: 3
|
|
856
1301
|
},
|
|
857
|
-
"
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
1302
|
+
"openai/gpt-4.1-nano": {
|
|
1303
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1304
|
+
latencyMs: 3088.3,
|
|
1305
|
+
p95LatencyMs: 3369.2,
|
|
1306
|
+
outputTokensPerSecond: 150.31,
|
|
1307
|
+
errorRate: 0,
|
|
1308
|
+
samples: 3
|
|
862
1309
|
},
|
|
863
|
-
"
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
1310
|
+
"openai/gpt-4o": {
|
|
1311
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1312
|
+
latencyMs: 2995.2,
|
|
1313
|
+
p95LatencyMs: 3174.2,
|
|
1314
|
+
outputTokensPerSecond: 171.32,
|
|
1315
|
+
errorRate: 0,
|
|
1316
|
+
samples: 3
|
|
868
1317
|
},
|
|
869
|
-
"
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
}
|
|
875
|
-
});
|
|
876
|
-
var model_profiles_generated_default = {
|
|
877
|
-
"openai/gpt-5.5": {
|
|
878
|
-
measuredAt: "2026-07-21T10:21:31Z",
|
|
879
|
-
latencyMs: 6243.1,
|
|
880
|
-
p95LatencyMs: 9865,
|
|
881
|
-
outputTokensPerSecond: 12.53,
|
|
1318
|
+
"openai/gpt-4o-mini": {
|
|
1319
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1320
|
+
latencyMs: 4751.5,
|
|
1321
|
+
p95LatencyMs: 4930.4,
|
|
1322
|
+
outputTokensPerSecond: 107.84,
|
|
882
1323
|
errorRate: 0,
|
|
883
1324
|
samples: 3
|
|
884
1325
|
},
|
|
885
|
-
"openai/gpt-5
|
|
886
|
-
measuredAt: "2026-
|
|
887
|
-
latencyMs:
|
|
888
|
-
p95LatencyMs:
|
|
889
|
-
outputTokensPerSecond:
|
|
1326
|
+
"openai/gpt-5-mini": {
|
|
1327
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1328
|
+
latencyMs: 4558.1,
|
|
1329
|
+
p95LatencyMs: 5081.9,
|
|
1330
|
+
outputTokensPerSecond: 113.25,
|
|
890
1331
|
errorRate: 0,
|
|
891
1332
|
samples: 3
|
|
892
1333
|
},
|
|
893
|
-
"openai/gpt-5.
|
|
894
|
-
measuredAt: "2026-
|
|
895
|
-
latencyMs:
|
|
896
|
-
p95LatencyMs:
|
|
897
|
-
outputTokensPerSecond:
|
|
898
|
-
errorRate: 0
|
|
1334
|
+
"openai/gpt-5.2": {
|
|
1335
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1336
|
+
latencyMs: 5436.6,
|
|
1337
|
+
p95LatencyMs: 5928.8,
|
|
1338
|
+
outputTokensPerSecond: 95.47,
|
|
1339
|
+
errorRate: 0,
|
|
899
1340
|
samples: 3
|
|
900
1341
|
},
|
|
901
1342
|
"openai/gpt-5.3-codex": {
|
|
902
|
-
measuredAt: "2026-
|
|
903
|
-
latencyMs:
|
|
904
|
-
p95LatencyMs:
|
|
905
|
-
outputTokensPerSecond:
|
|
906
|
-
errorRate: 0,
|
|
1343
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1344
|
+
latencyMs: 15290.4,
|
|
1345
|
+
p95LatencyMs: 15290.4,
|
|
1346
|
+
outputTokensPerSecond: 33.49,
|
|
1347
|
+
errorRate: 0.6667,
|
|
907
1348
|
samples: 3
|
|
908
1349
|
},
|
|
909
|
-
"
|
|
910
|
-
measuredAt: "2026-
|
|
911
|
-
latencyMs:
|
|
912
|
-
p95LatencyMs:
|
|
913
|
-
outputTokensPerSecond:
|
|
1350
|
+
"openai/gpt-5.4": {
|
|
1351
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1352
|
+
latencyMs: 5596,
|
|
1353
|
+
p95LatencyMs: 5919.4,
|
|
1354
|
+
outputTokensPerSecond: 91.67,
|
|
914
1355
|
errorRate: 0,
|
|
915
1356
|
samples: 3
|
|
916
1357
|
},
|
|
917
|
-
"
|
|
918
|
-
measuredAt: "2026-
|
|
919
|
-
latencyMs:
|
|
920
|
-
p95LatencyMs:
|
|
921
|
-
outputTokensPerSecond:
|
|
1358
|
+
"openai/gpt-5.4-mini": {
|
|
1359
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1360
|
+
latencyMs: 3377.8,
|
|
1361
|
+
p95LatencyMs: 3646.8,
|
|
1362
|
+
outputTokensPerSecond: 138.08,
|
|
922
1363
|
errorRate: 0,
|
|
923
1364
|
samples: 3
|
|
924
1365
|
},
|
|
925
|
-
"
|
|
926
|
-
measuredAt: "2026-
|
|
927
|
-
latencyMs:
|
|
928
|
-
p95LatencyMs:
|
|
929
|
-
outputTokensPerSecond:
|
|
1366
|
+
"openai/gpt-5.4-nano": {
|
|
1367
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1368
|
+
latencyMs: 4040.4,
|
|
1369
|
+
p95LatencyMs: 4205.9,
|
|
1370
|
+
outputTokensPerSecond: 118.52,
|
|
930
1371
|
errorRate: 0,
|
|
931
1372
|
samples: 3
|
|
932
1373
|
},
|
|
933
|
-
"
|
|
934
|
-
measuredAt: "2026-
|
|
935
|
-
latencyMs:
|
|
936
|
-
p95LatencyMs:
|
|
937
|
-
outputTokensPerSecond:
|
|
1374
|
+
"openai/gpt-5.5": {
|
|
1375
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1376
|
+
latencyMs: 6367.8,
|
|
1377
|
+
p95LatencyMs: 7330.7,
|
|
1378
|
+
outputTokensPerSecond: 81.29,
|
|
938
1379
|
errorRate: 0,
|
|
939
1380
|
samples: 3
|
|
940
1381
|
},
|
|
941
|
-
"
|
|
942
|
-
measuredAt: "2026-
|
|
943
|
-
latencyMs:
|
|
944
|
-
p95LatencyMs:
|
|
945
|
-
outputTokensPerSecond:
|
|
1382
|
+
"openai/gpt-5.6-luna": {
|
|
1383
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1384
|
+
latencyMs: 6064.5,
|
|
1385
|
+
p95LatencyMs: 7347.3,
|
|
1386
|
+
outputTokensPerSecond: 87.93,
|
|
946
1387
|
errorRate: 0,
|
|
947
1388
|
samples: 3
|
|
948
1389
|
},
|
|
949
|
-
"
|
|
950
|
-
measuredAt: "2026-
|
|
951
|
-
latencyMs:
|
|
952
|
-
p95LatencyMs:
|
|
953
|
-
outputTokensPerSecond:
|
|
1390
|
+
"openai/gpt-5.6-luna-pro": {
|
|
1391
|
+
measuredAt: "2026-08-29T16:59:54Z",
|
|
1392
|
+
latencyMs: 13914.9,
|
|
1393
|
+
p95LatencyMs: 13914.9,
|
|
1394
|
+
outputTokensPerSecond: 36.79,
|
|
1395
|
+
errorRate: 0.6667,
|
|
1396
|
+
samples: 3
|
|
1397
|
+
},
|
|
1398
|
+
"openai/gpt-5.6-sol": {
|
|
1399
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1400
|
+
latencyMs: 7720.2,
|
|
1401
|
+
p95LatencyMs: 9108.2,
|
|
1402
|
+
outputTokensPerSecond: 67.47,
|
|
954
1403
|
errorRate: 0,
|
|
955
1404
|
samples: 3
|
|
956
1405
|
},
|
|
957
|
-
"
|
|
958
|
-
measuredAt: "2026-
|
|
959
|
-
latencyMs:
|
|
960
|
-
p95LatencyMs:
|
|
961
|
-
outputTokensPerSecond:
|
|
1406
|
+
"openai/gpt-5.6-sol-pro": {
|
|
1407
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1408
|
+
latencyMs: 11442.7,
|
|
1409
|
+
p95LatencyMs: 13363.1,
|
|
1410
|
+
outputTokensPerSecond: 148.75,
|
|
962
1411
|
errorRate: 0,
|
|
963
1412
|
samples: 3
|
|
964
1413
|
},
|
|
965
|
-
"
|
|
966
|
-
measuredAt: "2026-
|
|
967
|
-
latencyMs:
|
|
968
|
-
p95LatencyMs:
|
|
969
|
-
outputTokensPerSecond:
|
|
1414
|
+
"openai/gpt-5.6-terra": {
|
|
1415
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1416
|
+
latencyMs: 4941,
|
|
1417
|
+
p95LatencyMs: 5095.3,
|
|
1418
|
+
outputTokensPerSecond: 103.69,
|
|
970
1419
|
errorRate: 0,
|
|
971
1420
|
samples: 3
|
|
972
1421
|
},
|
|
973
|
-
"
|
|
974
|
-
measuredAt: "2026-
|
|
975
|
-
latencyMs:
|
|
976
|
-
p95LatencyMs:
|
|
977
|
-
outputTokensPerSecond:
|
|
1422
|
+
"openai/gpt-5.6-terra-pro": {
|
|
1423
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1424
|
+
latencyMs: 3574.1,
|
|
1425
|
+
p95LatencyMs: 4126.3,
|
|
1426
|
+
outputTokensPerSecond: 133.59,
|
|
978
1427
|
errorRate: 0,
|
|
979
1428
|
samples: 3
|
|
980
1429
|
},
|
|
981
|
-
"
|
|
982
|
-
measuredAt: "2026-
|
|
983
|
-
latencyMs:
|
|
984
|
-
p95LatencyMs:
|
|
985
|
-
outputTokensPerSecond:
|
|
1430
|
+
"openai/o1": {
|
|
1431
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1432
|
+
latencyMs: 4324.9,
|
|
1433
|
+
p95LatencyMs: 5838.1,
|
|
1434
|
+
outputTokensPerSecond: 125.86,
|
|
986
1435
|
errorRate: 0,
|
|
987
1436
|
samples: 3
|
|
988
1437
|
},
|
|
989
|
-
"
|
|
990
|
-
measuredAt: "2026-
|
|
991
|
-
latencyMs:
|
|
992
|
-
p95LatencyMs:
|
|
993
|
-
outputTokensPerSecond:
|
|
1438
|
+
"openai/o3": {
|
|
1439
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1440
|
+
latencyMs: 5463.4,
|
|
1441
|
+
p95LatencyMs: 5613.1,
|
|
1442
|
+
outputTokensPerSecond: 93.8,
|
|
994
1443
|
errorRate: 0,
|
|
995
1444
|
samples: 3
|
|
996
1445
|
},
|
|
997
|
-
"
|
|
998
|
-
measuredAt: "2026-
|
|
999
|
-
latencyMs:
|
|
1000
|
-
p95LatencyMs:
|
|
1001
|
-
outputTokensPerSecond:
|
|
1446
|
+
"openai/o3-mini": {
|
|
1447
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1448
|
+
latencyMs: 2912.7,
|
|
1449
|
+
p95LatencyMs: 3092.1,
|
|
1450
|
+
outputTokensPerSecond: 176.49,
|
|
1002
1451
|
errorRate: 0,
|
|
1003
1452
|
samples: 3
|
|
1004
1453
|
},
|
|
1005
|
-
"
|
|
1006
|
-
measuredAt: "2026-
|
|
1007
|
-
latencyMs:
|
|
1008
|
-
p95LatencyMs:
|
|
1009
|
-
outputTokensPerSecond:
|
|
1010
|
-
errorRate: 0
|
|
1454
|
+
"openai/o4-mini": {
|
|
1455
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1456
|
+
latencyMs: 4958.7,
|
|
1457
|
+
p95LatencyMs: 5313,
|
|
1458
|
+
outputTokensPerSecond: 103.81,
|
|
1459
|
+
errorRate: 0,
|
|
1011
1460
|
samples: 3
|
|
1012
1461
|
},
|
|
1013
|
-
"
|
|
1014
|
-
measuredAt: "2026-
|
|
1015
|
-
latencyMs:
|
|
1016
|
-
p95LatencyMs:
|
|
1017
|
-
outputTokensPerSecond:
|
|
1462
|
+
"qwen/qwen3.7-flash": {
|
|
1463
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1464
|
+
latencyMs: 3385.5,
|
|
1465
|
+
p95LatencyMs: 4042.7,
|
|
1466
|
+
outputTokensPerSecond: 153.94,
|
|
1018
1467
|
errorRate: 0,
|
|
1019
1468
|
samples: 3
|
|
1020
1469
|
},
|
|
1021
|
-
"
|
|
1022
|
-
measuredAt: "2026-
|
|
1023
|
-
latencyMs:
|
|
1024
|
-
p95LatencyMs:
|
|
1025
|
-
outputTokensPerSecond:
|
|
1470
|
+
"qwen/qwen3.7-max": {
|
|
1471
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1472
|
+
latencyMs: 9387.1,
|
|
1473
|
+
p95LatencyMs: 10490.2,
|
|
1474
|
+
outputTokensPerSecond: 54.92,
|
|
1026
1475
|
errorRate: 0,
|
|
1027
1476
|
samples: 3
|
|
1028
1477
|
},
|
|
1029
|
-
"
|
|
1030
|
-
measuredAt: "2026-
|
|
1031
|
-
latencyMs:
|
|
1032
|
-
p95LatencyMs:
|
|
1033
|
-
outputTokensPerSecond:
|
|
1034
|
-
errorRate: 0
|
|
1478
|
+
"qwen/qwen3.7-plus": {
|
|
1479
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1480
|
+
latencyMs: 9766.6,
|
|
1481
|
+
p95LatencyMs: 9798.2,
|
|
1482
|
+
outputTokensPerSecond: 52.42,
|
|
1483
|
+
errorRate: 0,
|
|
1035
1484
|
samples: 3
|
|
1036
1485
|
},
|
|
1037
|
-
"
|
|
1038
|
-
measuredAt: "2026-
|
|
1039
|
-
latencyMs:
|
|
1040
|
-
p95LatencyMs:
|
|
1041
|
-
outputTokensPerSecond:
|
|
1486
|
+
"tencent/hy3": {
|
|
1487
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1488
|
+
latencyMs: 6062.3,
|
|
1489
|
+
p95LatencyMs: 7070.2,
|
|
1490
|
+
outputTokensPerSecond: 87.3,
|
|
1042
1491
|
errorRate: 0,
|
|
1043
1492
|
samples: 3
|
|
1044
1493
|
},
|
|
1045
|
-
"
|
|
1046
|
-
measuredAt: "2026-
|
|
1047
|
-
latencyMs:
|
|
1048
|
-
p95LatencyMs:
|
|
1049
|
-
outputTokensPerSecond:
|
|
1494
|
+
"xai/grok-4.3": {
|
|
1495
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1496
|
+
latencyMs: 9467.7,
|
|
1497
|
+
p95LatencyMs: 10087.9,
|
|
1498
|
+
outputTokensPerSecond: 48.36,
|
|
1050
1499
|
errorRate: 0,
|
|
1051
1500
|
samples: 3
|
|
1052
1501
|
},
|
|
1053
|
-
"
|
|
1054
|
-
measuredAt: "2026-
|
|
1055
|
-
latencyMs:
|
|
1056
|
-
p95LatencyMs:
|
|
1057
|
-
outputTokensPerSecond:
|
|
1502
|
+
"xai/grok-4.5": {
|
|
1503
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1504
|
+
latencyMs: 13564.8,
|
|
1505
|
+
p95LatencyMs: 17351.9,
|
|
1506
|
+
outputTokensPerSecond: 60.71,
|
|
1058
1507
|
errorRate: 0,
|
|
1059
1508
|
samples: 3
|
|
1060
1509
|
},
|
|
1061
|
-
"
|
|
1062
|
-
measuredAt: "2026-
|
|
1063
|
-
latencyMs:
|
|
1064
|
-
p95LatencyMs:
|
|
1065
|
-
outputTokensPerSecond:
|
|
1510
|
+
"xai/grok-build-0.1": {
|
|
1511
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1512
|
+
latencyMs: 16394.8,
|
|
1513
|
+
p95LatencyMs: 18035.4,
|
|
1514
|
+
outputTokensPerSecond: 96.86,
|
|
1066
1515
|
errorRate: 0,
|
|
1067
1516
|
samples: 3
|
|
1068
1517
|
},
|
|
1069
|
-
"
|
|
1070
|
-
measuredAt: "2026-
|
|
1071
|
-
latencyMs:
|
|
1072
|
-
p95LatencyMs:
|
|
1073
|
-
outputTokensPerSecond:
|
|
1518
|
+
"xiaomi/mimo-v2.5-pro": {
|
|
1519
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1520
|
+
latencyMs: 12070.7,
|
|
1521
|
+
p95LatencyMs: 12386.8,
|
|
1522
|
+
outputTokensPerSecond: 42.44,
|
|
1074
1523
|
errorRate: 0,
|
|
1075
1524
|
samples: 3
|
|
1076
1525
|
},
|
|
1077
1526
|
"zai/glm-5": {
|
|
1078
|
-
measuredAt: "2026-
|
|
1079
|
-
latencyMs:
|
|
1080
|
-
p95LatencyMs:
|
|
1081
|
-
outputTokensPerSecond:
|
|
1527
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1528
|
+
latencyMs: 6839.7,
|
|
1529
|
+
p95LatencyMs: 7261.4,
|
|
1530
|
+
outputTokensPerSecond: 75.16,
|
|
1531
|
+
errorRate: 0,
|
|
1532
|
+
samples: 3
|
|
1533
|
+
},
|
|
1534
|
+
"zai/glm-5-turbo": {
|
|
1535
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1536
|
+
latencyMs: 55348.5,
|
|
1537
|
+
p95LatencyMs: 114086.6,
|
|
1538
|
+
outputTokensPerSecond: 14.64,
|
|
1082
1539
|
errorRate: 0,
|
|
1083
1540
|
samples: 3
|
|
1084
1541
|
},
|
|
1085
|
-
"
|
|
1086
|
-
measuredAt: "2026-
|
|
1087
|
-
latencyMs:
|
|
1088
|
-
p95LatencyMs:
|
|
1089
|
-
outputTokensPerSecond:
|
|
1542
|
+
"zai/glm-5.1": {
|
|
1543
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1544
|
+
latencyMs: 15658.4,
|
|
1545
|
+
p95LatencyMs: 17307.1,
|
|
1546
|
+
outputTokensPerSecond: 32.9,
|
|
1090
1547
|
errorRate: 0,
|
|
1091
1548
|
samples: 3
|
|
1092
1549
|
},
|
|
1093
|
-
"
|
|
1094
|
-
measuredAt: "2026-
|
|
1095
|
-
latencyMs:
|
|
1096
|
-
p95LatencyMs:
|
|
1097
|
-
outputTokensPerSecond:
|
|
1550
|
+
"zai/glm-5.2": {
|
|
1551
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1552
|
+
latencyMs: 10308.5,
|
|
1553
|
+
p95LatencyMs: 15127.6,
|
|
1554
|
+
outputTokensPerSecond: 54.87,
|
|
1098
1555
|
errorRate: 0,
|
|
1099
1556
|
samples: 3
|
|
1100
1557
|
},
|
|
1101
|
-
"
|
|
1102
|
-
measuredAt: "2026-
|
|
1103
|
-
latencyMs:
|
|
1104
|
-
p95LatencyMs:
|
|
1105
|
-
outputTokensPerSecond:
|
|
1558
|
+
"zai/glm-5.3": {
|
|
1559
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1560
|
+
latencyMs: 7272.4,
|
|
1561
|
+
p95LatencyMs: 7998.1,
|
|
1562
|
+
outputTokensPerSecond: 71.09,
|
|
1106
1563
|
errorRate: 0,
|
|
1107
1564
|
samples: 3
|
|
1108
1565
|
},
|
|
1109
|
-
"
|
|
1110
|
-
measuredAt: "2026-
|
|
1111
|
-
latencyMs:
|
|
1112
|
-
p95LatencyMs:
|
|
1113
|
-
outputTokensPerSecond:
|
|
1566
|
+
"zai/glm-5.3-flash": {
|
|
1567
|
+
measuredAt: "2026-08-29T16:51:33Z",
|
|
1568
|
+
latencyMs: 10545.3,
|
|
1569
|
+
p95LatencyMs: 11672.4,
|
|
1570
|
+
outputTokensPerSecond: 49.01,
|
|
1114
1571
|
errorRate: 0,
|
|
1115
1572
|
samples: 3
|
|
1116
1573
|
}
|
|
@@ -1124,11 +1581,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1124
1581
|
latencyMs: 2305,
|
|
1125
1582
|
outputTokensPerSecond: 140.6
|
|
1126
1583
|
},
|
|
1127
|
-
"anthropic/claude-opus-4.6": {
|
|
1128
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1129
|
-
latencyMs: 2139,
|
|
1130
|
-
outputTokensPerSecond: 119.7
|
|
1131
|
-
},
|
|
1132
1584
|
"anthropic/claude-sonnet-4.6": {
|
|
1133
1585
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1134
1586
|
latencyMs: 2110,
|
|
@@ -1162,11 +1614,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1162
1614
|
latencyMs: 1609,
|
|
1163
1615
|
outputTokensPerSecond: 167.2
|
|
1164
1616
|
},
|
|
1165
|
-
"moonshot/kimi-k2.5": {
|
|
1166
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1167
|
-
latencyMs: 1646,
|
|
1168
|
-
outputTokensPerSecond: 155.7
|
|
1169
|
-
},
|
|
1170
1617
|
"openai/gpt-4o-mini": {
|
|
1171
1618
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1172
1619
|
latencyMs: 2764,
|
|
@@ -1176,18 +1623,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
|
|
|
1176
1623
|
measuredAt: "2026-03-16T13:50:48Z",
|
|
1177
1624
|
latencyMs: 7935,
|
|
1178
1625
|
outputTokensPerSecond: 32.3
|
|
1179
|
-
},
|
|
1180
|
-
"xai/grok-4-1-fast-non-reasoning": {
|
|
1181
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1182
|
-
latencyMs: 1244,
|
|
1183
|
-
outputTokensPerSecond: 205.8,
|
|
1184
|
-
intelligenceIndex: 41
|
|
1185
|
-
},
|
|
1186
|
-
"xai/grok-4-1-fast-reasoning": {
|
|
1187
|
-
measuredAt: "2026-03-16T13:50:48Z",
|
|
1188
|
-
latencyMs: 1454,
|
|
1189
|
-
outputTokensPerSecond: 176.2,
|
|
1190
|
-
intelligenceIndex: 41
|
|
1191
1626
|
}
|
|
1192
1627
|
});
|
|
1193
1628
|
function inferToolRequirement(prompt, _systemPrompt, toolChoice) {
|
|
@@ -1680,7 +2115,7 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1680
2115
|
match(["gpt-5.3-codex"], 1),
|
|
1681
2116
|
match(["claude-sonnet-4.6"], 0.94),
|
|
1682
2117
|
match(["glm-5.2"], 0.9),
|
|
1683
|
-
match(["
|
|
2118
|
+
match(["deepseek-v4-pro"], 0.86)
|
|
1684
2119
|
);
|
|
1685
2120
|
case "reasoning":
|
|
1686
2121
|
return Math.max(
|
|
@@ -1704,14 +2139,13 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1704
2139
|
base,
|
|
1705
2140
|
match(["gemini-3.5-flash"], 1),
|
|
1706
2141
|
match(["grok-4.5"], 0.93),
|
|
1707
|
-
match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
|
|
1708
|
-
match(["kimi-k2.7"], 0.84)
|
|
2142
|
+
match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
|
|
1709
2143
|
);
|
|
1710
2144
|
case "vision":
|
|
1711
2145
|
return Math.max(
|
|
1712
2146
|
base,
|
|
1713
2147
|
match(["gemini-3.1-pro"], 0.96),
|
|
1714
|
-
match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-
|
|
2148
|
+
match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9)
|
|
1715
2149
|
);
|
|
1716
2150
|
case "long_context":
|
|
1717
2151
|
return Math.max(
|
|
@@ -1723,17 +2157,18 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
|
|
|
1723
2157
|
);
|
|
1724
2158
|
case "extraction": {
|
|
1725
2159
|
const kimiExtractionAffinity = language === "zh" ? 1 : 0.9;
|
|
2160
|
+
const otherExtractionAffinity = language === "zh" ? 0.88 : 0.9;
|
|
1726
2161
|
return Math.max(
|
|
1727
2162
|
base,
|
|
1728
|
-
match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"],
|
|
1729
|
-
match(["claude-sonnet-5", "claude-sonnet-4.6"],
|
|
1730
|
-
match(["kimi-k3"
|
|
2163
|
+
match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], otherExtractionAffinity),
|
|
2164
|
+
match(["claude-sonnet-5", "claude-sonnet-4.6"], otherExtractionAffinity),
|
|
2165
|
+
match(["kimi-k3"], kimiExtractionAffinity)
|
|
1731
2166
|
);
|
|
1732
2167
|
}
|
|
1733
2168
|
default:
|
|
1734
2169
|
return Math.max(
|
|
1735
2170
|
base,
|
|
1736
|
-
match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"
|
|
2171
|
+
match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)
|
|
1737
2172
|
);
|
|
1738
2173
|
}
|
|
1739
2174
|
}
|
|
@@ -1792,6 +2227,9 @@ function evidenceCandidates(task) {
|
|
|
1792
2227
|
"deepseek/deepseek-v4-pro"
|
|
1793
2228
|
];
|
|
1794
2229
|
}
|
|
2230
|
+
if (task === "extraction") {
|
|
2231
|
+
return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"];
|
|
2232
|
+
}
|
|
1795
2233
|
if (task === "reasoning_math") {
|
|
1796
2234
|
return [
|
|
1797
2235
|
"google/gemini-3.5-flash",
|
|
@@ -1848,9 +2286,12 @@ var PortfolioStrategy = class {
|
|
|
1848
2286
|
const targetTier = (features.taskType === "reasoning_mcq" || features.taskType === "reasoning_math") && (base.tier === "SIMPLE" || base.tier === "MEDIUM") ? "REASONING" : base.tier;
|
|
1849
2287
|
const tierConfig = tierConfigs[targetTier];
|
|
1850
2288
|
const configuredCandidates = tierConfig ? getFallbackChain(targetTier, tierConfigs) : [];
|
|
2289
|
+
const unavailable = new Set(options.unavailableModels ?? []);
|
|
1851
2290
|
const chain = [
|
|
1852
2291
|
.../* @__PURE__ */ new Set([...configuredCandidates, ...evidenceCandidates(features.taskType)])
|
|
1853
|
-
].filter(
|
|
2292
|
+
].filter(
|
|
2293
|
+
(model2) => typeof model2 === "string" && model2.length > 0 && !unavailable.has(model2)
|
|
2294
|
+
);
|
|
1854
2295
|
const eligible = chain.filter(
|
|
1855
2296
|
(model2) => isEligible(model2, features, maxOutputTokens, options)
|
|
1856
2297
|
);
|
|
@@ -1927,13 +2368,7 @@ var PortfolioStrategy = class {
|
|
|
1927
2368
|
...eligibleCandidates.filter(
|
|
1928
2369
|
(model2) => !scoredModels.includes(model2) && !webResearchFallbackOrder.includes(model2)
|
|
1929
2370
|
)
|
|
1930
|
-
] :
|
|
1931
|
-
...scoredModels,
|
|
1932
|
-
...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
|
|
1933
|
-
] : [
|
|
1934
|
-
...scoredModels,
|
|
1935
|
-
...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
|
|
1936
|
-
];
|
|
2371
|
+
] : [...scoredModels, ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))];
|
|
1937
2372
|
const model = ranked[0] ?? base.model;
|
|
1938
2373
|
const selectedTierConfigs = {
|
|
1939
2374
|
...tierConfigs,
|
|
@@ -1970,7 +2405,7 @@ var PortfolioStrategy = class {
|
|
|
1970
2405
|
}
|
|
1971
2406
|
};
|
|
1972
2407
|
var DEFAULT_ROUTING_CONFIG = {
|
|
1973
|
-
version: "3.
|
|
2408
|
+
version: "3.5",
|
|
1974
2409
|
strategy: "portfolio",
|
|
1975
2410
|
portfolio: {
|
|
1976
2411
|
auto: {
|
|
@@ -3024,185 +3459,249 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3024
3459
|
// Below this confidence → ambiguous (null tier)
|
|
3025
3460
|
confidenceThreshold: 0.7
|
|
3026
3461
|
},
|
|
3462
|
+
// ─── Tier chains ───
|
|
3463
|
+
//
|
|
3464
|
+
// Catalog refresh 2026-08-29 (V3.5). Every chain below names only models
|
|
3465
|
+
// the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the
|
|
3466
|
+
// gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast
|
|
3467
|
+
// and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview,
|
|
3468
|
+
// the whole `free/*` namespace — were removed everywhere, including fallback
|
|
3469
|
+
// rungs, so a routed model is always one a user can find on blockrun.ai/models.
|
|
3470
|
+
//
|
|
3471
|
+
// Primaries moved only where portfolio.ts already carries calibration
|
|
3472
|
+
// evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for
|
|
3473
|
+
// agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no
|
|
3474
|
+
// trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash,
|
|
3475
|
+
// gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs;
|
|
3476
|
+
// promotion waits for a calibration run, because version recency is not a
|
|
3477
|
+
// quality signal.
|
|
3478
|
+
//
|
|
3479
|
+
// Latency figures in comments are the 2026-08-29 gateway probe
|
|
3480
|
+
// (model-profiles.generated.json); prices are the catalog list.
|
|
3027
3481
|
// Auto (balanced) tier configs - current default smart routing
|
|
3028
|
-
// Benchmark-tuned 2026-03-16: balancing quality (retention) + latency
|
|
3029
3482
|
tiers: {
|
|
3030
3483
|
SIMPLE: {
|
|
3031
3484
|
primary: "google/gemini-2.5-flash",
|
|
3032
|
-
//
|
|
3485
|
+
// $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer
|
|
3033
3486
|
fallback: [
|
|
3034
3487
|
"google/gemini-3-flash-preview",
|
|
3035
|
-
//
|
|
3488
|
+
// $0.50/$3 — GPQA 5/6 in the 2026-07 calibration
|
|
3489
|
+
"google/gemini-3.5-flash-lite",
|
|
3490
|
+
// $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation
|
|
3036
3491
|
"deepseek/deepseek-chat",
|
|
3037
|
-
//
|
|
3038
|
-
"moonshot/kimi-k2.5",
|
|
3039
|
-
// 1,646ms, IQ 47, strong quality
|
|
3492
|
+
// $0.14/$0.28, 1M ctx
|
|
3040
3493
|
"google/gemini-3.1-flash-lite",
|
|
3041
|
-
// $0.25/$1.50, 1M
|
|
3042
|
-
"
|
|
3043
|
-
//
|
|
3494
|
+
// $0.25/$1.50, 1M ctx
|
|
3495
|
+
"openai/gpt-5.6-luna",
|
|
3496
|
+
// $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30)
|
|
3044
3497
|
"openai/gpt-5.4-nano",
|
|
3045
|
-
// $0.20/$1.25, 1M
|
|
3046
|
-
"
|
|
3047
|
-
//
|
|
3048
|
-
"
|
|
3049
|
-
//
|
|
3498
|
+
// $0.20/$1.25, 1M ctx
|
|
3499
|
+
"google/gemini-2.5-flash-lite",
|
|
3500
|
+
// $0.10/$0.40
|
|
3501
|
+
"nvidia/step-3.7-flash"
|
|
3502
|
+
// FREE backstop — NVIDIA free tier (probed 2026-08-21)
|
|
3050
3503
|
]
|
|
3051
3504
|
},
|
|
3052
3505
|
MEDIUM: {
|
|
3053
|
-
|
|
3054
|
-
//
|
|
3506
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the
|
|
3507
|
+
// calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts).
|
|
3508
|
+
primary: "google/gemini-3.5-flash",
|
|
3509
|
+
// $1.50/$9, 1M ctx, vision + tools
|
|
3055
3510
|
fallback: [
|
|
3056
|
-
"
|
|
3057
|
-
//
|
|
3058
|
-
"
|
|
3059
|
-
// $0.
|
|
3511
|
+
"google/gemini-3.6-flash",
|
|
3512
|
+
// $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration
|
|
3513
|
+
"zai/glm-5.3-flash",
|
|
3514
|
+
// $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27
|
|
3515
|
+
"openai/gpt-5.6-terra",
|
|
3516
|
+
// $2/$12, 1M ctx — GPT-5.6 balanced tier
|
|
3060
3517
|
"google/gemini-3-flash-preview",
|
|
3061
|
-
//
|
|
3518
|
+
// $0.50/$3
|
|
3062
3519
|
"deepseek/deepseek-chat",
|
|
3063
|
-
//
|
|
3520
|
+
// $0.14/$0.28
|
|
3064
3521
|
"google/gemini-2.5-flash",
|
|
3065
|
-
//
|
|
3522
|
+
// $0.30/$2.50
|
|
3523
|
+
"minimax/minimax-m3",
|
|
3524
|
+
// $0.30/$1.20, 1M ctx
|
|
3066
3525
|
"google/gemini-3.1-flash-lite",
|
|
3067
|
-
// $0.25/$1.50
|
|
3068
|
-
"
|
|
3069
|
-
//
|
|
3070
|
-
"
|
|
3071
|
-
//
|
|
3072
|
-
"xai/grok-3-mini"
|
|
3073
|
-
// 1,202ms, $0.30/$0.50
|
|
3526
|
+
// $0.25/$1.50
|
|
3527
|
+
"openai/gpt-5.6-luna",
|
|
3528
|
+
// $0.20/$1.20
|
|
3529
|
+
"google/gemini-2.5-flash-lite"
|
|
3530
|
+
// $0.10/$0.40
|
|
3074
3531
|
]
|
|
3075
3532
|
},
|
|
3076
3533
|
COMPLEX: {
|
|
3077
3534
|
primary: "google/gemini-3.1-pro",
|
|
3078
|
-
//
|
|
3535
|
+
// $2/$12 — proven long-context flagship (portfolio.ts long_context lead)
|
|
3079
3536
|
fallback: [
|
|
3080
|
-
"google/gemini-3-flash
|
|
3081
|
-
// 1
|
|
3082
|
-
"
|
|
3083
|
-
// 1
|
|
3084
|
-
"google/gemini-2.5-pro",
|
|
3085
|
-
// 1,294ms
|
|
3537
|
+
"google/gemini-3.6-flash",
|
|
3538
|
+
// $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here)
|
|
3539
|
+
"google/gemini-3.5-flash",
|
|
3540
|
+
// $1.50/$9 — calibrated
|
|
3086
3541
|
"anthropic/claude-sonnet-5",
|
|
3087
|
-
// near-Opus quality
|
|
3542
|
+
// $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated
|
|
3543
|
+
"xai/grok-4.5",
|
|
3544
|
+
// $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden)
|
|
3545
|
+
"google/gemini-2.5-pro",
|
|
3546
|
+
// $1.25/$10
|
|
3088
3547
|
"anthropic/claude-sonnet-4.6",
|
|
3089
|
-
//
|
|
3090
|
-
"deepseek/deepseek-chat",
|
|
3091
|
-
// 1,431ms, IQ 32
|
|
3092
|
-
"google/gemini-2.5-flash",
|
|
3093
|
-
// 1,238ms, IQ 20 — cheap last resort
|
|
3548
|
+
// $3/$15
|
|
3094
3549
|
"openai/gpt-5.6-terra",
|
|
3095
|
-
// GPT-5.6 balanced tier
|
|
3550
|
+
// $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202)
|
|
3096
3551
|
"openai/gpt-5.5",
|
|
3097
|
-
//
|
|
3098
|
-
"openai/gpt-5.4"
|
|
3099
|
-
//
|
|
3552
|
+
// $5/$30 — prior OpenAI flagship
|
|
3553
|
+
"openai/gpt-5.4",
|
|
3554
|
+
// $2.50/$15 — previous flagship, benchmarked
|
|
3555
|
+
"zai/glm-5.3",
|
|
3556
|
+
// $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19
|
|
3557
|
+
"moonshot/kimi-k3",
|
|
3558
|
+
// $3/$15, 1M ctx — Moonshot flagship (K2.7 successor)
|
|
3559
|
+
"deepseek/deepseek-v4-pro",
|
|
3560
|
+
// $0.435/$0.87 — strongest open-weight reasoner
|
|
3561
|
+
"deepseek/deepseek-chat",
|
|
3562
|
+
// $0.14/$0.28 — cheap last resort
|
|
3563
|
+
"google/gemini-2.5-flash"
|
|
3564
|
+
// $0.30/$2.50
|
|
3100
3565
|
]
|
|
3101
3566
|
},
|
|
3102
3567
|
REASONING: {
|
|
3103
|
-
|
|
3104
|
-
//
|
|
3568
|
+
// Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek
|
|
3569
|
+
// Reasoner is the cheapest listed reasoner at the same 1M context.
|
|
3570
|
+
primary: "deepseek/deepseek-reasoner",
|
|
3571
|
+
// $0.14/$0.28, 1M ctx
|
|
3105
3572
|
fallback: [
|
|
3106
|
-
"xai/grok-4-fast-reasoning",
|
|
3107
|
-
// 1,298ms, $0.20/$0.50
|
|
3108
|
-
"deepseek/deepseek-reasoner",
|
|
3109
|
-
// V4 Flash thinking ($0.20/$0.40, 1M ctx)
|
|
3110
3573
|
"deepseek/deepseek-v4-pro",
|
|
3111
|
-
//
|
|
3574
|
+
// $0.435/$0.87 — calibrated reasoning band 0.95
|
|
3575
|
+
"xai/grok-4.3",
|
|
3576
|
+
// $1.50/$4, 1M ctx — xAI reasoning model, vision
|
|
3577
|
+
"qwen/qwen3.7-plus",
|
|
3578
|
+
// $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed)
|
|
3579
|
+
"google/gemini-3.5-flash",
|
|
3580
|
+
// $1.50/$9 — MGSM 5/5
|
|
3112
3581
|
"openai/o4-mini",
|
|
3113
|
-
//
|
|
3582
|
+
// $1.10/$4.40
|
|
3114
3583
|
"openai/o3"
|
|
3115
|
-
// 2
|
|
3584
|
+
// $2/$8
|
|
3116
3585
|
]
|
|
3117
3586
|
}
|
|
3118
3587
|
},
|
|
3119
3588
|
// Eco tier configs - absolute cheapest (blockrun/eco)
|
|
3120
3589
|
ecoTiers: {
|
|
3121
3590
|
SIMPLE: {
|
|
3122
|
-
primary: "
|
|
3123
|
-
// FREE
|
|
3591
|
+
primary: "nvidia/step-3.7-flash",
|
|
3592
|
+
// FREE — NVIDIA free tier flagship
|
|
3124
3593
|
fallback: [
|
|
3125
|
-
"
|
|
3126
|
-
// FREE —
|
|
3127
|
-
//
|
|
3128
|
-
//
|
|
3129
|
-
//
|
|
3130
|
-
|
|
3131
|
-
// $0.25/$1.50 — newest flash-lite
|
|
3132
|
-
"openai/gpt-5.4-nano",
|
|
3133
|
-
// $0.20/$1.25 — fast nano
|
|
3594
|
+
"nvidia/nemotron-nano-9b-v2",
|
|
3595
|
+
// FREE — compact + fast, high-volume light tasks
|
|
3596
|
+
// The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash
|
|
3597
|
+
// 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400
|
|
3598
|
+
// 2026-08-21). Each retirement retargets the two free rungs to the
|
|
3599
|
+
// current free tier; the paid rungs below never move.
|
|
3134
3600
|
"google/gemini-2.5-flash-lite",
|
|
3135
|
-
// $0.10/$0.40
|
|
3136
|
-
"
|
|
3137
|
-
// $0.
|
|
3601
|
+
// $0.10/$0.40 — cheapest paid rung
|
|
3602
|
+
"zai/glm-5.3-flash",
|
|
3603
|
+
// $0.15/$0.50, 1M ctx, vision + tools
|
|
3604
|
+
"openai/gpt-5.6-luna",
|
|
3605
|
+
// $0.20/$1.20, 1M ctx
|
|
3606
|
+
"openai/gpt-5.4-nano",
|
|
3607
|
+
// $0.20/$1.25
|
|
3608
|
+
"google/gemini-3.1-flash-lite"
|
|
3609
|
+
// $0.25/$1.50
|
|
3138
3610
|
]
|
|
3139
3611
|
},
|
|
3140
3612
|
MEDIUM: {
|
|
3141
|
-
primary: "
|
|
3142
|
-
// $0.
|
|
3613
|
+
primary: "zai/glm-5.3-flash",
|
|
3614
|
+
// $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model
|
|
3143
3615
|
fallback: [
|
|
3616
|
+
"deepseek/deepseek-chat",
|
|
3617
|
+
// $0.14/$0.28
|
|
3618
|
+
"google/gemini-3.1-flash-lite",
|
|
3619
|
+
// $0.25/$1.50
|
|
3620
|
+
"openai/gpt-5.6-luna",
|
|
3621
|
+
// $0.20/$1.20
|
|
3144
3622
|
"openai/gpt-5.4-nano",
|
|
3145
3623
|
// $0.20/$1.25
|
|
3146
3624
|
"google/gemini-2.5-flash-lite",
|
|
3147
3625
|
// $0.10/$0.40
|
|
3148
|
-
"xai/grok-4-fast-non-reasoning",
|
|
3149
3626
|
"google/gemini-2.5-flash"
|
|
3627
|
+
// $0.30/$2.50
|
|
3150
3628
|
]
|
|
3151
3629
|
},
|
|
3152
3630
|
COMPLEX: {
|
|
3153
|
-
primary: "
|
|
3154
|
-
// $0.
|
|
3631
|
+
primary: "zai/glm-5.3-flash",
|
|
3632
|
+
// $0.15/$0.50, 1M ctx
|
|
3155
3633
|
fallback: [
|
|
3156
|
-
"
|
|
3157
|
-
|
|
3158
|
-
"
|
|
3159
|
-
|
|
3634
|
+
"deepseek/deepseek-chat",
|
|
3635
|
+
// $0.14/$0.28, 1M ctx
|
|
3636
|
+
"minimax/minimax-m3",
|
|
3637
|
+
// $0.30/$1.20, 1M ctx
|
|
3638
|
+
"deepseek/deepseek-v4-pro",
|
|
3639
|
+
// $0.435/$0.87
|
|
3640
|
+
"google/gemini-3.1-flash-lite",
|
|
3641
|
+
// $0.25/$1.50
|
|
3642
|
+
"google/gemini-2.5-flash"
|
|
3643
|
+
// $0.30/$2.50
|
|
3160
3644
|
]
|
|
3161
3645
|
},
|
|
3162
3646
|
REASONING: {
|
|
3163
|
-
primary: "
|
|
3164
|
-
// $0.
|
|
3647
|
+
primary: "deepseek/deepseek-reasoner",
|
|
3648
|
+
// $0.14/$0.28, 1M ctx — cheapest listed reasoner
|
|
3165
3649
|
fallback: [
|
|
3166
|
-
"
|
|
3167
|
-
|
|
3168
|
-
|
|
3169
|
-
|
|
3170
|
-
|
|
3650
|
+
"deepseek/deepseek-v4-pro",
|
|
3651
|
+
// $0.435/$0.87
|
|
3652
|
+
"qwen/qwen3.7-plus",
|
|
3653
|
+
// $0.32/$1.28 — reasoning
|
|
3654
|
+
"minimax/minimax-m3",
|
|
3655
|
+
// $0.30/$1.20 — reasoning + coding
|
|
3656
|
+
"zai/glm-5.3-flash"
|
|
3657
|
+
// $0.15/$0.50 — reasoning tokens alongside content
|
|
3171
3658
|
]
|
|
3172
3659
|
}
|
|
3173
3660
|
},
|
|
3174
3661
|
// Premium tier configs - best quality (blockrun/premium)
|
|
3175
|
-
// codex=complex coding,
|
|
3662
|
+
// codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits
|
|
3176
3663
|
premiumTiers: {
|
|
3177
3664
|
SIMPLE: {
|
|
3178
|
-
|
|
3179
|
-
|
|
3665
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08).
|
|
3666
|
+
primary: "google/gemini-3.5-flash",
|
|
3667
|
+
// $1.50/$9, 1M ctx, vision + tools — calibrated
|
|
3180
3668
|
fallback: [
|
|
3181
|
-
"
|
|
3182
|
-
//
|
|
3183
|
-
"moonshot/kimi-k2.5",
|
|
3184
|
-
// $0.60/$3.00 - proven reliable backstop when Moonshot direct API falters
|
|
3185
|
-
"google/gemini-2.5-flash",
|
|
3186
|
-
// 60% retention, fast growth
|
|
3669
|
+
"google/gemini-3.6-flash",
|
|
3670
|
+
// $1.50/$7.50 — newest Flash
|
|
3187
3671
|
"anthropic/claude-haiku-4.5",
|
|
3188
|
-
|
|
3672
|
+
// $1/$5
|
|
3673
|
+
"zai/glm-5.3",
|
|
3674
|
+
// $1.40/$4.40, 1M ctx
|
|
3675
|
+
"google/gemini-2.5-flash",
|
|
3676
|
+
// $0.30/$2.50
|
|
3677
|
+
"google/gemini-3.5-flash-lite",
|
|
3678
|
+
// $0.30/$2.50
|
|
3189
3679
|
"deepseek/deepseek-chat"
|
|
3680
|
+
// $0.14/$0.28
|
|
3190
3681
|
]
|
|
3191
3682
|
},
|
|
3192
3683
|
MEDIUM: {
|
|
3193
3684
|
primary: "openai/gpt-5.3-codex",
|
|
3194
|
-
// $1.75/$14 - 400K context, 128K output
|
|
3685
|
+
// $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts)
|
|
3195
3686
|
fallback: [
|
|
3196
|
-
"moonshot/kimi-k2.7",
|
|
3197
|
-
// Moonshot flagship
|
|
3198
|
-
"moonshot/kimi-k2.6",
|
|
3199
|
-
"moonshot/kimi-k2.5",
|
|
3200
|
-
"google/gemini-2.5-flash",
|
|
3201
|
-
// 60% retention, good coding capability
|
|
3202
|
-
"google/gemini-2.5-pro",
|
|
3203
|
-
"xai/grok-4-0709",
|
|
3204
3687
|
"anthropic/claude-sonnet-5",
|
|
3205
|
-
|
|
3688
|
+
// $3/$15 — code_agent band 0.98
|
|
3689
|
+
"moonshot/kimi-k3",
|
|
3690
|
+
// $3/$15, 1M ctx — Moonshot flagship
|
|
3691
|
+
"zai/glm-5.3",
|
|
3692
|
+
// $1.40/$4.40 — long-horizon coding
|
|
3693
|
+
"google/gemini-3.6-flash",
|
|
3694
|
+
// $1.50/$7.50
|
|
3695
|
+
"google/gemini-3.5-flash",
|
|
3696
|
+
// $1.50/$9
|
|
3697
|
+
"google/gemini-2.5-pro",
|
|
3698
|
+
// $1.25/$10
|
|
3699
|
+
"xai/grok-4.5",
|
|
3700
|
+
// $2.50/$9
|
|
3701
|
+
"anthropic/claude-sonnet-4.6",
|
|
3702
|
+
// $3/$15
|
|
3703
|
+
"openai/gpt-5.6-terra"
|
|
3704
|
+
// $2/$12
|
|
3206
3705
|
]
|
|
3207
3706
|
},
|
|
3208
3707
|
COMPLEX: {
|
|
@@ -3212,8 +3711,8 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3212
3711
|
// Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking)
|
|
3213
3712
|
// Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is
|
|
3214
3713
|
// also prone to "high demand" 503s (correlated failure — everyone falls
|
|
3215
|
-
// back to Google at the same time). Prefer xAI
|
|
3216
|
-
// flagship → DeepSeek → NVIDIA free instead.
|
|
3714
|
+
// back to Google at the same time). Prefer in-family → xAI → Moonshot →
|
|
3715
|
+
// OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead.
|
|
3217
3716
|
fallback: [
|
|
3218
3717
|
"anthropic/claude-opus-5",
|
|
3219
3718
|
// in-family hot swap first (half the price, 1M ctx + adaptive thinking)
|
|
@@ -3221,52 +3720,54 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3221
3720
|
// in-family hot swap (identical cost to 5)
|
|
3222
3721
|
"anthropic/claude-opus-4.7",
|
|
3223
3722
|
// in-family hot swap (identical cost to 4.8)
|
|
3224
|
-
"anthropic/claude-opus-4.6",
|
|
3225
|
-
// in-family hot swap
|
|
3226
3723
|
"anthropic/claude-sonnet-5",
|
|
3227
3724
|
// Sonnet-tier drop-down, near-Opus quality
|
|
3228
3725
|
"anthropic/claude-sonnet-4.6",
|
|
3229
3726
|
"xai/grok-4.5",
|
|
3230
|
-
// xAI flagship — 503-resistant, direct-xAI SKU
|
|
3231
|
-
"
|
|
3232
|
-
// 503-resistant flagship
|
|
3233
|
-
"moonshot/kimi-k2.7",
|
|
3727
|
+
// xAI flagship — 503-resistant, direct-xAI SKU
|
|
3728
|
+
"moonshot/kimi-k3",
|
|
3234
3729
|
// Moonshot flagship, independent infra
|
|
3235
|
-
"moonshot/kimi-k2.6",
|
|
3236
|
-
"moonshot/kimi-k2.5",
|
|
3237
3730
|
"openai/gpt-5.6-terra",
|
|
3238
|
-
// GPT-5.6 balanced tier —
|
|
3731
|
+
// GPT-5.6 balanced tier — stable (Sol excluded: #202)
|
|
3239
3732
|
"openai/gpt-5.5",
|
|
3240
3733
|
// Prior OpenAI flagship — 1M+ ctx, native agent + computer use
|
|
3241
3734
|
"openai/gpt-5.4",
|
|
3242
3735
|
// Previous flagship (slow but stable, benchmarked at 6,213ms)
|
|
3243
3736
|
"openai/gpt-5.3-codex",
|
|
3737
|
+
"zai/glm-5.3",
|
|
3738
|
+
// Z.AI flagship, 1M ctx
|
|
3739
|
+
"deepseek/deepseek-v4-pro",
|
|
3740
|
+
// strongest open-weight reasoner
|
|
3244
3741
|
"deepseek/deepseek-chat",
|
|
3245
3742
|
// Cheap, reliable
|
|
3246
|
-
"
|
|
3247
|
-
// NVIDIA free ultimate backstop
|
|
3743
|
+
"nvidia/step-3.7-flash"
|
|
3744
|
+
// NVIDIA free ultimate backstop
|
|
3248
3745
|
]
|
|
3249
3746
|
},
|
|
3250
3747
|
REASONING: {
|
|
3251
|
-
|
|
3252
|
-
//
|
|
3748
|
+
// Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both,
|
|
3749
|
+
// plus Sonnet 5's tau2/BrowseComp trajectory evidence).
|
|
3750
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3751
|
+
// $3/$15, 1M ctx, adaptive thinking
|
|
3253
3752
|
fallback: [
|
|
3254
|
-
"anthropic/claude-sonnet-
|
|
3255
|
-
// in-family hot swap — same cost
|
|
3753
|
+
"anthropic/claude-sonnet-4.6",
|
|
3754
|
+
// in-family hot swap — same cost
|
|
3256
3755
|
"anthropic/claude-opus-5",
|
|
3257
3756
|
// Newest flagship Opus w/ adaptive thinking
|
|
3258
3757
|
"anthropic/claude-opus-4.8",
|
|
3259
3758
|
// Prior flagship Opus — identical cost to 5
|
|
3260
3759
|
"anthropic/claude-opus-4.7",
|
|
3261
3760
|
// Flagship Opus w/ adaptive thinking
|
|
3262
|
-
"
|
|
3263
|
-
//
|
|
3264
|
-
"
|
|
3265
|
-
//
|
|
3761
|
+
"xai/grok-4.5",
|
|
3762
|
+
// reasoning band 0.94
|
|
3763
|
+
"deepseek/deepseek-v4-pro",
|
|
3764
|
+
// reasoning band 0.95
|
|
3765
|
+
"xai/grok-4.3",
|
|
3766
|
+
// $1.50/$4 — xAI reasoning model
|
|
3266
3767
|
"openai/o4-mini",
|
|
3267
|
-
//
|
|
3768
|
+
// $1.10/$4.40
|
|
3268
3769
|
"openai/o3"
|
|
3269
|
-
// 2
|
|
3770
|
+
// $2/$8
|
|
3270
3771
|
]
|
|
3271
3772
|
}
|
|
3272
3773
|
},
|
|
@@ -3276,101 +3777,102 @@ var DEFAULT_ROUTING_CONFIG = {
|
|
|
3276
3777
|
primary: "openai/gpt-4o-mini",
|
|
3277
3778
|
// $0.15/$0.60 - best tool compliance at lowest cost
|
|
3278
3779
|
fallback: [
|
|
3279
|
-
"
|
|
3280
|
-
// 1
|
|
3780
|
+
"openai/gpt-5.6-luna",
|
|
3781
|
+
// $0.20/$1.20 — lightweight agentic tier of GPT-5.6
|
|
3782
|
+
"zai/glm-5.3-flash",
|
|
3783
|
+
// $0.15/$0.50 — tool calls verified live 2026-08-27
|
|
3281
3784
|
"anthropic/claude-haiku-4.5",
|
|
3282
|
-
//
|
|
3283
|
-
"
|
|
3284
|
-
//
|
|
3785
|
+
// $1/$5
|
|
3786
|
+
"google/gemini-2.5-flash"
|
|
3787
|
+
// $0.30/$2.50
|
|
3285
3788
|
]
|
|
3286
3789
|
},
|
|
3287
3790
|
MEDIUM: {
|
|
3288
|
-
|
|
3289
|
-
//
|
|
3791
|
+
// Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the
|
|
3792
|
+
// Terminal-Bench and tau2 trajectory evidence in portfolio.ts.
|
|
3793
|
+
primary: "openai/gpt-5-mini",
|
|
3794
|
+
// $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline
|
|
3290
3795
|
fallback: [
|
|
3291
|
-
"
|
|
3292
|
-
//
|
|
3293
|
-
"
|
|
3294
|
-
// $0.
|
|
3295
|
-
"
|
|
3296
|
-
//
|
|
3796
|
+
"google/gemini-3.5-flash",
|
|
3797
|
+
// $1.50/$9 — tool_agent band 0.88
|
|
3798
|
+
"zai/glm-5.3-flash",
|
|
3799
|
+
// $0.15/$0.50 — tools verified
|
|
3800
|
+
"openai/gpt-5.6-terra",
|
|
3801
|
+
// $2/$12
|
|
3297
3802
|
"openai/gpt-4o-mini",
|
|
3298
|
-
//
|
|
3803
|
+
// $0.15/$0.60 — reliable tool calling
|
|
3299
3804
|
"anthropic/claude-haiku-4.5",
|
|
3300
|
-
//
|
|
3301
|
-
"deepseek/deepseek-chat"
|
|
3302
|
-
//
|
|
3805
|
+
// $1/$5
|
|
3806
|
+
"deepseek/deepseek-chat",
|
|
3807
|
+
// $0.14/$0.28
|
|
3808
|
+
"moonshot/kimi-k3"
|
|
3809
|
+
// $3/$15 — tool_agent band 0.85
|
|
3303
3810
|
]
|
|
3304
3811
|
},
|
|
3305
3812
|
COMPLEX: {
|
|
3306
|
-
|
|
3307
|
-
//
|
|
3813
|
+
// Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0,
|
|
3814
|
+
// Terminal-Bench safety band lead (portfolio.ts).
|
|
3815
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3816
|
+
// $3/$15 — best agentic quality per trajectory evidence
|
|
3308
3817
|
// Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s
|
|
3309
3818
|
// correlate with Anthropic outages (everyone falls back together).
|
|
3310
3819
|
// Prefer 503-resistant providers first.
|
|
3311
3820
|
fallback: [
|
|
3312
|
-
"anthropic/claude-sonnet-
|
|
3313
|
-
// in-family hot swap — same cost
|
|
3821
|
+
"anthropic/claude-sonnet-4.6",
|
|
3822
|
+
// in-family hot swap — same cost
|
|
3314
3823
|
"anthropic/claude-opus-5",
|
|
3315
3824
|
// Newest flagship Opus — in-family hot swap
|
|
3316
3825
|
"anthropic/claude-opus-4.8",
|
|
3317
3826
|
// Prior flagship Opus — identical cost to 5
|
|
3318
3827
|
"anthropic/claude-opus-4.7",
|
|
3319
3828
|
// Flagship Opus — in-family hot swap
|
|
3320
|
-
"
|
|
3321
|
-
//
|
|
3322
|
-
"
|
|
3323
|
-
//
|
|
3324
|
-
"moonshot/kimi-k2.7",
|
|
3325
|
-
// Moonshot flagship — strong tool use, independent infra
|
|
3326
|
-
"moonshot/kimi-k2.5",
|
|
3327
|
-
// cost-stability backstop
|
|
3829
|
+
"xai/grok-4.5",
|
|
3830
|
+
// xAI flagship — strong tool use, independent infra
|
|
3831
|
+
"moonshot/kimi-k3",
|
|
3832
|
+
// Moonshot flagship — independent infra
|
|
3328
3833
|
"openai/gpt-5.6-terra",
|
|
3329
|
-
// GPT-5.6 balanced tier —
|
|
3834
|
+
// GPT-5.6 balanced tier — stable (Sol excluded: #202)
|
|
3330
3835
|
"openai/gpt-5.5",
|
|
3331
3836
|
// Prior flagship — native agent + computer use (exactly the agentic-tier use case)
|
|
3332
3837
|
"openai/gpt-5.4",
|
|
3333
|
-
// Previous flagship —
|
|
3838
|
+
// Previous flagship — reliable
|
|
3839
|
+
"openai/gpt-5.3-codex",
|
|
3840
|
+
// code_agent lead
|
|
3841
|
+
"zai/glm-5.3",
|
|
3842
|
+
// long-horizon coding
|
|
3843
|
+
"deepseek/deepseek-v4-pro",
|
|
3844
|
+
// retail high-risk 3/3
|
|
3334
3845
|
"deepseek/deepseek-chat",
|
|
3335
|
-
//
|
|
3336
|
-
"
|
|
3337
|
-
// NVIDIA free ultimate backstop
|
|
3846
|
+
// cheap, reliable
|
|
3847
|
+
"nvidia/step-3.7-flash"
|
|
3848
|
+
// NVIDIA free ultimate backstop
|
|
3338
3849
|
]
|
|
3339
3850
|
},
|
|
3340
3851
|
REASONING: {
|
|
3341
|
-
primary: "anthropic/claude-sonnet-
|
|
3342
|
-
//
|
|
3852
|
+
primary: "anthropic/claude-sonnet-5",
|
|
3853
|
+
// $3/$15 — strong tool use + adaptive thinking
|
|
3343
3854
|
fallback: [
|
|
3344
|
-
"anthropic/claude-sonnet-
|
|
3345
|
-
// in-family hot swap — same cost
|
|
3855
|
+
"anthropic/claude-sonnet-4.6",
|
|
3856
|
+
// in-family hot swap — same cost
|
|
3346
3857
|
"anthropic/claude-opus-5",
|
|
3347
3858
|
// Newest flagship Opus w/ adaptive thinking
|
|
3348
3859
|
"anthropic/claude-opus-4.8",
|
|
3349
3860
|
// Prior flagship Opus — identical cost to 5
|
|
3350
3861
|
"anthropic/claude-opus-4.7",
|
|
3351
3862
|
// Flagship Opus w/ adaptive thinking
|
|
3352
|
-
"
|
|
3353
|
-
//
|
|
3354
|
-
"
|
|
3355
|
-
//
|
|
3863
|
+
"xai/grok-4.5",
|
|
3864
|
+
// reasoning band 0.94
|
|
3865
|
+
"deepseek/deepseek-v4-pro",
|
|
3866
|
+
// reasoning band 0.95
|
|
3356
3867
|
"deepseek/deepseek-reasoner"
|
|
3357
|
-
//
|
|
3868
|
+
// $0.14/$0.28
|
|
3358
3869
|
]
|
|
3359
3870
|
}
|
|
3360
3871
|
},
|
|
3361
|
-
// Time-windowed promotions — auto-applied when active, ignored when expired
|
|
3362
|
-
|
|
3363
|
-
|
|
3364
|
-
|
|
3365
|
-
startDate: "2026-04-01",
|
|
3366
|
-
endDate: "2026-05-01",
|
|
3367
|
-
tierOverrides: {
|
|
3368
|
-
SIMPLE: { primary: "zai/glm-5.1" }
|
|
3369
|
-
},
|
|
3370
|
-
profiles: ["auto"]
|
|
3371
|
-
// only auto profile — eco stays free, premium stays premium
|
|
3372
|
-
}
|
|
3373
|
-
],
|
|
3872
|
+
// Time-windowed promotions — auto-applied when active, ignored when expired.
|
|
3873
|
+
// The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and
|
|
3874
|
+
// has expired; the list is kept empty so the mechanism stays wired.
|
|
3875
|
+
promotions: [],
|
|
3374
3876
|
overrides: {
|
|
3375
3877
|
maxTokensForceComplex: 1e5,
|
|
3376
3878
|
structuredOutputMinTier: "MEDIUM",
|
|
@@ -3712,7 +4214,7 @@ async function createSolanaPaymentPayload(secretKey, fromAddress, recipient, amo
|
|
|
3712
4214
|
}
|
|
3713
4215
|
return null;
|
|
3714
4216
|
};
|
|
3715
|
-
let entry = await getBlockhashEntry(connection, rpcUrl, false);
|
|
4217
|
+
let entry = await getBlockhashEntry(connection, rpcUrl, options.forceFreshBlockhash ?? false);
|
|
3716
4218
|
let serializedTx = findDistinctTx(entry);
|
|
3717
4219
|
if (serializedTx === null) {
|
|
3718
4220
|
entry = await getBlockhashEntry(connection, rpcUrl, true);
|
|
@@ -3963,7 +4465,7 @@ function getCostSummary() {
|
|
|
3963
4465
|
}
|
|
3964
4466
|
|
|
3965
4467
|
// src/version.ts
|
|
3966
|
-
var SDK_VERSION = "3.13.
|
|
4468
|
+
var SDK_VERSION = "3.13.5";
|
|
3967
4469
|
var USER_AGENT = `blockrun-ts/${SDK_VERSION}`;
|
|
3968
4470
|
|
|
3969
4471
|
// src/client.ts
|
|
@@ -8617,6 +9119,68 @@ async function getOrCreateSolanaWallet() {
|
|
|
8617
9119
|
var SOLANA_API_URL = "https://sol.blockrun.ai/api";
|
|
8618
9120
|
var DEFAULT_MAX_TOKENS2 = 1024;
|
|
8619
9121
|
var DEFAULT_TIMEOUT14 = 6e4;
|
|
9122
|
+
var STALE_BLOCKHASH_RETRY_BACKOFFS_MS = [500, 2e3];
|
|
9123
|
+
var MAX_PAYMENT_FAILURE_BYTES = 64 * 1024;
|
|
9124
|
+
var SafeStaleBlockhashError = class extends PaymentError {
|
|
9125
|
+
constructor() {
|
|
9126
|
+
super("Payment verification used an expired Solana blockhash; retrying with a fresh quote.");
|
|
9127
|
+
this.name = "SafeStaleBlockhashError";
|
|
9128
|
+
}
|
|
9129
|
+
};
|
|
9130
|
+
function normalizePaymentSignal(value) {
|
|
9131
|
+
return typeof value === "string" ? value.toLowerCase().replace(/[_\-\s:]/g, "") : "";
|
|
9132
|
+
}
|
|
9133
|
+
async function readPaymentFailureBody(response) {
|
|
9134
|
+
const reader = response.body?.getReader();
|
|
9135
|
+
if (!reader) return "";
|
|
9136
|
+
const decoder = new TextDecoder();
|
|
9137
|
+
let total = 0;
|
|
9138
|
+
let text = "";
|
|
9139
|
+
try {
|
|
9140
|
+
for (; ; ) {
|
|
9141
|
+
const { done, value } = await reader.read();
|
|
9142
|
+
if (done) return text + decoder.decode();
|
|
9143
|
+
total += value.byteLength;
|
|
9144
|
+
if (total > MAX_PAYMENT_FAILURE_BYTES) {
|
|
9145
|
+
void reader.cancel();
|
|
9146
|
+
return null;
|
|
9147
|
+
}
|
|
9148
|
+
text += decoder.decode(value, { stream: true });
|
|
9149
|
+
}
|
|
9150
|
+
} catch {
|
|
9151
|
+
return null;
|
|
9152
|
+
}
|
|
9153
|
+
}
|
|
9154
|
+
async function isSafeStaleBlockhashResponse(response) {
|
|
9155
|
+
const length = Number(response.headers.get("content-length") || "0");
|
|
9156
|
+
if (Number.isFinite(length) && length > MAX_PAYMENT_FAILURE_BYTES) return false;
|
|
9157
|
+
const text = await readPaymentFailureBody(response);
|
|
9158
|
+
if (text === null) return false;
|
|
9159
|
+
let body;
|
|
9160
|
+
try {
|
|
9161
|
+
const parsed = JSON.parse(text);
|
|
9162
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
|
|
9163
|
+
body = parsed;
|
|
9164
|
+
} catch {
|
|
9165
|
+
return false;
|
|
9166
|
+
}
|
|
9167
|
+
const nested = body.error && typeof body.error === "object" && !Array.isArray(body.error) ? body.error : void 0;
|
|
9168
|
+
const errorLabel = typeof body.error === "string" ? body.error : "";
|
|
9169
|
+
const code = normalizePaymentSignal(body.code ?? nested?.code);
|
|
9170
|
+
const reason = normalizePaymentSignal(body.reason);
|
|
9171
|
+
const detail = normalizePaymentSignal(body.invalidMessage);
|
|
9172
|
+
const message = normalizePaymentSignal(nested?.message ?? body.message);
|
|
9173
|
+
const label = normalizePaymentSignal(errorLabel);
|
|
9174
|
+
if (code.includes("settlementfailed") || label.includes("settlementfailed") || message.includes("settlementfailed")) return false;
|
|
9175
|
+
const verifyPhase = code === "paymentinvalid" || label.includes("verificationfailed") || message.includes("verificationfailed");
|
|
9176
|
+
if (!verifyPhase) return false;
|
|
9177
|
+
return code === "paymentblockhashstale" || detail.includes("blockhashnotfound") || detail.includes("blockheightexceeded") || reason === "expiredsignature" || message.includes("expiredsignature");
|
|
9178
|
+
}
|
|
9179
|
+
async function waitForStaleRetry(attempt) {
|
|
9180
|
+
await new Promise(
|
|
9181
|
+
(resolve) => setTimeout(resolve, STALE_BLOCKHASH_RETRY_BACKOFFS_MS[attempt])
|
|
9182
|
+
);
|
|
9183
|
+
}
|
|
8620
9184
|
var DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc";
|
|
8621
9185
|
function resolveRpcConfig(rpcUrl, rpcHeaders) {
|
|
8622
9186
|
const env = typeof process !== "undefined" && process.env ? process.env : {};
|
|
@@ -9063,26 +9627,34 @@ var SolanaLLMClient = class {
|
|
|
9063
9627
|
}
|
|
9064
9628
|
async requestWithPayment(endpoint, body) {
|
|
9065
9629
|
const url = `${this.apiUrl}${endpoint}`;
|
|
9066
|
-
|
|
9067
|
-
|
|
9068
|
-
|
|
9069
|
-
|
|
9070
|
-
|
|
9071
|
-
|
|
9072
|
-
|
|
9073
|
-
|
|
9074
|
-
|
|
9075
|
-
|
|
9076
|
-
|
|
9077
|
-
|
|
9078
|
-
|
|
9079
|
-
|
|
9630
|
+
for (let staleRetries = 0; ; ) {
|
|
9631
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9632
|
+
method: "POST",
|
|
9633
|
+
headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
|
|
9634
|
+
body: JSON.stringify(body)
|
|
9635
|
+
});
|
|
9636
|
+
if (response.status === 402) {
|
|
9637
|
+
try {
|
|
9638
|
+
return await this.handlePaymentAndRetry(url, body, response, staleRetries > 0);
|
|
9639
|
+
} catch (error) {
|
|
9640
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9641
|
+
await waitForStaleRetry(staleRetries++);
|
|
9642
|
+
continue;
|
|
9643
|
+
}
|
|
9080
9644
|
}
|
|
9081
|
-
|
|
9645
|
+
if (!response.ok) {
|
|
9646
|
+
let errorBody;
|
|
9647
|
+
try {
|
|
9648
|
+
errorBody = await response.json();
|
|
9649
|
+
} catch {
|
|
9650
|
+
errorBody = { error: "Request failed" };
|
|
9651
|
+
}
|
|
9652
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9653
|
+
}
|
|
9654
|
+
return response.json();
|
|
9082
9655
|
}
|
|
9083
|
-
return response.json();
|
|
9084
9656
|
}
|
|
9085
|
-
async handlePaymentAndRetry(url, body, response) {
|
|
9657
|
+
async handlePaymentAndRetry(url, body, response, forceFreshBlockhash = false) {
|
|
9086
9658
|
let paymentHeader = response.headers.get("payment-required");
|
|
9087
9659
|
if (!paymentHeader) {
|
|
9088
9660
|
try {
|
|
@@ -9124,7 +9696,8 @@ var SolanaLLMClient = class {
|
|
|
9124
9696
|
extra: details.extra,
|
|
9125
9697
|
extensions,
|
|
9126
9698
|
rpcUrl: this.rpcUrl,
|
|
9127
|
-
rpcHeaders: this.rpcHeaders
|
|
9699
|
+
rpcHeaders: this.rpcHeaders,
|
|
9700
|
+
forceFreshBlockhash
|
|
9128
9701
|
}
|
|
9129
9702
|
);
|
|
9130
9703
|
const retryResponse = await this.fetchWithTimeout(url, {
|
|
@@ -9137,6 +9710,9 @@ var SolanaLLMClient = class {
|
|
|
9137
9710
|
body: JSON.stringify(body)
|
|
9138
9711
|
});
|
|
9139
9712
|
if (retryResponse.status === 402) {
|
|
9713
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9714
|
+
throw new SafeStaleBlockhashError();
|
|
9715
|
+
}
|
|
9140
9716
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9141
9717
|
}
|
|
9142
9718
|
if (!retryResponse.ok) {
|
|
@@ -9155,26 +9731,34 @@ var SolanaLLMClient = class {
|
|
|
9155
9731
|
}
|
|
9156
9732
|
async requestWithPaymentRaw(endpoint, body) {
|
|
9157
9733
|
const url = `${this.apiUrl}${endpoint}`;
|
|
9158
|
-
|
|
9159
|
-
|
|
9160
|
-
|
|
9161
|
-
|
|
9162
|
-
|
|
9163
|
-
|
|
9164
|
-
|
|
9165
|
-
|
|
9166
|
-
|
|
9167
|
-
|
|
9168
|
-
|
|
9169
|
-
|
|
9170
|
-
|
|
9171
|
-
|
|
9734
|
+
for (let staleRetries = 0; ; ) {
|
|
9735
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9736
|
+
method: "POST",
|
|
9737
|
+
headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
|
|
9738
|
+
body: JSON.stringify(body)
|
|
9739
|
+
});
|
|
9740
|
+
if (response.status === 402) {
|
|
9741
|
+
try {
|
|
9742
|
+
return await this.handlePaymentAndRetryRaw(url, body, response, staleRetries > 0);
|
|
9743
|
+
} catch (error) {
|
|
9744
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9745
|
+
await waitForStaleRetry(staleRetries++);
|
|
9746
|
+
continue;
|
|
9747
|
+
}
|
|
9172
9748
|
}
|
|
9173
|
-
|
|
9749
|
+
if (!response.ok) {
|
|
9750
|
+
let errorBody;
|
|
9751
|
+
try {
|
|
9752
|
+
errorBody = await response.json();
|
|
9753
|
+
} catch {
|
|
9754
|
+
errorBody = { error: "Request failed" };
|
|
9755
|
+
}
|
|
9756
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9757
|
+
}
|
|
9758
|
+
return response.json();
|
|
9174
9759
|
}
|
|
9175
|
-
return response.json();
|
|
9176
9760
|
}
|
|
9177
|
-
async handlePaymentAndRetryRaw(url, body, response) {
|
|
9761
|
+
async handlePaymentAndRetryRaw(url, body, response, forceFreshBlockhash = false) {
|
|
9178
9762
|
let paymentHeader = response.headers.get("payment-required");
|
|
9179
9763
|
if (!paymentHeader) {
|
|
9180
9764
|
try {
|
|
@@ -9216,7 +9800,8 @@ var SolanaLLMClient = class {
|
|
|
9216
9800
|
extra: details.extra,
|
|
9217
9801
|
extensions,
|
|
9218
9802
|
rpcUrl: this.rpcUrl,
|
|
9219
|
-
rpcHeaders: this.rpcHeaders
|
|
9803
|
+
rpcHeaders: this.rpcHeaders,
|
|
9804
|
+
forceFreshBlockhash
|
|
9220
9805
|
}
|
|
9221
9806
|
);
|
|
9222
9807
|
const retryResponse = await this.fetchWithTimeout(url, {
|
|
@@ -9229,6 +9814,9 @@ var SolanaLLMClient = class {
|
|
|
9229
9814
|
body: JSON.stringify(body)
|
|
9230
9815
|
});
|
|
9231
9816
|
if (retryResponse.status === 402) {
|
|
9817
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9818
|
+
throw new SafeStaleBlockhashError();
|
|
9819
|
+
}
|
|
9232
9820
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9233
9821
|
}
|
|
9234
9822
|
if (!retryResponse.ok) {
|
|
@@ -9248,25 +9836,33 @@ var SolanaLLMClient = class {
|
|
|
9248
9836
|
async getWithPaymentRaw(endpoint, params) {
|
|
9249
9837
|
const query = params ? "?" + new URLSearchParams(params).toString() : "";
|
|
9250
9838
|
const url = `${this.apiUrl}${endpoint}${query}`;
|
|
9251
|
-
|
|
9252
|
-
|
|
9253
|
-
|
|
9254
|
-
|
|
9255
|
-
|
|
9256
|
-
|
|
9257
|
-
|
|
9258
|
-
|
|
9259
|
-
|
|
9260
|
-
|
|
9261
|
-
|
|
9262
|
-
|
|
9263
|
-
|
|
9839
|
+
for (let staleRetries = 0; ; ) {
|
|
9840
|
+
const response = await this.fetchWithTimeout(url, {
|
|
9841
|
+
method: "GET",
|
|
9842
|
+
headers: { "User-Agent": USER_AGENT }
|
|
9843
|
+
});
|
|
9844
|
+
if (response.status === 402) {
|
|
9845
|
+
try {
|
|
9846
|
+
return await this.handleGetPaymentAndRetryRaw(url, endpoint, params, response, staleRetries > 0);
|
|
9847
|
+
} catch (error) {
|
|
9848
|
+
if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
|
|
9849
|
+
await waitForStaleRetry(staleRetries++);
|
|
9850
|
+
continue;
|
|
9851
|
+
}
|
|
9264
9852
|
}
|
|
9265
|
-
|
|
9853
|
+
if (!response.ok) {
|
|
9854
|
+
let errorBody;
|
|
9855
|
+
try {
|
|
9856
|
+
errorBody = await response.json();
|
|
9857
|
+
} catch {
|
|
9858
|
+
errorBody = { error: "Request failed" };
|
|
9859
|
+
}
|
|
9860
|
+
throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
|
|
9861
|
+
}
|
|
9862
|
+
return response.json();
|
|
9266
9863
|
}
|
|
9267
|
-
return response.json();
|
|
9268
9864
|
}
|
|
9269
|
-
async handleGetPaymentAndRetryRaw(url, endpoint, params, response) {
|
|
9865
|
+
async handleGetPaymentAndRetryRaw(url, endpoint, params, response, forceFreshBlockhash = false) {
|
|
9270
9866
|
let paymentHeader = response.headers.get("payment-required");
|
|
9271
9867
|
if (!paymentHeader) {
|
|
9272
9868
|
try {
|
|
@@ -9308,7 +9904,8 @@ var SolanaLLMClient = class {
|
|
|
9308
9904
|
extra: details.extra,
|
|
9309
9905
|
extensions,
|
|
9310
9906
|
rpcUrl: this.rpcUrl,
|
|
9311
|
-
rpcHeaders: this.rpcHeaders
|
|
9907
|
+
rpcHeaders: this.rpcHeaders,
|
|
9908
|
+
forceFreshBlockhash
|
|
9312
9909
|
}
|
|
9313
9910
|
);
|
|
9314
9911
|
const query = params ? "?" + new URLSearchParams(params).toString() : "";
|
|
@@ -9321,6 +9918,9 @@ var SolanaLLMClient = class {
|
|
|
9321
9918
|
}
|
|
9322
9919
|
});
|
|
9323
9920
|
if (retryResponse.status === 402) {
|
|
9921
|
+
if (await isSafeStaleBlockhashResponse(retryResponse)) {
|
|
9922
|
+
throw new SafeStaleBlockhashError();
|
|
9923
|
+
}
|
|
9324
9924
|
throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
|
|
9325
9925
|
}
|
|
9326
9926
|
if (!retryResponse.ok) {
|