@blockrun/llm 3.13.1 → 3.13.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -145,7 +145,7 @@ var APIError = class extends BlockrunError {
145
145
  }
146
146
  };
147
147
 
148
- // node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+d4308049348e1_gni7wlazdyzf7iqax5nhqdpoqi/node_modules/@blockrun/router-core/dist/index.js
148
+ // node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+5d911879d7f1e_fbldkf3f3jmlwtwu53ftq2mj24/node_modules/@blockrun/router-core/dist/index.js
149
149
  function scoreTokenCount(estimatedTokens, thresholds) {
150
150
  if (estimatedTokens < thresholds.simple) {
151
151
  return { name: "tokenCount", score: -1, signal: `short (${estimatedTokens} tokens)` };
@@ -479,6 +479,21 @@ function applyPromotions(tierConfigs, promotions, profile, now = /* @__PURE__ */
479
479
  }
480
480
  return result;
481
481
  }
482
+ function applyUnavailableModels(tierConfigs, unavailableModels) {
483
+ if (!unavailableModels || unavailableModels.length === 0) return tierConfigs;
484
+ const dead = new Set(unavailableModels);
485
+ let result = tierConfigs;
486
+ for (const tier of Object.keys(tierConfigs)) {
487
+ const config = tierConfigs[tier];
488
+ const alive = [config.primary, ...config.fallback].filter((model) => !dead.has(model));
489
+ if (alive.length === 0 || alive[0] === config.primary && alive.length === config.fallback.length + 1) {
490
+ continue;
491
+ }
492
+ if (result === tierConfigs) result = { ...tierConfigs };
493
+ result[tier] = { primary: alive[0], fallback: alive.slice(1) };
494
+ }
495
+ return result;
496
+ }
482
497
  var RulesStrategy = class {
483
498
  name = "rules";
484
499
  route(prompt, systemPrompt, maxOutputTokens, options) {
@@ -530,6 +545,7 @@ ${value.slice(-(scanLimit - prefixLength))}`;
530
545
  profile = useAgenticTiers ? "agentic" : "auto";
531
546
  }
532
547
  tierConfigs = applyPromotions(tierConfigs, config.promotions, profile, options.now);
548
+ tierConfigs = applyUnavailableModels(tierConfigs, options.unavailableModels);
533
549
  const agenticScoreValue = ruleResult.agenticScore;
534
550
  if (estimatedTokens > config.overrides.maxTokensForceComplex) {
535
551
  const decision2 = selectModel(
@@ -603,14 +619,15 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
603
619
  supportsVision: true
604
620
  },
605
621
  "anthropic/claude-haiku-4.5": {
622
+ // override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
606
623
  contextWindow: 2e5,
607
- maxOutputTokens: 8192,
624
+ maxOutputTokens: 64e3,
608
625
  supportsTools: true,
609
626
  supportsVision: true
610
627
  },
611
- "anthropic/claude-opus-4.6": {
612
- contextWindow: 1e6,
613
- maxOutputTokens: 128e3,
628
+ "anthropic/claude-opus-4.5": {
629
+ contextWindow: 2e5,
630
+ maxOutputTokens: 64e3,
614
631
  supportsTools: true,
615
632
  supportsVision: true
616
633
  },
@@ -632,12 +649,19 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
632
649
  supportsTools: true,
633
650
  supportsVision: true
634
651
  },
635
- "anthropic/claude-sonnet-4.6": {
652
+ "anthropic/claude-sonnet-4.5": {
636
653
  contextWindow: 2e5,
637
654
  maxOutputTokens: 64e3,
638
655
  supportsTools: true,
639
656
  supportsVision: true
640
657
  },
658
+ "anthropic/claude-sonnet-4.6": {
659
+ // override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
660
+ contextWindow: 1e6,
661
+ maxOutputTokens: 128e3,
662
+ supportsTools: true,
663
+ supportsVision: true
664
+ },
641
665
  "anthropic/claude-sonnet-5": {
642
666
  contextWindow: 1e6,
643
667
  maxOutputTokens: 128e3,
@@ -645,14 +669,14 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
645
669
  supportsVision: true
646
670
  },
647
671
  "deepseek/deepseek-chat": {
648
- contextWindow: 1e6,
649
- maxOutputTokens: 8192,
672
+ contextWindow: 1048576,
673
+ maxOutputTokens: 65536,
650
674
  supportsTools: true,
651
675
  supportsVision: false
652
676
  },
653
677
  "deepseek/deepseek-reasoner": {
654
- contextWindow: 1e6,
655
- maxOutputTokens: 8192,
678
+ contextWindow: 1048576,
679
+ maxOutputTokens: 65536,
656
680
  supportsTools: true,
657
681
  supportsVision: false
658
682
  },
@@ -662,62 +686,38 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
662
686
  supportsTools: true,
663
687
  supportsVision: false
664
688
  },
665
- "free/deepseek-v4-flash": {
666
- contextWindow: 1e6,
667
- maxOutputTokens: 16384,
668
- supportsTools: false,
669
- supportsVision: false
670
- },
671
- "free/gpt-oss-120b": {
672
- contextWindow: 128e3,
673
- maxOutputTokens: 16384,
674
- supportsTools: false,
675
- supportsVision: false
676
- },
677
- "free/gpt-oss-20b": {
678
- contextWindow: 128e3,
679
- maxOutputTokens: 16384,
680
- supportsTools: false,
681
- supportsVision: false
682
- },
683
- "free/seed-oss-36b": {
684
- contextWindow: 131072,
685
- maxOutputTokens: 16384,
686
- supportsTools: false,
687
- supportsVision: false
688
- },
689
689
  "google/gemini-2.5-flash": {
690
- contextWindow: 1e6,
690
+ contextWindow: 1048576,
691
691
  maxOutputTokens: 65536,
692
692
  supportsTools: true,
693
693
  supportsVision: true
694
694
  },
695
695
  "google/gemini-2.5-flash-lite": {
696
- contextWindow: 1e6,
696
+ contextWindow: 1048576,
697
697
  maxOutputTokens: 65536,
698
698
  supportsTools: true,
699
699
  supportsVision: false
700
700
  },
701
701
  "google/gemini-2.5-pro": {
702
- contextWindow: 105e4,
702
+ contextWindow: 1048576,
703
703
  maxOutputTokens: 65536,
704
704
  supportsTools: true,
705
705
  supportsVision: true
706
706
  },
707
707
  "google/gemini-3-flash-preview": {
708
- contextWindow: 1e6,
708
+ contextWindow: 1048576,
709
709
  maxOutputTokens: 65536,
710
- supportsTools: false,
710
+ supportsTools: true,
711
711
  supportsVision: true
712
712
  },
713
713
  "google/gemini-3.1-flash-lite": {
714
- contextWindow: 1e6,
715
- maxOutputTokens: 8192,
714
+ contextWindow: 1048576,
715
+ maxOutputTokens: 65536,
716
716
  supportsTools: true,
717
717
  supportsVision: false
718
718
  },
719
719
  "google/gemini-3.1-pro": {
720
- contextWindow: 105e4,
720
+ contextWindow: 1048576,
721
721
  maxOutputTokens: 65536,
722
722
  supportsTools: true,
723
723
  supportsVision: true
@@ -728,23 +728,29 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
728
728
  supportsTools: true,
729
729
  supportsVision: true
730
730
  },
731
- "moonshot/kimi-k2.5": {
732
- contextWindow: 262144,
733
- maxOutputTokens: 16384,
731
+ "google/gemini-3.5-flash-lite": {
732
+ contextWindow: 1048576,
733
+ maxOutputTokens: 65536,
734
734
  supportsTools: true,
735
- supportsVision: true
735
+ supportsVision: false
736
736
  },
737
- "moonshot/kimi-k2.6": {
738
- contextWindow: 262144,
737
+ "google/gemini-3.6-flash": {
738
+ contextWindow: 1048576,
739
739
  maxOutputTokens: 65536,
740
740
  supportsTools: true,
741
741
  supportsVision: true
742
742
  },
743
- "moonshot/kimi-k2.7": {
744
- contextWindow: 262144,
743
+ "minimax/minimax-m2.7": {
744
+ contextWindow: 204800,
745
+ maxOutputTokens: 16384,
746
+ supportsTools: true,
747
+ supportsVision: false
748
+ },
749
+ "minimax/minimax-m3": {
750
+ contextWindow: 1048576,
745
751
  maxOutputTokens: 65536,
746
752
  supportsTools: true,
747
- supportsVision: true
753
+ supportsVision: false
748
754
  },
749
755
  "moonshot/kimi-k3": {
750
756
  contextWindow: 1048576,
@@ -752,7 +758,63 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
752
758
  supportsTools: true,
753
759
  supportsVision: true
754
760
  },
761
+ "nvidia/mistral-nemotron": {
762
+ // supportsTools: gateway unavailable at probe time — fails closed
763
+ contextWindow: 131072,
764
+ maxOutputTokens: 16384,
765
+ supportsTools: false,
766
+ supportsVision: false
767
+ },
768
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
769
+ contextWindow: 256e3,
770
+ maxOutputTokens: 16384,
771
+ supportsTools: false,
772
+ supportsVision: true
773
+ },
774
+ "nvidia/nemotron-nano-12b-v2-vl": {
775
+ // supportsTools: gateway unavailable at probe time — fails closed
776
+ contextWindow: 131072,
777
+ maxOutputTokens: 16384,
778
+ supportsTools: false,
779
+ supportsVision: true
780
+ },
781
+ "nvidia/nemotron-nano-9b-v2": {
782
+ contextWindow: 131072,
783
+ maxOutputTokens: 16384,
784
+ supportsTools: false,
785
+ supportsVision: false
786
+ },
787
+ "nvidia/step-3.7-flash": {
788
+ contextWindow: 131072,
789
+ maxOutputTokens: 16384,
790
+ supportsTools: false,
791
+ supportsVision: false
792
+ },
793
+ "openai/chat-latest": {
794
+ contextWindow: 128e3,
795
+ maxOutputTokens: 128e3,
796
+ supportsTools: true,
797
+ supportsVision: true
798
+ },
755
799
  "openai/gpt-4.1": {
800
+ contextWindow: 128e3,
801
+ maxOutputTokens: 32768,
802
+ supportsTools: true,
803
+ supportsVision: true
804
+ },
805
+ "openai/gpt-4.1-mini": {
806
+ contextWindow: 128e3,
807
+ maxOutputTokens: 32768,
808
+ supportsTools: true,
809
+ supportsVision: false
810
+ },
811
+ "openai/gpt-4.1-nano": {
812
+ contextWindow: 128e3,
813
+ maxOutputTokens: 32768,
814
+ supportsTools: true,
815
+ supportsVision: false
816
+ },
817
+ "openai/gpt-4o": {
756
818
  contextWindow: 128e3,
757
819
  maxOutputTokens: 16384,
758
820
  supportsTools: true,
@@ -766,17 +828,44 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
766
828
  },
767
829
  "openai/gpt-5-mini": {
768
830
  contextWindow: 2e5,
769
- maxOutputTokens: 65536,
831
+ maxOutputTokens: 128e3,
770
832
  supportsTools: true,
771
833
  supportsVision: false
772
834
  },
835
+ "openai/gpt-5.2": {
836
+ contextWindow: 4e5,
837
+ maxOutputTokens: 128e3,
838
+ supportsTools: true,
839
+ supportsVision: true
840
+ },
841
+ "openai/gpt-5.2-pro": {
842
+ // supportsTools: not probed — fails closed
843
+ contextWindow: 4e5,
844
+ maxOutputTokens: 128e3,
845
+ supportsTools: false,
846
+ supportsVision: true
847
+ },
848
+ "openai/gpt-5.3": {
849
+ // supportsTools: gateway unavailable at probe time — fails closed
850
+ contextWindow: 128e3,
851
+ maxOutputTokens: 128e3,
852
+ supportsTools: false,
853
+ supportsVision: true
854
+ },
773
855
  "openai/gpt-5.3-codex": {
856
+ // supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe measured an incident, not the model. Codex's function calling is established by the 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot claiming the model cannot call tools.
774
857
  contextWindow: 4e5,
775
858
  maxOutputTokens: 128e3,
776
859
  supportsTools: true,
777
860
  supportsVision: false
778
861
  },
779
862
  "openai/gpt-5.4": {
863
+ contextWindow: 105e4,
864
+ maxOutputTokens: 128e3,
865
+ supportsTools: true,
866
+ supportsVision: true
867
+ },
868
+ "openai/gpt-5.4-mini": {
780
869
  contextWindow: 4e5,
781
870
  maxOutputTokens: 128e3,
782
871
  supportsTools: true,
@@ -784,30 +873,92 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
784
873
  },
785
874
  "openai/gpt-5.4-nano": {
786
875
  contextWindow: 105e4,
787
- maxOutputTokens: 32768,
876
+ maxOutputTokens: 128e3,
788
877
  supportsTools: true,
789
878
  supportsVision: false
790
879
  },
880
+ "openai/gpt-5.4-pro": {
881
+ // supportsTools: not probed — fails closed
882
+ contextWindow: 105e4,
883
+ maxOutputTokens: 128e3,
884
+ supportsTools: false,
885
+ supportsVision: true
886
+ },
791
887
  "openai/gpt-5.5": {
792
888
  contextWindow: 105e4,
793
889
  maxOutputTokens: 128e3,
794
890
  supportsTools: true,
795
891
  supportsVision: true
796
892
  },
893
+ "openai/gpt-5.5-pro": {
894
+ // supportsTools: not probed — fails closed
895
+ contextWindow: 105e4,
896
+ maxOutputTokens: 128e3,
897
+ supportsTools: false,
898
+ supportsVision: true
899
+ },
900
+ "openai/gpt-5.6-luna": {
901
+ contextWindow: 105e4,
902
+ maxOutputTokens: 128e3,
903
+ supportsTools: true,
904
+ supportsVision: true
905
+ },
906
+ "openai/gpt-5.6-luna-pro": {
907
+ contextWindow: 105e4,
908
+ maxOutputTokens: 128e3,
909
+ supportsTools: false,
910
+ supportsVision: true
911
+ },
912
+ "openai/gpt-5.6-sol": {
913
+ contextWindow: 105e4,
914
+ maxOutputTokens: 128e3,
915
+ supportsTools: true,
916
+ supportsVision: true
917
+ },
918
+ "openai/gpt-5.6-sol-pro": {
919
+ contextWindow: 105e4,
920
+ maxOutputTokens: 128e3,
921
+ supportsTools: true,
922
+ supportsVision: true
923
+ },
797
924
  "openai/gpt-5.6-terra": {
798
925
  contextWindow: 105e4,
799
926
  maxOutputTokens: 128e3,
800
927
  supportsTools: true,
801
928
  supportsVision: true
802
929
  },
930
+ "openai/gpt-5.6-terra-pro": {
931
+ contextWindow: 105e4,
932
+ maxOutputTokens: 128e3,
933
+ supportsTools: true,
934
+ supportsVision: true
935
+ },
936
+ "openai/o1": {
937
+ contextWindow: 2e5,
938
+ maxOutputTokens: 1e5,
939
+ supportsTools: true,
940
+ supportsVision: false
941
+ },
803
942
  "openai/o3": {
804
943
  contextWindow: 2e5,
805
944
  maxOutputTokens: 1e5,
806
945
  supportsTools: true,
807
946
  supportsVision: false
808
947
  },
948
+ "openai/o3-mini": {
949
+ contextWindow: 128e3,
950
+ maxOutputTokens: 1e5,
951
+ supportsTools: true,
952
+ supportsVision: false
953
+ },
809
954
  "openai/o4-mini": {
810
955
  contextWindow: 128e3,
956
+ maxOutputTokens: 1e5,
957
+ supportsTools: true,
958
+ supportsVision: false
959
+ },
960
+ "qwen/qwen3.7-flash": {
961
+ contextWindow: 1e6,
811
962
  maxOutputTokens: 65536,
812
963
  supportsTools: true,
813
964
  supportsVision: false
@@ -818,299 +969,605 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
818
969
  supportsTools: true,
819
970
  supportsVision: false
820
971
  },
821
- "xai/grok-3-mini": {
822
- contextWindow: 131072,
823
- maxOutputTokens: 16384,
972
+ "qwen/qwen3.7-plus": {
973
+ contextWindow: 1e6,
974
+ maxOutputTokens: 131072,
824
975
  supportsTools: true,
825
976
  supportsVision: false
826
977
  },
827
- "xai/grok-4-0709": {
828
- contextWindow: 131072,
829
- maxOutputTokens: 16384,
978
+ "tencent/hy3": {
979
+ contextWindow: 262144,
980
+ maxOutputTokens: 128e3,
830
981
  supportsTools: true,
831
982
  supportsVision: false
832
983
  },
833
- "xai/grok-4-1-fast-non-reasoning": {
834
- contextWindow: 131072,
984
+ "xai/grok-4.3": {
985
+ contextWindow: 1e6,
835
986
  maxOutputTokens: 16384,
836
987
  supportsTools: true,
837
- supportsVision: false
988
+ supportsVision: true
838
989
  },
839
- "xai/grok-4-1-fast-reasoning": {
840
- contextWindow: 131072,
990
+ "xai/grok-4.5": {
991
+ contextWindow: 5e5,
841
992
  maxOutputTokens: 16384,
842
993
  supportsTools: true,
843
- supportsVision: false
994
+ supportsVision: true
844
995
  },
845
- "xai/grok-4-fast-non-reasoning": {
846
- contextWindow: 131072,
996
+ "xai/grok-build-0.1": {
997
+ contextWindow: 256e3,
847
998
  maxOutputTokens: 16384,
848
999
  supportsTools: true,
849
1000
  supportsVision: false
850
1001
  },
851
- "xai/grok-4-fast-reasoning": {
852
- contextWindow: 131072,
853
- maxOutputTokens: 16384,
854
- supportsTools: true,
855
- supportsVision: false
1002
+ "xiaomi/mimo-v2.5-pro": {
1003
+ contextWindow: 1048576,
1004
+ maxOutputTokens: 131072,
1005
+ supportsTools: true,
1006
+ supportsVision: false
1007
+ },
1008
+ "zai/glm-5": {
1009
+ contextWindow: 2e5,
1010
+ maxOutputTokens: 128e3,
1011
+ supportsTools: true,
1012
+ supportsVision: false
1013
+ },
1014
+ "zai/glm-5-turbo": {
1015
+ contextWindow: 2e5,
1016
+ maxOutputTokens: 128e3,
1017
+ supportsTools: true,
1018
+ supportsVision: false
1019
+ },
1020
+ "zai/glm-5.1": {
1021
+ contextWindow: 2e5,
1022
+ maxOutputTokens: 128e3,
1023
+ supportsTools: true,
1024
+ supportsVision: false
1025
+ },
1026
+ "zai/glm-5.2": {
1027
+ contextWindow: 1e6,
1028
+ maxOutputTokens: 131072,
1029
+ supportsTools: true,
1030
+ supportsVision: false
1031
+ },
1032
+ "zai/glm-5.3": {
1033
+ contextWindow: 1e6,
1034
+ maxOutputTokens: 131072,
1035
+ supportsTools: true,
1036
+ supportsVision: false
1037
+ },
1038
+ "zai/glm-5.3-flash": {
1039
+ contextWindow: 1e6,
1040
+ maxOutputTokens: 131072,
1041
+ supportsTools: true,
1042
+ supportsVision: true
1043
+ }
1044
+ });
1045
+ var model_profiles_generated_default = {
1046
+ "anthropic/claude-fable-5": {
1047
+ measuredAt: "2026-08-29T16:51:33Z",
1048
+ latencyMs: 9298.5,
1049
+ p95LatencyMs: 9873.4,
1050
+ outputTokensPerSecond: 55.17,
1051
+ errorRate: 0,
1052
+ samples: 3
1053
+ },
1054
+ "anthropic/claude-haiku-4.5": {
1055
+ measuredAt: "2026-08-29T16:51:33Z",
1056
+ latencyMs: 3157.4,
1057
+ p95LatencyMs: 3170.7,
1058
+ outputTokensPerSecond: 162.16,
1059
+ errorRate: 0,
1060
+ samples: 3
1061
+ },
1062
+ "anthropic/claude-opus-4.5": {
1063
+ measuredAt: "2026-08-29T16:51:33Z",
1064
+ latencyMs: 6497.7,
1065
+ p95LatencyMs: 6953.7,
1066
+ outputTokensPerSecond: 78.99,
1067
+ errorRate: 0,
1068
+ samples: 3
1069
+ },
1070
+ "anthropic/claude-opus-4.7": {
1071
+ measuredAt: "2026-08-29T16:51:33Z",
1072
+ latencyMs: 5316.5,
1073
+ p95LatencyMs: 6121.5,
1074
+ outputTokensPerSecond: 97.34,
1075
+ errorRate: 0,
1076
+ samples: 3
1077
+ },
1078
+ "anthropic/claude-opus-4.8": {
1079
+ measuredAt: "2026-08-29T16:51:33Z",
1080
+ latencyMs: 6216.1,
1081
+ p95LatencyMs: 6847.7,
1082
+ outputTokensPerSecond: 82.81,
1083
+ errorRate: 0,
1084
+ samples: 3
1085
+ },
1086
+ "anthropic/claude-opus-5": {
1087
+ measuredAt: "2026-08-29T16:51:33Z",
1088
+ latencyMs: 7309,
1089
+ p95LatencyMs: 7745.2,
1090
+ outputTokensPerSecond: 70.17,
1091
+ errorRate: 0,
1092
+ samples: 3
1093
+ },
1094
+ "anthropic/claude-sonnet-4.5": {
1095
+ measuredAt: "2026-08-29T16:51:33Z",
1096
+ latencyMs: 6330.4,
1097
+ p95LatencyMs: 6631.6,
1098
+ outputTokensPerSecond: 81.03,
1099
+ errorRate: 0,
1100
+ samples: 3
1101
+ },
1102
+ "anthropic/claude-sonnet-4.6": {
1103
+ measuredAt: "2026-08-29T16:51:33Z",
1104
+ latencyMs: 6508,
1105
+ p95LatencyMs: 6698.3,
1106
+ outputTokensPerSecond: 78.6,
1107
+ errorRate: 0,
1108
+ samples: 3
1109
+ },
1110
+ "anthropic/claude-sonnet-5": {
1111
+ measuredAt: "2026-08-29T16:51:33Z",
1112
+ latencyMs: 6165.4,
1113
+ p95LatencyMs: 6582.9,
1114
+ outputTokensPerSecond: 83.62,
1115
+ errorRate: 0,
1116
+ samples: 3
1117
+ },
1118
+ "deepseek/deepseek-chat": {
1119
+ measuredAt: "2026-08-29T16:51:33Z",
1120
+ latencyMs: 4351.4,
1121
+ p95LatencyMs: 4543.7,
1122
+ outputTokensPerSecond: 117.78,
1123
+ errorRate: 0,
1124
+ samples: 3
1125
+ },
1126
+ "deepseek/deepseek-reasoner": {
1127
+ measuredAt: "2026-08-29T16:51:33Z",
1128
+ latencyMs: 5201.2,
1129
+ p95LatencyMs: 6079.6,
1130
+ outputTokensPerSecond: 99.77,
1131
+ errorRate: 0,
1132
+ samples: 3
1133
+ },
1134
+ "deepseek/deepseek-v4-pro": {
1135
+ measuredAt: "2026-08-29T16:51:33Z",
1136
+ latencyMs: 8781.2,
1137
+ p95LatencyMs: 9881.1,
1138
+ outputTokensPerSecond: 58.98,
1139
+ errorRate: 0,
1140
+ samples: 3
1141
+ },
1142
+ "google/gemini-2.5-flash": {
1143
+ measuredAt: "2026-08-29T16:51:33Z",
1144
+ latencyMs: 5416.4,
1145
+ p95LatencyMs: 6442.8,
1146
+ outputTokensPerSecond: 213.07,
1147
+ errorRate: 0,
1148
+ samples: 3
1149
+ },
1150
+ "google/gemini-2.5-flash-lite": {
1151
+ measuredAt: "2026-08-29T16:51:33Z",
1152
+ latencyMs: 5002.6,
1153
+ p95LatencyMs: 5780.3,
1154
+ outputTokensPerSecond: 408.43,
1155
+ errorRate: 0,
1156
+ samples: 3
1157
+ },
1158
+ "google/gemini-2.5-pro": {
1159
+ measuredAt: "2026-08-29T16:51:33Z",
1160
+ latencyMs: 28169.5,
1161
+ p95LatencyMs: 29491.4,
1162
+ outputTokensPerSecond: 147.3,
1163
+ errorRate: 0,
1164
+ samples: 3
1165
+ },
1166
+ "google/gemini-3-flash-preview": {
1167
+ measuredAt: "2026-08-29T16:51:33Z",
1168
+ latencyMs: 4717.1,
1169
+ p95LatencyMs: 5037.1,
1170
+ outputTokensPerSecond: 198.71,
1171
+ errorRate: 0,
1172
+ samples: 3
1173
+ },
1174
+ "google/gemini-3.1-flash-lite": {
1175
+ measuredAt: "2026-08-29T16:51:33Z",
1176
+ latencyMs: 2855.8,
1177
+ p95LatencyMs: 3172.7,
1178
+ outputTokensPerSecond: 286.91,
1179
+ errorRate: 0,
1180
+ samples: 3
1181
+ },
1182
+ "google/gemini-3.1-pro": {
1183
+ measuredAt: "2026-08-29T16:59:54Z",
1184
+ latencyMs: 24194.1,
1185
+ p95LatencyMs: 27269.6,
1186
+ outputTokensPerSecond: 109.47,
1187
+ errorRate: 0,
1188
+ samples: 3
1189
+ },
1190
+ "google/gemini-3.5-flash": {
1191
+ measuredAt: "2026-08-29T16:51:33Z",
1192
+ latencyMs: 5320.6,
1193
+ p95LatencyMs: 5429.8,
1194
+ outputTokensPerSecond: 226.21,
1195
+ errorRate: 0,
1196
+ samples: 3
1197
+ },
1198
+ "google/gemini-3.5-flash-lite": {
1199
+ measuredAt: "2026-08-29T16:51:33Z",
1200
+ latencyMs: 3515.8,
1201
+ p95LatencyMs: 4363.4,
1202
+ outputTokensPerSecond: 248.9,
1203
+ errorRate: 0,
1204
+ samples: 3
1205
+ },
1206
+ "google/gemini-3.6-flash": {
1207
+ measuredAt: "2026-08-29T16:51:33Z",
1208
+ latencyMs: 13020,
1209
+ p95LatencyMs: 15383.1,
1210
+ outputTokensPerSecond: 187.87,
1211
+ errorRate: 0,
1212
+ samples: 3
1213
+ },
1214
+ "minimax/minimax-m2.7": {
1215
+ measuredAt: "2026-08-29T16:51:33Z",
1216
+ latencyMs: 8761.1,
1217
+ p95LatencyMs: 10199.3,
1218
+ outputTokensPerSecond: 59.18,
1219
+ errorRate: 0,
1220
+ samples: 3
1221
+ },
1222
+ "minimax/minimax-m3": {
1223
+ measuredAt: "2026-08-29T16:51:33Z",
1224
+ latencyMs: 11101.9,
1225
+ p95LatencyMs: 26087.1,
1226
+ outputTokensPerSecond: 101.12,
1227
+ errorRate: 0,
1228
+ samples: 3
1229
+ },
1230
+ "moonshot/kimi-k3": {
1231
+ measuredAt: "2026-08-29T16:51:33Z",
1232
+ latencyMs: 24498.9,
1233
+ p95LatencyMs: 40365.3,
1234
+ outputTokensPerSecond: 25.11,
1235
+ errorRate: 0,
1236
+ samples: 3
1237
+ },
1238
+ "nvidia/mistral-nemotron": {
1239
+ measuredAt: "2026-08-29T16:51:33Z",
1240
+ latencyMs: 7349.6,
1241
+ p95LatencyMs: 9932.3,
1242
+ outputTokensPerSecond: 79.48,
1243
+ errorRate: 0.3333,
1244
+ samples: 3
1245
+ },
1246
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
1247
+ measuredAt: "2026-08-29T16:59:54Z",
1248
+ latencyMs: 9324.6,
1249
+ p95LatencyMs: 12992,
1250
+ outputTokensPerSecond: 64.96,
1251
+ errorRate: 0.3333,
1252
+ samples: 3
1253
+ },
1254
+ "nvidia/nemotron-nano-12b-v2-vl": {
1255
+ measuredAt: "2026-08-29T16:51:33Z",
1256
+ latencyMs: 5846.9,
1257
+ p95LatencyMs: 5846.9,
1258
+ outputTokensPerSecond: 87.57,
1259
+ errorRate: 0.6667,
1260
+ samples: 3
1261
+ },
1262
+ "nvidia/nemotron-nano-9b-v2": {
1263
+ measuredAt: "2026-08-29T16:51:33Z",
1264
+ latencyMs: 5282.5,
1265
+ p95LatencyMs: 5282.5,
1266
+ outputTokensPerSecond: 96.92,
1267
+ errorRate: 0.6667,
1268
+ samples: 3
1269
+ },
1270
+ "nvidia/step-3.7-flash": {
1271
+ measuredAt: "2026-08-29T16:51:33Z",
1272
+ latencyMs: 4617.4,
1273
+ p95LatencyMs: 5237.4,
1274
+ outputTokensPerSecond: 112.92,
1275
+ errorRate: 0.3333,
1276
+ samples: 3
1277
+ },
1278
+ "openai/chat-latest": {
1279
+ measuredAt: "2026-08-29T16:51:33Z",
1280
+ latencyMs: 3690.9,
1281
+ p95LatencyMs: 4344,
1282
+ outputTokensPerSecond: 111.85,
1283
+ errorRate: 0,
1284
+ samples: 3
1285
+ },
1286
+ "openai/gpt-4.1": {
1287
+ measuredAt: "2026-08-29T16:51:33Z",
1288
+ latencyMs: 3527.9,
1289
+ p95LatencyMs: 3831.7,
1290
+ outputTokensPerSecond: 141.27,
1291
+ errorRate: 0,
1292
+ samples: 3
1293
+ },
1294
+ "openai/gpt-4.1-mini": {
1295
+ measuredAt: "2026-08-29T16:51:33Z",
1296
+ latencyMs: 4268.2,
1297
+ p95LatencyMs: 5101.5,
1298
+ outputTokensPerSecond: 103.42,
1299
+ errorRate: 0,
1300
+ samples: 3
856
1301
  },
857
- "xai/grok-4.5": {
858
- contextWindow: 5e5,
859
- maxOutputTokens: 16384,
860
- supportsTools: true,
861
- supportsVision: true
1302
+ "openai/gpt-4.1-nano": {
1303
+ measuredAt: "2026-08-29T16:51:33Z",
1304
+ latencyMs: 3088.3,
1305
+ p95LatencyMs: 3369.2,
1306
+ outputTokensPerSecond: 150.31,
1307
+ errorRate: 0,
1308
+ samples: 3
862
1309
  },
863
- "zai/glm-5.1": {
864
- contextWindow: 2e5,
865
- maxOutputTokens: 128e3,
866
- supportsTools: true,
867
- supportsVision: false
1310
+ "openai/gpt-4o": {
1311
+ measuredAt: "2026-08-29T16:51:33Z",
1312
+ latencyMs: 2995.2,
1313
+ p95LatencyMs: 3174.2,
1314
+ outputTokensPerSecond: 171.32,
1315
+ errorRate: 0,
1316
+ samples: 3
868
1317
  },
869
- "zai/glm-5.2": {
870
- contextWindow: 1e6,
871
- maxOutputTokens: 262144,
872
- supportsTools: true,
873
- supportsVision: false
874
- }
875
- });
876
- var model_profiles_generated_default = {
877
- "openai/gpt-5.5": {
878
- measuredAt: "2026-07-21T10:21:31Z",
879
- latencyMs: 6243.1,
880
- p95LatencyMs: 9865,
881
- outputTokensPerSecond: 12.53,
1318
+ "openai/gpt-4o-mini": {
1319
+ measuredAt: "2026-08-29T16:51:33Z",
1320
+ latencyMs: 4751.5,
1321
+ p95LatencyMs: 4930.4,
1322
+ outputTokensPerSecond: 107.84,
882
1323
  errorRate: 0,
883
1324
  samples: 3
884
1325
  },
885
- "openai/gpt-5.4-pro": {
886
- measuredAt: "2026-07-21T10:21:31Z",
887
- latencyMs: 13015.5,
888
- p95LatencyMs: 23976.4,
889
- outputTokensPerSecond: 6.42,
1326
+ "openai/gpt-5-mini": {
1327
+ measuredAt: "2026-08-29T16:51:33Z",
1328
+ latencyMs: 4558.1,
1329
+ p95LatencyMs: 5081.9,
1330
+ outputTokensPerSecond: 113.25,
890
1331
  errorRate: 0,
891
1332
  samples: 3
892
1333
  },
893
- "openai/gpt-5.4-mini": {
894
- measuredAt: "2026-07-21T10:21:31Z",
895
- latencyMs: 5550,
896
- p95LatencyMs: 6595.7,
897
- outputTokensPerSecond: 11.96,
898
- errorRate: 0.3333,
1334
+ "openai/gpt-5.2": {
1335
+ measuredAt: "2026-08-29T16:51:33Z",
1336
+ latencyMs: 5436.6,
1337
+ p95LatencyMs: 5928.8,
1338
+ outputTokensPerSecond: 95.47,
1339
+ errorRate: 0,
899
1340
  samples: 3
900
1341
  },
901
1342
  "openai/gpt-5.3-codex": {
902
- measuredAt: "2026-07-21T10:21:31Z",
903
- latencyMs: 4617.1,
904
- p95LatencyMs: 5800.7,
905
- outputTokensPerSecond: 12.48,
906
- errorRate: 0,
1343
+ measuredAt: "2026-08-29T16:59:54Z",
1344
+ latencyMs: 15290.4,
1345
+ p95LatencyMs: 15290.4,
1346
+ outputTokensPerSecond: 33.49,
1347
+ errorRate: 0.6667,
907
1348
  samples: 3
908
1349
  },
909
- "anthropic/claude-opus-4.8": {
910
- measuredAt: "2026-07-21T10:21:31Z",
911
- latencyMs: 3915.1,
912
- p95LatencyMs: 6130.8,
913
- outputTokensPerSecond: 16.33,
1350
+ "openai/gpt-5.4": {
1351
+ measuredAt: "2026-08-29T16:51:33Z",
1352
+ latencyMs: 5596,
1353
+ p95LatencyMs: 5919.4,
1354
+ outputTokensPerSecond: 91.67,
914
1355
  errorRate: 0,
915
1356
  samples: 3
916
1357
  },
917
- "anthropic/claude-opus-4.6": {
918
- measuredAt: "2026-07-21T10:21:31Z",
919
- latencyMs: 3765.5,
920
- p95LatencyMs: 4257.2,
921
- outputTokensPerSecond: 14.18,
1358
+ "openai/gpt-5.4-mini": {
1359
+ measuredAt: "2026-08-29T16:51:33Z",
1360
+ latencyMs: 3377.8,
1361
+ p95LatencyMs: 3646.8,
1362
+ outputTokensPerSecond: 138.08,
922
1363
  errorRate: 0,
923
1364
  samples: 3
924
1365
  },
925
- "anthropic/claude-sonnet-4.6": {
926
- measuredAt: "2026-07-21T10:21:31Z",
927
- latencyMs: 3860.6,
928
- p95LatencyMs: 5093.5,
929
- outputTokensPerSecond: 13.85,
1366
+ "openai/gpt-5.4-nano": {
1367
+ measuredAt: "2026-08-29T16:51:33Z",
1368
+ latencyMs: 4040.4,
1369
+ p95LatencyMs: 4205.9,
1370
+ outputTokensPerSecond: 118.52,
930
1371
  errorRate: 0,
931
1372
  samples: 3
932
1373
  },
933
- "anthropic/claude-haiku-4.5": {
934
- measuredAt: "2026-07-21T10:21:31Z",
935
- latencyMs: 2734.9,
936
- p95LatencyMs: 3181.6,
937
- outputTokensPerSecond: 19.58,
1374
+ "openai/gpt-5.5": {
1375
+ measuredAt: "2026-08-29T16:51:33Z",
1376
+ latencyMs: 6367.8,
1377
+ p95LatencyMs: 7330.7,
1378
+ outputTokensPerSecond: 81.29,
938
1379
  errorRate: 0,
939
1380
  samples: 3
940
1381
  },
941
- "google/gemini-3.1-pro": {
942
- measuredAt: "2026-07-21T10:21:31Z",
943
- latencyMs: 13935.7,
944
- p95LatencyMs: 26675.3,
945
- outputTokensPerSecond: 77.47,
1382
+ "openai/gpt-5.6-luna": {
1383
+ measuredAt: "2026-08-29T16:51:33Z",
1384
+ latencyMs: 6064.5,
1385
+ p95LatencyMs: 7347.3,
1386
+ outputTokensPerSecond: 87.93,
946
1387
  errorRate: 0,
947
1388
  samples: 3
948
1389
  },
949
- "google/gemini-3.5-flash": {
950
- measuredAt: "2026-07-21T10:21:31Z",
951
- latencyMs: 4608.7,
952
- p95LatencyMs: 8420.9,
953
- outputTokensPerSecond: 57.88,
1390
+ "openai/gpt-5.6-luna-pro": {
1391
+ measuredAt: "2026-08-29T16:59:54Z",
1392
+ latencyMs: 13914.9,
1393
+ p95LatencyMs: 13914.9,
1394
+ outputTokensPerSecond: 36.79,
1395
+ errorRate: 0.6667,
1396
+ samples: 3
1397
+ },
1398
+ "openai/gpt-5.6-sol": {
1399
+ measuredAt: "2026-08-29T16:51:33Z",
1400
+ latencyMs: 7720.2,
1401
+ p95LatencyMs: 9108.2,
1402
+ outputTokensPerSecond: 67.47,
954
1403
  errorRate: 0,
955
1404
  samples: 3
956
1405
  },
957
- "google/gemini-3.1-flash-lite": {
958
- measuredAt: "2026-07-21T10:21:31Z",
959
- latencyMs: 4619.7,
960
- p95LatencyMs: 9927.1,
961
- outputTokensPerSecond: 42.01,
1406
+ "openai/gpt-5.6-sol-pro": {
1407
+ measuredAt: "2026-08-29T16:51:33Z",
1408
+ latencyMs: 11442.7,
1409
+ p95LatencyMs: 13363.1,
1410
+ outputTokensPerSecond: 148.75,
962
1411
  errorRate: 0,
963
1412
  samples: 3
964
1413
  },
965
- "google/gemini-2.5-flash": {
966
- measuredAt: "2026-07-21T10:21:31Z",
967
- latencyMs: 5506.9,
968
- p95LatencyMs: 11462.5,
969
- outputTokensPerSecond: 65.19,
1414
+ "openai/gpt-5.6-terra": {
1415
+ measuredAt: "2026-08-29T16:51:33Z",
1416
+ latencyMs: 4941,
1417
+ p95LatencyMs: 5095.3,
1418
+ outputTokensPerSecond: 103.69,
970
1419
  errorRate: 0,
971
1420
  samples: 3
972
1421
  },
973
- "deepseek/deepseek-v4-pro": {
974
- measuredAt: "2026-07-21T10:21:31Z",
975
- latencyMs: 6044.8,
976
- p95LatencyMs: 10782.3,
977
- outputTokensPerSecond: 22.47,
1422
+ "openai/gpt-5.6-terra-pro": {
1423
+ measuredAt: "2026-08-29T16:51:33Z",
1424
+ latencyMs: 3574.1,
1425
+ p95LatencyMs: 4126.3,
1426
+ outputTokensPerSecond: 133.59,
978
1427
  errorRate: 0,
979
1428
  samples: 3
980
1429
  },
981
- "deepseek/deepseek-reasoner": {
982
- measuredAt: "2026-07-21T10:21:31Z",
983
- latencyMs: 4111.9,
984
- p95LatencyMs: 5305.7,
985
- outputTokensPerSecond: 16.46,
1430
+ "openai/o1": {
1431
+ measuredAt: "2026-08-29T16:51:33Z",
1432
+ latencyMs: 4324.9,
1433
+ p95LatencyMs: 5838.1,
1434
+ outputTokensPerSecond: 125.86,
986
1435
  errorRate: 0,
987
1436
  samples: 3
988
1437
  },
989
- "deepseek/deepseek-chat": {
990
- measuredAt: "2026-07-21T10:21:31Z",
991
- latencyMs: 2648.6,
992
- p95LatencyMs: 3524.1,
993
- outputTokensPerSecond: 16.73,
1438
+ "openai/o3": {
1439
+ measuredAt: "2026-08-29T16:51:33Z",
1440
+ latencyMs: 5463.4,
1441
+ p95LatencyMs: 5613.1,
1442
+ outputTokensPerSecond: 93.8,
994
1443
  errorRate: 0,
995
1444
  samples: 3
996
1445
  },
997
- "moonshot/kimi-k2.7": {
998
- measuredAt: "2026-07-21T10:21:31Z",
999
- latencyMs: 4295.4,
1000
- p95LatencyMs: 6153.8,
1001
- outputTokensPerSecond: 18.54,
1446
+ "openai/o3-mini": {
1447
+ measuredAt: "2026-08-29T16:51:33Z",
1448
+ latencyMs: 2912.7,
1449
+ p95LatencyMs: 3092.1,
1450
+ outputTokensPerSecond: 176.49,
1002
1451
  errorRate: 0,
1003
1452
  samples: 3
1004
1453
  },
1005
- "qwen/qwen3.7-max": {
1006
- measuredAt: "2026-07-21T10:21:31Z",
1007
- latencyMs: 30729.4,
1008
- p95LatencyMs: 39622,
1009
- outputTokensPerSecond: 36.89,
1010
- errorRate: 0.3333,
1454
+ "openai/o4-mini": {
1455
+ measuredAt: "2026-08-29T16:51:33Z",
1456
+ latencyMs: 4958.7,
1457
+ p95LatencyMs: 5313,
1458
+ outputTokensPerSecond: 103.81,
1459
+ errorRate: 0,
1011
1460
  samples: 3
1012
1461
  },
1013
- "xai/grok-4.3": {
1014
- measuredAt: "2026-07-21T10:21:31Z",
1015
- latencyMs: 6946.1,
1016
- p95LatencyMs: 9495.4,
1017
- outputTokensPerSecond: 65.3,
1462
+ "qwen/qwen3.7-flash": {
1463
+ measuredAt: "2026-08-29T16:51:33Z",
1464
+ latencyMs: 3385.5,
1465
+ p95LatencyMs: 4042.7,
1466
+ outputTokensPerSecond: 153.94,
1018
1467
  errorRate: 0,
1019
1468
  samples: 3
1020
1469
  },
1021
- "xai/grok-4.20-reasoning": {
1022
- measuredAt: "2026-07-21T10:21:31Z",
1023
- latencyMs: 3472.4,
1024
- p95LatencyMs: 5332.4,
1025
- outputTokensPerSecond: 13.27,
1470
+ "qwen/qwen3.7-max": {
1471
+ measuredAt: "2026-08-29T16:51:33Z",
1472
+ latencyMs: 9387.1,
1473
+ p95LatencyMs: 10490.2,
1474
+ outputTokensPerSecond: 54.92,
1026
1475
  errorRate: 0,
1027
1476
  samples: 3
1028
1477
  },
1029
- "xai/grok-4.20-non-reasoning": {
1030
- measuredAt: "2026-07-21T10:21:31Z",
1031
- latencyMs: 5174.4,
1032
- p95LatencyMs: 6081.7,
1033
- outputTokensPerSecond: 10.21,
1034
- errorRate: 0.3333,
1478
+ "qwen/qwen3.7-plus": {
1479
+ measuredAt: "2026-08-29T16:51:33Z",
1480
+ latencyMs: 9766.6,
1481
+ p95LatencyMs: 9798.2,
1482
+ outputTokensPerSecond: 52.42,
1483
+ errorRate: 0,
1035
1484
  samples: 3
1036
1485
  },
1037
- "xai/grok-4-1-fast-reasoning": {
1038
- measuredAt: "2026-07-21T10:21:31Z",
1039
- latencyMs: 13148.2,
1040
- p95LatencyMs: 19104.2,
1041
- outputTokensPerSecond: 4.28,
1486
+ "tencent/hy3": {
1487
+ measuredAt: "2026-08-29T16:51:33Z",
1488
+ latencyMs: 6062.3,
1489
+ p95LatencyMs: 7070.2,
1490
+ outputTokensPerSecond: 87.3,
1042
1491
  errorRate: 0,
1043
1492
  samples: 3
1044
1493
  },
1045
- "minimax/minimax-m3": {
1046
- measuredAt: "2026-07-21T10:21:31Z",
1047
- latencyMs: 3385,
1048
- p95LatencyMs: 4247.2,
1049
- outputTokensPerSecond: 15.16,
1494
+ "xai/grok-4.3": {
1495
+ measuredAt: "2026-08-29T16:51:33Z",
1496
+ latencyMs: 9467.7,
1497
+ p95LatencyMs: 10087.9,
1498
+ outputTokensPerSecond: 48.36,
1050
1499
  errorRate: 0,
1051
1500
  samples: 3
1052
1501
  },
1053
- "minimax/minimax-m2.7": {
1054
- measuredAt: "2026-07-21T10:21:31Z",
1055
- latencyMs: 4596.7,
1056
- p95LatencyMs: 6884.6,
1057
- outputTokensPerSecond: 17.03,
1502
+ "xai/grok-4.5": {
1503
+ measuredAt: "2026-08-29T16:51:33Z",
1504
+ latencyMs: 13564.8,
1505
+ p95LatencyMs: 17351.9,
1506
+ outputTokensPerSecond: 60.71,
1058
1507
  errorRate: 0,
1059
1508
  samples: 3
1060
1509
  },
1061
- "zai/glm-5.2": {
1062
- measuredAt: "2026-07-21T10:21:31Z",
1063
- latencyMs: 4406.3,
1064
- p95LatencyMs: 6139.7,
1065
- outputTokensPerSecond: 10.41,
1510
+ "xai/grok-build-0.1": {
1511
+ measuredAt: "2026-08-29T16:51:33Z",
1512
+ latencyMs: 16394.8,
1513
+ p95LatencyMs: 18035.4,
1514
+ outputTokensPerSecond: 96.86,
1066
1515
  errorRate: 0,
1067
1516
  samples: 3
1068
1517
  },
1069
- "zai/glm-5.1": {
1070
- measuredAt: "2026-07-21T10:21:31Z",
1071
- latencyMs: 7775.4,
1072
- p95LatencyMs: 9182.1,
1073
- outputTokensPerSecond: 6.08,
1518
+ "xiaomi/mimo-v2.5-pro": {
1519
+ measuredAt: "2026-08-29T16:51:33Z",
1520
+ latencyMs: 12070.7,
1521
+ p95LatencyMs: 12386.8,
1522
+ outputTokensPerSecond: 42.44,
1074
1523
  errorRate: 0,
1075
1524
  samples: 3
1076
1525
  },
1077
1526
  "zai/glm-5": {
1078
- measuredAt: "2026-07-21T10:21:31Z",
1079
- latencyMs: 4159.4,
1080
- p95LatencyMs: 4992.7,
1081
- outputTokensPerSecond: 10.28,
1527
+ measuredAt: "2026-08-29T16:51:33Z",
1528
+ latencyMs: 6839.7,
1529
+ p95LatencyMs: 7261.4,
1530
+ outputTokensPerSecond: 75.16,
1531
+ errorRate: 0,
1532
+ samples: 3
1533
+ },
1534
+ "zai/glm-5-turbo": {
1535
+ measuredAt: "2026-08-29T16:51:33Z",
1536
+ latencyMs: 55348.5,
1537
+ p95LatencyMs: 114086.6,
1538
+ outputTokensPerSecond: 14.64,
1082
1539
  errorRate: 0,
1083
1540
  samples: 3
1084
1541
  },
1085
- "free/qwen3-coder-480b": {
1086
- measuredAt: "2026-07-21T10:21:31Z",
1087
- latencyMs: 2063.9,
1088
- p95LatencyMs: 3646.3,
1089
- outputTokensPerSecond: 39.8,
1542
+ "zai/glm-5.1": {
1543
+ measuredAt: "2026-08-29T16:51:33Z",
1544
+ latencyMs: 15658.4,
1545
+ p95LatencyMs: 17307.1,
1546
+ outputTokensPerSecond: 32.9,
1090
1547
  errorRate: 0,
1091
1548
  samples: 3
1092
1549
  },
1093
- "free/mistral-large-3-675b": {
1094
- measuredAt: "2026-07-21T10:21:31Z",
1095
- latencyMs: 3147.5,
1096
- p95LatencyMs: 5555.3,
1097
- outputTokensPerSecond: 27.76,
1550
+ "zai/glm-5.2": {
1551
+ measuredAt: "2026-08-29T16:51:33Z",
1552
+ latencyMs: 10308.5,
1553
+ p95LatencyMs: 15127.6,
1554
+ outputTokensPerSecond: 54.87,
1098
1555
  errorRate: 0,
1099
1556
  samples: 3
1100
1557
  },
1101
- "free/nemotron-3-nano-omni-30b-a3b-reasoning": {
1102
- measuredAt: "2026-07-21T10:21:31Z",
1103
- latencyMs: 6508.4,
1104
- p95LatencyMs: 14252.7,
1105
- outputTokensPerSecond: 68.26,
1558
+ "zai/glm-5.3": {
1559
+ measuredAt: "2026-08-29T16:51:33Z",
1560
+ latencyMs: 7272.4,
1561
+ p95LatencyMs: 7998.1,
1562
+ outputTokensPerSecond: 71.09,
1106
1563
  errorRate: 0,
1107
1564
  samples: 3
1108
1565
  },
1109
- "free/glm-4.7": {
1110
- measuredAt: "2026-07-21T10:21:31Z",
1111
- latencyMs: 2014.8,
1112
- p95LatencyMs: 3039.9,
1113
- outputTokensPerSecond: 39.92,
1566
+ "zai/glm-5.3-flash": {
1567
+ measuredAt: "2026-08-29T16:51:33Z",
1568
+ latencyMs: 10545.3,
1569
+ p95LatencyMs: 11672.4,
1570
+ outputTokensPerSecond: 49.01,
1114
1571
  errorRate: 0,
1115
1572
  samples: 3
1116
1573
  }
@@ -1124,11 +1581,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1124
1581
  latencyMs: 2305,
1125
1582
  outputTokensPerSecond: 140.6
1126
1583
  },
1127
- "anthropic/claude-opus-4.6": {
1128
- measuredAt: "2026-03-16T13:50:48Z",
1129
- latencyMs: 2139,
1130
- outputTokensPerSecond: 119.7
1131
- },
1132
1584
  "anthropic/claude-sonnet-4.6": {
1133
1585
  measuredAt: "2026-03-16T13:50:48Z",
1134
1586
  latencyMs: 2110,
@@ -1162,11 +1614,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1162
1614
  latencyMs: 1609,
1163
1615
  outputTokensPerSecond: 167.2
1164
1616
  },
1165
- "moonshot/kimi-k2.5": {
1166
- measuredAt: "2026-03-16T13:50:48Z",
1167
- latencyMs: 1646,
1168
- outputTokensPerSecond: 155.7
1169
- },
1170
1617
  "openai/gpt-4o-mini": {
1171
1618
  measuredAt: "2026-03-16T13:50:48Z",
1172
1619
  latencyMs: 2764,
@@ -1176,18 +1623,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1176
1623
  measuredAt: "2026-03-16T13:50:48Z",
1177
1624
  latencyMs: 7935,
1178
1625
  outputTokensPerSecond: 32.3
1179
- },
1180
- "xai/grok-4-1-fast-non-reasoning": {
1181
- measuredAt: "2026-03-16T13:50:48Z",
1182
- latencyMs: 1244,
1183
- outputTokensPerSecond: 205.8,
1184
- intelligenceIndex: 41
1185
- },
1186
- "xai/grok-4-1-fast-reasoning": {
1187
- measuredAt: "2026-03-16T13:50:48Z",
1188
- latencyMs: 1454,
1189
- outputTokensPerSecond: 176.2,
1190
- intelligenceIndex: 41
1191
1626
  }
1192
1627
  });
1193
1628
  function inferToolRequirement(prompt, _systemPrompt, toolChoice) {
@@ -1680,7 +2115,7 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1680
2115
  match(["gpt-5.3-codex"], 1),
1681
2116
  match(["claude-sonnet-4.6"], 0.94),
1682
2117
  match(["glm-5.2"], 0.9),
1683
- match(["kimi-k2.7", "deepseek-v4-pro"], 0.86)
2118
+ match(["deepseek-v4-pro"], 0.86)
1684
2119
  );
1685
2120
  case "reasoning":
1686
2121
  return Math.max(
@@ -1704,14 +2139,13 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1704
2139
  base,
1705
2140
  match(["gemini-3.5-flash"], 1),
1706
2141
  match(["grok-4.5"], 0.93),
1707
- match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9),
1708
- match(["kimi-k2.7"], 0.84)
2142
+ match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
1709
2143
  );
1710
2144
  case "vision":
1711
2145
  return Math.max(
1712
2146
  base,
1713
2147
  match(["gemini-3.1-pro"], 0.96),
1714
- match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k2.7", "grok-4.3"], 0.9)
2148
+ match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9)
1715
2149
  );
1716
2150
  case "long_context":
1717
2151
  return Math.max(
@@ -1723,17 +2157,18 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1723
2157
  );
1724
2158
  case "extraction": {
1725
2159
  const kimiExtractionAffinity = language === "zh" ? 1 : 0.9;
2160
+ const otherExtractionAffinity = language === "zh" ? 0.88 : 0.9;
1726
2161
  return Math.max(
1727
2162
  base,
1728
- match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], 0.9),
1729
- match(["claude-sonnet-5", "claude-sonnet-4.6"], 0.9),
1730
- match(["kimi-k3", "kimi-k2.7"], kimiExtractionAffinity)
2163
+ match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], otherExtractionAffinity),
2164
+ match(["claude-sonnet-5", "claude-sonnet-4.6"], otherExtractionAffinity),
2165
+ match(["kimi-k3"], kimiExtractionAffinity)
1731
2166
  );
1732
2167
  }
1733
2168
  default:
1734
2169
  return Math.max(
1735
2170
  base,
1736
- match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3", "kimi-k2.7"], 0.86)
2171
+ match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)
1737
2172
  );
1738
2173
  }
1739
2174
  }
@@ -1792,6 +2227,9 @@ function evidenceCandidates(task) {
1792
2227
  "deepseek/deepseek-v4-pro"
1793
2228
  ];
1794
2229
  }
2230
+ if (task === "extraction") {
2231
+ return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"];
2232
+ }
1795
2233
  if (task === "reasoning_math") {
1796
2234
  return [
1797
2235
  "google/gemini-3.5-flash",
@@ -1848,9 +2286,12 @@ var PortfolioStrategy = class {
1848
2286
  const targetTier = (features.taskType === "reasoning_mcq" || features.taskType === "reasoning_math") && (base.tier === "SIMPLE" || base.tier === "MEDIUM") ? "REASONING" : base.tier;
1849
2287
  const tierConfig = tierConfigs[targetTier];
1850
2288
  const configuredCandidates = tierConfig ? getFallbackChain(targetTier, tierConfigs) : [];
2289
+ const unavailable = new Set(options.unavailableModels ?? []);
1851
2290
  const chain = [
1852
2291
  .../* @__PURE__ */ new Set([...configuredCandidates, ...evidenceCandidates(features.taskType)])
1853
- ].filter((model2) => typeof model2 === "string" && model2.length > 0);
2292
+ ].filter(
2293
+ (model2) => typeof model2 === "string" && model2.length > 0 && !unavailable.has(model2)
2294
+ );
1854
2295
  const eligible = chain.filter(
1855
2296
  (model2) => isEligible(model2, features, maxOutputTokens, options)
1856
2297
  );
@@ -1927,13 +2368,7 @@ var PortfolioStrategy = class {
1927
2368
  ...eligibleCandidates.filter(
1928
2369
  (model2) => !scoredModels.includes(model2) && !webResearchFallbackOrder.includes(model2)
1929
2370
  )
1930
- ] : features.taskType === "tool_agent" || features.taskType === "tool_agent_parallel" && features.agentDomain !== "other" ? [
1931
- ...scoredModels,
1932
- ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
1933
- ] : [
1934
- ...scoredModels,
1935
- ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
1936
- ];
2371
+ ] : [...scoredModels, ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))];
1937
2372
  const model = ranked[0] ?? base.model;
1938
2373
  const selectedTierConfigs = {
1939
2374
  ...tierConfigs,
@@ -1970,7 +2405,7 @@ var PortfolioStrategy = class {
1970
2405
  }
1971
2406
  };
1972
2407
  var DEFAULT_ROUTING_CONFIG = {
1973
- version: "3.4",
2408
+ version: "3.5",
1974
2409
  strategy: "portfolio",
1975
2410
  portfolio: {
1976
2411
  auto: {
@@ -3024,186 +3459,249 @@ var DEFAULT_ROUTING_CONFIG = {
3024
3459
  // Below this confidence → ambiguous (null tier)
3025
3460
  confidenceThreshold: 0.7
3026
3461
  },
3462
+ // ─── Tier chains ───
3463
+ //
3464
+ // Catalog refresh 2026-08-29 (V3.5). Every chain below names only models
3465
+ // the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the
3466
+ // gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast
3467
+ // and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview,
3468
+ // the whole `free/*` namespace — were removed everywhere, including fallback
3469
+ // rungs, so a routed model is always one a user can find on blockrun.ai/models.
3470
+ //
3471
+ // Primaries moved only where portfolio.ts already carries calibration
3472
+ // evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for
3473
+ // agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no
3474
+ // trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash,
3475
+ // gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs;
3476
+ // promotion waits for a calibration run, because version recency is not a
3477
+ // quality signal.
3478
+ //
3479
+ // Latency figures in comments are the 2026-08-29 gateway probe
3480
+ // (model-profiles.generated.json); prices are the catalog list.
3027
3481
  // Auto (balanced) tier configs - current default smart routing
3028
- // Benchmark-tuned 2026-03-16: balancing quality (retention) + latency
3029
3482
  tiers: {
3030
3483
  SIMPLE: {
3031
3484
  primary: "google/gemini-2.5-flash",
3032
- // 1,238ms, IQ 20, 60% retention (best) — fast AND quality
3485
+ // $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer
3033
3486
  fallback: [
3034
3487
  "google/gemini-3-flash-preview",
3035
- // 1,398ms, IQ 46 — smarter fallback
3488
+ // $0.50/$3 — GPQA 5/6 in the 2026-07 calibration
3489
+ "google/gemini-3.5-flash-lite",
3490
+ // $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation
3036
3491
  "deepseek/deepseek-chat",
3037
- // V4 Flash chat ($0.20/$0.40, 1M ctx) — repriced 2026-04-24
3038
- "moonshot/kimi-k2.5",
3039
- // 1,646ms, IQ 47, strong quality
3492
+ // $0.14/$0.28, 1M ctx
3040
3493
  "google/gemini-3.1-flash-lite",
3041
- // $0.25/$1.50, 1M context — newest flash-lite
3042
- "google/gemini-2.5-flash-lite",
3043
- // 1,353ms, $0.10/$0.40
3494
+ // $0.25/$1.50, 1M ctx
3495
+ "openai/gpt-5.6-luna",
3496
+ // $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30)
3044
3497
  "openai/gpt-5.4-nano",
3045
- // $0.20/$1.25, 1M context
3046
- "xai/grok-4-fast-non-reasoning",
3047
- // 1,143ms, $0.20/$0.50 — fast fallback
3048
- "free/gpt-oss-120b"
3049
- // 1,252ms, FREE fallback (hidden from /v1/models but direct calls work)
3498
+ // $0.20/$1.25, 1M ctx
3499
+ "google/gemini-2.5-flash-lite",
3500
+ // $0.10/$0.40
3501
+ "nvidia/step-3.7-flash"
3502
+ // FREE backstop — NVIDIA free tier (probed 2026-08-21)
3050
3503
  ]
3051
3504
  },
3052
3505
  MEDIUM: {
3053
- primary: "moonshot/kimi-k2.7",
3054
- // $0.95/$4.00, 256K ctx, multi-modal + reasoning — Moonshot flagship; promoted from K2.6 (2026-06-14) after BlockRun added K2.7 + hid K2.6. Same price as K2.6.
3506
+ // Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the
3507
+ // calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts).
3508
+ primary: "google/gemini-3.5-flash",
3509
+ // $1.50/$9, 1M ctx, vision + tools
3055
3510
  fallback: [
3056
- "moonshot/kimi-k2.6",
3057
- // identical-cost in-family hot swap (K2.6 still routable)
3058
- "moonshot/kimi-k2.5",
3059
- // $0.60/$3.00 — graceful-degradation backstop
3511
+ "google/gemini-3.6-flash",
3512
+ // $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration
3513
+ "zai/glm-5.3-flash",
3514
+ // $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27
3515
+ "openai/gpt-5.6-terra",
3516
+ // $2/$12, 1M ctx — GPT-5.6 balanced tier
3060
3517
  "google/gemini-3-flash-preview",
3061
- // 1,398ms, IQ 46 — nearly same IQ, faster + cheaper
3518
+ // $0.50/$3
3062
3519
  "deepseek/deepseek-chat",
3063
- // 1,431ms, IQ 32, 41% retention
3520
+ // $0.14/$0.28
3064
3521
  "google/gemini-2.5-flash",
3065
- // 1,238ms, 60% retention
3522
+ // $0.30/$2.50
3523
+ "minimax/minimax-m3",
3524
+ // $0.30/$1.20, 1M ctx
3066
3525
  "google/gemini-3.1-flash-lite",
3067
- // $0.25/$1.50, 1M context
3068
- "google/gemini-2.5-flash-lite",
3069
- // 1,353ms, $0.10/$0.40
3070
- "xai/grok-4-1-fast-non-reasoning",
3071
- // 1,244ms, fast fallback
3072
- "xai/grok-3-mini"
3073
- // 1,202ms, $0.30/$0.50
3526
+ // $0.25/$1.50
3527
+ "openai/gpt-5.6-luna",
3528
+ // $0.20/$1.20
3529
+ "google/gemini-2.5-flash-lite"
3530
+ // $0.10/$0.40
3074
3531
  ]
3075
3532
  },
3076
3533
  COMPLEX: {
3077
3534
  primary: "google/gemini-3.1-pro",
3078
- // 1,609ms, IQ 57 — fast flagship quality
3535
+ // $2/$12 — proven long-context flagship (portfolio.ts long_context lead)
3079
3536
  fallback: [
3080
- "google/gemini-3-flash-preview",
3081
- // 1,398ms, IQ 46 — fast + smart
3082
- "xai/grok-4-0709",
3083
- // 1,348ms, IQ 41
3084
- "google/gemini-2.5-pro",
3085
- // 1,294ms
3537
+ "google/gemini-3.6-flash",
3538
+ // $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here)
3539
+ "google/gemini-3.5-flash",
3540
+ // $1.50/$9 — calibrated
3086
3541
  "anthropic/claude-sonnet-5",
3087
- // near-Opus quality at Sonnet cost, 1M ctx
3542
+ // $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated
3543
+ "xai/grok-4.5",
3544
+ // $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden)
3545
+ "google/gemini-2.5-pro",
3546
+ // $1.25/$10
3088
3547
  "anthropic/claude-sonnet-4.6",
3089
- // 2,110ms, IQ 52 — quality fallback
3090
- "deepseek/deepseek-chat",
3091
- // 1,431ms, IQ 32
3092
- "google/gemini-2.5-flash",
3093
- // 1,238ms, IQ 20 — cheap last resort
3548
+ // $3/$15
3094
3549
  "openai/gpt-5.6-terra",
3095
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3550
+ // $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202)
3096
3551
  "openai/gpt-5.5",
3097
- // Prior OpenAI flagship — 1M+ ctx, native agent + computer use; benchmark TBD
3098
- "openai/gpt-5.4"
3099
- // 6,213ms, IQ 57 — previous flagship, benchmarked
3552
+ // $5/$30 — prior OpenAI flagship
3553
+ "openai/gpt-5.4",
3554
+ // $2.50/$15 — previous flagship, benchmarked
3555
+ "zai/glm-5.3",
3556
+ // $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19
3557
+ "moonshot/kimi-k3",
3558
+ // $3/$15, 1M ctx — Moonshot flagship (K2.7 successor)
3559
+ "deepseek/deepseek-v4-pro",
3560
+ // $0.435/$0.87 — strongest open-weight reasoner
3561
+ "deepseek/deepseek-chat",
3562
+ // $0.14/$0.28 — cheap last resort
3563
+ "google/gemini-2.5-flash"
3564
+ // $0.30/$2.50
3100
3565
  ]
3101
3566
  },
3102
3567
  REASONING: {
3103
- primary: "xai/grok-4-1-fast-reasoning",
3104
- // 1,454ms, $0.20/$0.50
3568
+ // Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek
3569
+ // Reasoner is the cheapest listed reasoner at the same 1M context.
3570
+ primary: "deepseek/deepseek-reasoner",
3571
+ // $0.14/$0.28, 1M ctx
3105
3572
  fallback: [
3106
- "xai/grok-4-fast-reasoning",
3107
- // 1,298ms, $0.20/$0.50
3108
- "deepseek/deepseek-reasoner",
3109
- // V4 Flash thinking ($0.20/$0.40, 1M ctx)
3110
3573
  "deepseek/deepseek-v4-pro",
3111
- // V4 Pro flagship ($0.50/$1.00 promo through 2026-05-31, list $2/$4) — strongest open-weight reasoner
3574
+ // $0.435/$0.87 — calibrated reasoning band 0.95
3575
+ "xai/grok-4.3",
3576
+ // $1.50/$4, 1M ctx — xAI reasoning model, vision
3577
+ "qwen/qwen3.7-plus",
3578
+ // $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed)
3579
+ "google/gemini-3.5-flash",
3580
+ // $1.50/$9 — MGSM 5/5
3112
3581
  "openai/o4-mini",
3113
- // 2,328ms ($1.10/$4.40)
3582
+ // $1.10/$4.40
3114
3583
  "openai/o3"
3115
- // 2,862ms
3584
+ // $2/$8
3116
3585
  ]
3117
3586
  }
3118
3587
  },
3119
3588
  // Eco tier configs - absolute cheapest (blockrun/eco)
3120
3589
  ecoTiers: {
3121
3590
  SIMPLE: {
3122
- primary: "free/gpt-oss-120b",
3123
- // FREE! $0.00/$0.00 — heavy user default
3591
+ primary: "nvidia/step-3.7-flash",
3592
+ // FREE — NVIDIA free tier flagship
3124
3593
  fallback: [
3125
- "free/gpt-oss-20b",
3126
- // FREE — smaller, faster
3127
- "free/deepseek-v4-flash",
3128
- // FREE — 1M ctx; slow (~10 tok/s, 07-28 probe) but completes
3129
- // seed-oss-36b sat here as the free coder until it EOL'd 2026-08-03 (HTTP 410).
3130
- // gpt-oss-120b/20b already head this chain, so the rung is dropped, not retargeted.
3131
- "google/gemini-3.1-flash-lite",
3132
- // $0.25/$1.50 — newest flash-lite
3133
- "openai/gpt-5.4-nano",
3134
- // $0.20/$1.25 — fast nano
3594
+ "nvidia/nemotron-nano-9b-v2",
3595
+ // FREE — compact + fast, high-volume light tasks
3596
+ // The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash
3597
+ // 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400
3598
+ // 2026-08-21). Each retirement retargets the two free rungs to the
3599
+ // current free tier; the paid rungs below never move.
3135
3600
  "google/gemini-2.5-flash-lite",
3136
- // $0.10/$0.40
3137
- "xai/grok-4-fast-non-reasoning"
3138
- // $0.20/$0.50
3601
+ // $0.10/$0.40 — cheapest paid rung
3602
+ "zai/glm-5.3-flash",
3603
+ // $0.15/$0.50, 1M ctx, vision + tools
3604
+ "openai/gpt-5.6-luna",
3605
+ // $0.20/$1.20, 1M ctx
3606
+ "openai/gpt-5.4-nano",
3607
+ // $0.20/$1.25
3608
+ "google/gemini-3.1-flash-lite"
3609
+ // $0.25/$1.50
3139
3610
  ]
3140
3611
  },
3141
3612
  MEDIUM: {
3142
- primary: "google/gemini-3.1-flash-lite",
3143
- // $0.25/$1.50 — newest flash-lite
3613
+ primary: "zai/glm-5.3-flash",
3614
+ // $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model
3144
3615
  fallback: [
3616
+ "deepseek/deepseek-chat",
3617
+ // $0.14/$0.28
3618
+ "google/gemini-3.1-flash-lite",
3619
+ // $0.25/$1.50
3620
+ "openai/gpt-5.6-luna",
3621
+ // $0.20/$1.20
3145
3622
  "openai/gpt-5.4-nano",
3146
3623
  // $0.20/$1.25
3147
3624
  "google/gemini-2.5-flash-lite",
3148
3625
  // $0.10/$0.40
3149
- "xai/grok-4-fast-non-reasoning",
3150
3626
  "google/gemini-2.5-flash"
3627
+ // $0.30/$2.50
3151
3628
  ]
3152
3629
  },
3153
3630
  COMPLEX: {
3154
- primary: "google/gemini-3.1-flash-lite",
3155
- // $0.25/$1.50
3631
+ primary: "zai/glm-5.3-flash",
3632
+ // $0.15/$0.50, 1M ctx
3156
3633
  fallback: [
3157
- "google/gemini-2.5-flash-lite",
3158
- "xai/grok-4-0709",
3159
- "google/gemini-2.5-flash",
3160
- "deepseek/deepseek-chat"
3634
+ "deepseek/deepseek-chat",
3635
+ // $0.14/$0.28, 1M ctx
3636
+ "minimax/minimax-m3",
3637
+ // $0.30/$1.20, 1M ctx
3638
+ "deepseek/deepseek-v4-pro",
3639
+ // $0.435/$0.87
3640
+ "google/gemini-3.1-flash-lite",
3641
+ // $0.25/$1.50
3642
+ "google/gemini-2.5-flash"
3643
+ // $0.30/$2.50
3161
3644
  ]
3162
3645
  },
3163
3646
  REASONING: {
3164
- primary: "xai/grok-4-1-fast-reasoning",
3165
- // $0.20/$0.50
3647
+ primary: "deepseek/deepseek-reasoner",
3648
+ // $0.14/$0.28, 1M ctx — cheapest listed reasoner
3166
3649
  fallback: [
3167
- "xai/grok-4-fast-reasoning",
3168
- "deepseek/deepseek-reasoner",
3169
- // V4 Flash thinking — $0.20/$0.40
3170
- "deepseek/deepseek-v4-pro"
3171
- // V4 Pro flagship — $0.50/$1.00 promo, post-promo $2/$4
3650
+ "deepseek/deepseek-v4-pro",
3651
+ // $0.435/$0.87
3652
+ "qwen/qwen3.7-plus",
3653
+ // $0.32/$1.28 — reasoning
3654
+ "minimax/minimax-m3",
3655
+ // $0.30/$1.20 — reasoning + coding
3656
+ "zai/glm-5.3-flash"
3657
+ // $0.15/$0.50 — reasoning tokens alongside content
3172
3658
  ]
3173
3659
  }
3174
3660
  },
3175
3661
  // Premium tier configs - best quality (blockrun/premium)
3176
- // codex=complex coding, kimi=simple coding, sonnet=reasoning/instructions, opus=architecture/PM/audits
3662
+ // codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits
3177
3663
  premiumTiers: {
3178
3664
  SIMPLE: {
3179
- primary: "moonshot/kimi-k2.7",
3180
- // $0.95/$4.00 - Moonshot flagship (256K ctx, multi-modal + reasoning); promoted from K2.6 (2026-06-14), same price
3665
+ // Was moonshot/kimi-k2.7 (hidden 2026-08).
3666
+ primary: "google/gemini-3.5-flash",
3667
+ // $1.50/$9, 1M ctx, vision + tools — calibrated
3181
3668
  fallback: [
3182
- "moonshot/kimi-k2.6",
3183
- // identical-cost in-family hot swap (K2.6 still routable)
3184
- "moonshot/kimi-k2.5",
3185
- // $0.60/$3.00 - proven reliable backstop when Moonshot direct API falters
3186
- "google/gemini-2.5-flash",
3187
- // 60% retention, fast growth
3669
+ "google/gemini-3.6-flash",
3670
+ // $1.50/$7.50 — newest Flash
3188
3671
  "anthropic/claude-haiku-4.5",
3189
- "google/gemini-2.5-flash-lite",
3672
+ // $1/$5
3673
+ "zai/glm-5.3",
3674
+ // $1.40/$4.40, 1M ctx
3675
+ "google/gemini-2.5-flash",
3676
+ // $0.30/$2.50
3677
+ "google/gemini-3.5-flash-lite",
3678
+ // $0.30/$2.50
3190
3679
  "deepseek/deepseek-chat"
3680
+ // $0.14/$0.28
3191
3681
  ]
3192
3682
  },
3193
3683
  MEDIUM: {
3194
3684
  primary: "openai/gpt-5.3-codex",
3195
- // $1.75/$14 - 400K context, 128K output, replaces 5.2
3685
+ // $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts)
3196
3686
  fallback: [
3197
- "moonshot/kimi-k2.7",
3198
- // Moonshot flagship
3199
- "moonshot/kimi-k2.6",
3200
- "moonshot/kimi-k2.5",
3201
- "google/gemini-2.5-flash",
3202
- // 60% retention, good coding capability
3203
- "google/gemini-2.5-pro",
3204
- "xai/grok-4-0709",
3205
3687
  "anthropic/claude-sonnet-5",
3206
- "anthropic/claude-sonnet-4.6"
3688
+ // $3/$15 — code_agent band 0.98
3689
+ "moonshot/kimi-k3",
3690
+ // $3/$15, 1M ctx — Moonshot flagship
3691
+ "zai/glm-5.3",
3692
+ // $1.40/$4.40 — long-horizon coding
3693
+ "google/gemini-3.6-flash",
3694
+ // $1.50/$7.50
3695
+ "google/gemini-3.5-flash",
3696
+ // $1.50/$9
3697
+ "google/gemini-2.5-pro",
3698
+ // $1.25/$10
3699
+ "xai/grok-4.5",
3700
+ // $2.50/$9
3701
+ "anthropic/claude-sonnet-4.6",
3702
+ // $3/$15
3703
+ "openai/gpt-5.6-terra"
3704
+ // $2/$12
3207
3705
  ]
3208
3706
  },
3209
3707
  COMPLEX: {
@@ -3213,8 +3711,8 @@ var DEFAULT_ROUTING_CONFIG = {
3213
3711
  // Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking)
3214
3712
  // Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is
3215
3713
  // also prone to "high demand" 503s (correlated failure — everyone falls
3216
- // back to Google at the same time). Prefer xAI Grok → Moonshot → OpenAI
3217
- // flagship → DeepSeek → NVIDIA free instead.
3714
+ // back to Google at the same time). Prefer in-family → xAI → Moonshot →
3715
+ // OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead.
3218
3716
  fallback: [
3219
3717
  "anthropic/claude-opus-5",
3220
3718
  // in-family hot swap first (half the price, 1M ctx + adaptive thinking)
@@ -3222,52 +3720,54 @@ var DEFAULT_ROUTING_CONFIG = {
3222
3720
  // in-family hot swap (identical cost to 5)
3223
3721
  "anthropic/claude-opus-4.7",
3224
3722
  // in-family hot swap (identical cost to 4.8)
3225
- "anthropic/claude-opus-4.6",
3226
- // in-family hot swap
3227
3723
  "anthropic/claude-sonnet-5",
3228
3724
  // Sonnet-tier drop-down, near-Opus quality
3229
3725
  "anthropic/claude-sonnet-4.6",
3230
3726
  "xai/grok-4.5",
3231
- // xAI flagship — 503-resistant, direct-xAI SKU (added 2026-07-14)
3232
- "xai/grok-4-0709",
3233
- // 503-resistant flagship
3234
- "moonshot/kimi-k2.7",
3727
+ // xAI flagship — 503-resistant, direct-xAI SKU
3728
+ "moonshot/kimi-k3",
3235
3729
  // Moonshot flagship, independent infra
3236
- "moonshot/kimi-k2.6",
3237
- "moonshot/kimi-k2.5",
3238
3730
  "openai/gpt-5.6-terra",
3239
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3731
+ // GPT-5.6 balanced tier — stable (Sol excluded: #202)
3240
3732
  "openai/gpt-5.5",
3241
3733
  // Prior OpenAI flagship — 1M+ ctx, native agent + computer use
3242
3734
  "openai/gpt-5.4",
3243
3735
  // Previous flagship (slow but stable, benchmarked at 6,213ms)
3244
3736
  "openai/gpt-5.3-codex",
3737
+ "zai/glm-5.3",
3738
+ // Z.AI flagship, 1M ctx
3739
+ "deepseek/deepseek-v4-pro",
3740
+ // strongest open-weight reasoner
3245
3741
  "deepseek/deepseek-chat",
3246
3742
  // Cheap, reliable
3247
- "free/gpt-oss-120b"
3248
- // NVIDIA free ultimate backstop (was seed-oss-36b; EOL'd 2026-08-03)
3743
+ "nvidia/step-3.7-flash"
3744
+ // NVIDIA free ultimate backstop
3249
3745
  ]
3250
3746
  },
3251
3747
  REASONING: {
3252
- primary: "anthropic/claude-sonnet-4.6",
3253
- // 2,110ms, $3/$15 - best for reasoning/instructions
3748
+ // Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both,
3749
+ // plus Sonnet 5's tau2/BrowseComp trajectory evidence).
3750
+ primary: "anthropic/claude-sonnet-5",
3751
+ // $3/$15, 1M ctx, adaptive thinking
3254
3752
  fallback: [
3255
- "anthropic/claude-sonnet-5",
3256
- // in-family hot swap — same cost, adaptive thinking, 1M ctx
3753
+ "anthropic/claude-sonnet-4.6",
3754
+ // in-family hot swap — same cost
3257
3755
  "anthropic/claude-opus-5",
3258
3756
  // Newest flagship Opus w/ adaptive thinking
3259
3757
  "anthropic/claude-opus-4.8",
3260
3758
  // Prior flagship Opus — identical cost to 5
3261
3759
  "anthropic/claude-opus-4.7",
3262
3760
  // Flagship Opus w/ adaptive thinking
3263
- "anthropic/claude-opus-4.6",
3264
- // 2,139ms
3265
- "xai/grok-4-1-fast-reasoning",
3266
- // 1,454ms, cheap fast reasoning
3761
+ "xai/grok-4.5",
3762
+ // reasoning band 0.94
3763
+ "deepseek/deepseek-v4-pro",
3764
+ // reasoning band 0.95
3765
+ "xai/grok-4.3",
3766
+ // $1.50/$4 — xAI reasoning model
3267
3767
  "openai/o4-mini",
3268
- // 2,328ms ($1.10/$4.40)
3768
+ // $1.10/$4.40
3269
3769
  "openai/o3"
3270
- // 2,862ms
3770
+ // $2/$8
3271
3771
  ]
3272
3772
  }
3273
3773
  },
@@ -3277,101 +3777,102 @@ var DEFAULT_ROUTING_CONFIG = {
3277
3777
  primary: "openai/gpt-4o-mini",
3278
3778
  // $0.15/$0.60 - best tool compliance at lowest cost
3279
3779
  fallback: [
3280
- "moonshot/kimi-k2.5",
3281
- // 1,646ms, strong tool use quality
3780
+ "openai/gpt-5.6-luna",
3781
+ // $0.20/$1.20 — lightweight agentic tier of GPT-5.6
3782
+ "zai/glm-5.3-flash",
3783
+ // $0.15/$0.50 — tool calls verified live 2026-08-27
3282
3784
  "anthropic/claude-haiku-4.5",
3283
- // 2,305ms
3284
- "xai/grok-4-1-fast-non-reasoning"
3285
- // 1,244ms, fast fallback
3785
+ // $1/$5
3786
+ "google/gemini-2.5-flash"
3787
+ // $0.30/$2.50
3286
3788
  ]
3287
3789
  },
3288
3790
  MEDIUM: {
3289
- primary: "moonshot/kimi-k2.7",
3290
- // $0.95/$4.00 — Moonshot flagship, strong tool use; promoted from K2.6 (2026-06-14) after BlockRun added K2.7 + hid K2.6. Same price.
3791
+ // Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the
3792
+ // Terminal-Bench and tau2 trajectory evidence in portfolio.ts.
3793
+ primary: "openai/gpt-5-mini",
3794
+ // $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline
3291
3795
  fallback: [
3292
- "moonshot/kimi-k2.6",
3293
- // identical-cost in-family hot swap (K2.6 still routable)
3294
- "moonshot/kimi-k2.5",
3295
- // $0.60/$3.00 — graceful-degradation backstop
3296
- "xai/grok-4-1-fast-non-reasoning",
3297
- // 1,244ms, fast fallback
3796
+ "google/gemini-3.5-flash",
3797
+ // $1.50/$9 — tool_agent band 0.88
3798
+ "zai/glm-5.3-flash",
3799
+ // $0.15/$0.50 — tools verified
3800
+ "openai/gpt-5.6-terra",
3801
+ // $2/$12
3298
3802
  "openai/gpt-4o-mini",
3299
- // 2,764ms, reliable tool calling
3803
+ // $0.15/$0.60 — reliable tool calling
3300
3804
  "anthropic/claude-haiku-4.5",
3301
- // 2,305ms
3302
- "deepseek/deepseek-chat"
3303
- // 1,431ms
3805
+ // $1/$5
3806
+ "deepseek/deepseek-chat",
3807
+ // $0.14/$0.28
3808
+ "moonshot/kimi-k3"
3809
+ // $3/$15 — tool_agent band 0.85
3304
3810
  ]
3305
3811
  },
3306
3812
  COMPLEX: {
3307
- primary: "anthropic/claude-sonnet-4.6",
3308
- // 2,110ms — best agentic quality
3813
+ // Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0,
3814
+ // Terminal-Bench safety band lead (portfolio.ts).
3815
+ primary: "anthropic/claude-sonnet-5",
3816
+ // $3/$15 — best agentic quality per trajectory evidence
3309
3817
  // Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s
3310
3818
  // correlate with Anthropic outages (everyone falls back together).
3311
3819
  // Prefer 503-resistant providers first.
3312
3820
  fallback: [
3313
- "anthropic/claude-sonnet-5",
3314
- // in-family hot swap — same cost, near-Opus agentic quality
3821
+ "anthropic/claude-sonnet-4.6",
3822
+ // in-family hot swap — same cost
3315
3823
  "anthropic/claude-opus-5",
3316
3824
  // Newest flagship Opus — in-family hot swap
3317
3825
  "anthropic/claude-opus-4.8",
3318
3826
  // Prior flagship Opus — identical cost to 5
3319
3827
  "anthropic/claude-opus-4.7",
3320
3828
  // Flagship Opus — in-family hot swap
3321
- "anthropic/claude-opus-4.6",
3322
- // 2,139ms
3323
- "xai/grok-4-0709",
3324
- // 1,348ms — strong tool use, independent infra
3325
- "moonshot/kimi-k2.7",
3326
- // Moonshot flagship — strong tool use, independent infra
3327
- "moonshot/kimi-k2.5",
3328
- // cost-stability backstop
3829
+ "xai/grok-4.5",
3830
+ // xAI flagship — strong tool use, independent infra
3831
+ "moonshot/kimi-k3",
3832
+ // Moonshot flagship — independent infra
3329
3833
  "openai/gpt-5.6-terra",
3330
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3834
+ // GPT-5.6 balanced tier — stable (Sol excluded: #202)
3331
3835
  "openai/gpt-5.5",
3332
3836
  // Prior flagship — native agent + computer use (exactly the agentic-tier use case)
3333
3837
  "openai/gpt-5.4",
3334
- // Previous flagship — 6,213ms, reliable
3838
+ // Previous flagship — reliable
3839
+ "openai/gpt-5.3-codex",
3840
+ // code_agent lead
3841
+ "zai/glm-5.3",
3842
+ // long-horizon coding
3843
+ "deepseek/deepseek-v4-pro",
3844
+ // retail high-risk 3/3
3335
3845
  "deepseek/deepseek-chat",
3336
- // 1,431ms — cheap, reliable
3337
- "free/gpt-oss-120b"
3338
- // NVIDIA free ultimate backstop (was seed-oss-36b; EOL'd 2026-08-03)
3846
+ // cheap, reliable
3847
+ "nvidia/step-3.7-flash"
3848
+ // NVIDIA free ultimate backstop
3339
3849
  ]
3340
3850
  },
3341
3851
  REASONING: {
3342
- primary: "anthropic/claude-sonnet-4.6",
3343
- // 2,110ms — strong tool use + reasoning
3852
+ primary: "anthropic/claude-sonnet-5",
3853
+ // $3/$15 — strong tool use + adaptive thinking
3344
3854
  fallback: [
3345
- "anthropic/claude-sonnet-5",
3346
- // in-family hot swap — same cost, adaptive thinking
3855
+ "anthropic/claude-sonnet-4.6",
3856
+ // in-family hot swap — same cost
3347
3857
  "anthropic/claude-opus-5",
3348
3858
  // Newest flagship Opus w/ adaptive thinking
3349
3859
  "anthropic/claude-opus-4.8",
3350
3860
  // Prior flagship Opus — identical cost to 5
3351
3861
  "anthropic/claude-opus-4.7",
3352
3862
  // Flagship Opus w/ adaptive thinking
3353
- "anthropic/claude-opus-4.6",
3354
- // 2,139ms
3355
- "xai/grok-4-1-fast-reasoning",
3356
- // 1,454ms
3863
+ "xai/grok-4.5",
3864
+ // reasoning band 0.94
3865
+ "deepseek/deepseek-v4-pro",
3866
+ // reasoning band 0.95
3357
3867
  "deepseek/deepseek-reasoner"
3358
- // 1,454ms
3868
+ // $0.14/$0.28
3359
3869
  ]
3360
3870
  }
3361
3871
  },
3362
- // Time-windowed promotions — auto-applied when active, ignored when expired
3363
- promotions: [
3364
- {
3365
- name: "GLM-5.1 Launch Promo ($0.001 flat)",
3366
- startDate: "2026-04-01",
3367
- endDate: "2026-05-01",
3368
- tierOverrides: {
3369
- SIMPLE: { primary: "zai/glm-5.1" }
3370
- },
3371
- profiles: ["auto"]
3372
- // only auto profile — eco stays free, premium stays premium
3373
- }
3374
- ],
3872
+ // Time-windowed promotions — auto-applied when active, ignored when expired.
3873
+ // The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and
3874
+ // has expired; the list is kept empty so the mechanism stays wired.
3875
+ promotions: [],
3375
3876
  overrides: {
3376
3877
  maxTokensForceComplex: 1e5,
3377
3878
  structuredOutputMinTier: "MEDIUM",
@@ -3713,7 +4214,7 @@ async function createSolanaPaymentPayload(secretKey, fromAddress, recipient, amo
3713
4214
  }
3714
4215
  return null;
3715
4216
  };
3716
- let entry = await getBlockhashEntry(connection, rpcUrl, false);
4217
+ let entry = await getBlockhashEntry(connection, rpcUrl, options.forceFreshBlockhash ?? false);
3717
4218
  let serializedTx = findDistinctTx(entry);
3718
4219
  if (serializedTx === null) {
3719
4220
  entry = await getBlockhashEntry(connection, rpcUrl, true);
@@ -3964,7 +4465,7 @@ function getCostSummary() {
3964
4465
  }
3965
4466
 
3966
4467
  // src/version.ts
3967
- var SDK_VERSION = "3.13.1";
4468
+ var SDK_VERSION = "3.13.4";
3968
4469
  var USER_AGENT = `blockrun-ts/${SDK_VERSION}`;
3969
4470
 
3970
4471
  // src/client.ts
@@ -8618,6 +9119,68 @@ async function getOrCreateSolanaWallet() {
8618
9119
  var SOLANA_API_URL = "https://sol.blockrun.ai/api";
8619
9120
  var DEFAULT_MAX_TOKENS2 = 1024;
8620
9121
  var DEFAULT_TIMEOUT14 = 6e4;
9122
+ var STALE_BLOCKHASH_RETRY_BACKOFFS_MS = [500, 2e3];
9123
+ var MAX_PAYMENT_FAILURE_BYTES = 64 * 1024;
9124
+ var SafeStaleBlockhashError = class extends PaymentError {
9125
+ constructor() {
9126
+ super("Payment verification used an expired Solana blockhash; retrying with a fresh quote.");
9127
+ this.name = "SafeStaleBlockhashError";
9128
+ }
9129
+ };
9130
+ function normalizePaymentSignal(value) {
9131
+ return typeof value === "string" ? value.toLowerCase().replace(/[_\-\s:]/g, "") : "";
9132
+ }
9133
+ async function readPaymentFailureBody(response) {
9134
+ const reader = response.body?.getReader();
9135
+ if (!reader) return "";
9136
+ const decoder = new TextDecoder();
9137
+ let total = 0;
9138
+ let text = "";
9139
+ try {
9140
+ for (; ; ) {
9141
+ const { done, value } = await reader.read();
9142
+ if (done) return text + decoder.decode();
9143
+ total += value.byteLength;
9144
+ if (total > MAX_PAYMENT_FAILURE_BYTES) {
9145
+ void reader.cancel();
9146
+ return null;
9147
+ }
9148
+ text += decoder.decode(value, { stream: true });
9149
+ }
9150
+ } catch {
9151
+ return null;
9152
+ }
9153
+ }
9154
+ async function isSafeStaleBlockhashResponse(response) {
9155
+ const length = Number(response.headers.get("content-length") || "0");
9156
+ if (Number.isFinite(length) && length > MAX_PAYMENT_FAILURE_BYTES) return false;
9157
+ const text = await readPaymentFailureBody(response);
9158
+ if (text === null) return false;
9159
+ let body;
9160
+ try {
9161
+ const parsed = JSON.parse(text);
9162
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
9163
+ body = parsed;
9164
+ } catch {
9165
+ return false;
9166
+ }
9167
+ const nested = body.error && typeof body.error === "object" && !Array.isArray(body.error) ? body.error : void 0;
9168
+ const errorLabel = typeof body.error === "string" ? body.error : "";
9169
+ const code = normalizePaymentSignal(body.code ?? nested?.code);
9170
+ const reason = normalizePaymentSignal(body.reason);
9171
+ const detail = normalizePaymentSignal(body.invalidMessage);
9172
+ const message = normalizePaymentSignal(nested?.message ?? body.message);
9173
+ const label = normalizePaymentSignal(errorLabel);
9174
+ if (code.includes("settlementfailed") || label.includes("settlementfailed") || message.includes("settlementfailed")) return false;
9175
+ const verifyPhase = code === "paymentinvalid" || label.includes("verificationfailed") || message.includes("verificationfailed");
9176
+ if (!verifyPhase) return false;
9177
+ return code === "paymentblockhashstale" || detail.includes("blockhashnotfound") || detail.includes("blockheightexceeded") || reason === "expiredsignature" || message.includes("expiredsignature");
9178
+ }
9179
+ async function waitForStaleRetry(attempt) {
9180
+ await new Promise(
9181
+ (resolve) => setTimeout(resolve, STALE_BLOCKHASH_RETRY_BACKOFFS_MS[attempt])
9182
+ );
9183
+ }
8621
9184
  var DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc";
8622
9185
  function resolveRpcConfig(rpcUrl, rpcHeaders) {
8623
9186
  const env = typeof process !== "undefined" && process.env ? process.env : {};
@@ -9064,26 +9627,34 @@ var SolanaLLMClient = class {
9064
9627
  }
9065
9628
  async requestWithPayment(endpoint, body) {
9066
9629
  const url = `${this.apiUrl}${endpoint}`;
9067
- const response = await this.fetchWithTimeout(url, {
9068
- method: "POST",
9069
- headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9070
- body: JSON.stringify(body)
9071
- });
9072
- if (response.status === 402) {
9073
- return this.handlePaymentAndRetry(url, body, response);
9074
- }
9075
- if (!response.ok) {
9076
- let errorBody;
9077
- try {
9078
- errorBody = await response.json();
9079
- } catch {
9080
- errorBody = { error: "Request failed" };
9630
+ for (let staleRetries = 0; ; ) {
9631
+ const response = await this.fetchWithTimeout(url, {
9632
+ method: "POST",
9633
+ headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9634
+ body: JSON.stringify(body)
9635
+ });
9636
+ if (response.status === 402) {
9637
+ try {
9638
+ return await this.handlePaymentAndRetry(url, body, response, staleRetries > 0);
9639
+ } catch (error) {
9640
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9641
+ await waitForStaleRetry(staleRetries++);
9642
+ continue;
9643
+ }
9081
9644
  }
9082
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9645
+ if (!response.ok) {
9646
+ let errorBody;
9647
+ try {
9648
+ errorBody = await response.json();
9649
+ } catch {
9650
+ errorBody = { error: "Request failed" };
9651
+ }
9652
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9653
+ }
9654
+ return response.json();
9083
9655
  }
9084
- return response.json();
9085
9656
  }
9086
- async handlePaymentAndRetry(url, body, response) {
9657
+ async handlePaymentAndRetry(url, body, response, forceFreshBlockhash = false) {
9087
9658
  let paymentHeader = response.headers.get("payment-required");
9088
9659
  if (!paymentHeader) {
9089
9660
  try {
@@ -9125,7 +9696,8 @@ var SolanaLLMClient = class {
9125
9696
  extra: details.extra,
9126
9697
  extensions,
9127
9698
  rpcUrl: this.rpcUrl,
9128
- rpcHeaders: this.rpcHeaders
9699
+ rpcHeaders: this.rpcHeaders,
9700
+ forceFreshBlockhash
9129
9701
  }
9130
9702
  );
9131
9703
  const retryResponse = await this.fetchWithTimeout(url, {
@@ -9138,6 +9710,9 @@ var SolanaLLMClient = class {
9138
9710
  body: JSON.stringify(body)
9139
9711
  });
9140
9712
  if (retryResponse.status === 402) {
9713
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9714
+ throw new SafeStaleBlockhashError();
9715
+ }
9141
9716
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9142
9717
  }
9143
9718
  if (!retryResponse.ok) {
@@ -9156,26 +9731,34 @@ var SolanaLLMClient = class {
9156
9731
  }
9157
9732
  async requestWithPaymentRaw(endpoint, body) {
9158
9733
  const url = `${this.apiUrl}${endpoint}`;
9159
- const response = await this.fetchWithTimeout(url, {
9160
- method: "POST",
9161
- headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9162
- body: JSON.stringify(body)
9163
- });
9164
- if (response.status === 402) {
9165
- return this.handlePaymentAndRetryRaw(url, body, response);
9166
- }
9167
- if (!response.ok) {
9168
- let errorBody;
9169
- try {
9170
- errorBody = await response.json();
9171
- } catch {
9172
- errorBody = { error: "Request failed" };
9734
+ for (let staleRetries = 0; ; ) {
9735
+ const response = await this.fetchWithTimeout(url, {
9736
+ method: "POST",
9737
+ headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9738
+ body: JSON.stringify(body)
9739
+ });
9740
+ if (response.status === 402) {
9741
+ try {
9742
+ return await this.handlePaymentAndRetryRaw(url, body, response, staleRetries > 0);
9743
+ } catch (error) {
9744
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9745
+ await waitForStaleRetry(staleRetries++);
9746
+ continue;
9747
+ }
9173
9748
  }
9174
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9749
+ if (!response.ok) {
9750
+ let errorBody;
9751
+ try {
9752
+ errorBody = await response.json();
9753
+ } catch {
9754
+ errorBody = { error: "Request failed" };
9755
+ }
9756
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9757
+ }
9758
+ return response.json();
9175
9759
  }
9176
- return response.json();
9177
9760
  }
9178
- async handlePaymentAndRetryRaw(url, body, response) {
9761
+ async handlePaymentAndRetryRaw(url, body, response, forceFreshBlockhash = false) {
9179
9762
  let paymentHeader = response.headers.get("payment-required");
9180
9763
  if (!paymentHeader) {
9181
9764
  try {
@@ -9217,7 +9800,8 @@ var SolanaLLMClient = class {
9217
9800
  extra: details.extra,
9218
9801
  extensions,
9219
9802
  rpcUrl: this.rpcUrl,
9220
- rpcHeaders: this.rpcHeaders
9803
+ rpcHeaders: this.rpcHeaders,
9804
+ forceFreshBlockhash
9221
9805
  }
9222
9806
  );
9223
9807
  const retryResponse = await this.fetchWithTimeout(url, {
@@ -9230,6 +9814,9 @@ var SolanaLLMClient = class {
9230
9814
  body: JSON.stringify(body)
9231
9815
  });
9232
9816
  if (retryResponse.status === 402) {
9817
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9818
+ throw new SafeStaleBlockhashError();
9819
+ }
9233
9820
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9234
9821
  }
9235
9822
  if (!retryResponse.ok) {
@@ -9249,25 +9836,33 @@ var SolanaLLMClient = class {
9249
9836
  async getWithPaymentRaw(endpoint, params) {
9250
9837
  const query = params ? "?" + new URLSearchParams(params).toString() : "";
9251
9838
  const url = `${this.apiUrl}${endpoint}${query}`;
9252
- const response = await this.fetchWithTimeout(url, {
9253
- method: "GET",
9254
- headers: { "User-Agent": USER_AGENT }
9255
- });
9256
- if (response.status === 402) {
9257
- return this.handleGetPaymentAndRetryRaw(url, endpoint, params, response);
9258
- }
9259
- if (!response.ok) {
9260
- let errorBody;
9261
- try {
9262
- errorBody = await response.json();
9263
- } catch {
9264
- errorBody = { error: "Request failed" };
9839
+ for (let staleRetries = 0; ; ) {
9840
+ const response = await this.fetchWithTimeout(url, {
9841
+ method: "GET",
9842
+ headers: { "User-Agent": USER_AGENT }
9843
+ });
9844
+ if (response.status === 402) {
9845
+ try {
9846
+ return await this.handleGetPaymentAndRetryRaw(url, endpoint, params, response, staleRetries > 0);
9847
+ } catch (error) {
9848
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9849
+ await waitForStaleRetry(staleRetries++);
9850
+ continue;
9851
+ }
9265
9852
  }
9266
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9853
+ if (!response.ok) {
9854
+ let errorBody;
9855
+ try {
9856
+ errorBody = await response.json();
9857
+ } catch {
9858
+ errorBody = { error: "Request failed" };
9859
+ }
9860
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9861
+ }
9862
+ return response.json();
9267
9863
  }
9268
- return response.json();
9269
9864
  }
9270
- async handleGetPaymentAndRetryRaw(url, endpoint, params, response) {
9865
+ async handleGetPaymentAndRetryRaw(url, endpoint, params, response, forceFreshBlockhash = false) {
9271
9866
  let paymentHeader = response.headers.get("payment-required");
9272
9867
  if (!paymentHeader) {
9273
9868
  try {
@@ -9309,7 +9904,8 @@ var SolanaLLMClient = class {
9309
9904
  extra: details.extra,
9310
9905
  extensions,
9311
9906
  rpcUrl: this.rpcUrl,
9312
- rpcHeaders: this.rpcHeaders
9907
+ rpcHeaders: this.rpcHeaders,
9908
+ forceFreshBlockhash
9313
9909
  }
9314
9910
  );
9315
9911
  const query = params ? "?" + new URLSearchParams(params).toString() : "";
@@ -9322,6 +9918,9 @@ var SolanaLLMClient = class {
9322
9918
  }
9323
9919
  });
9324
9920
  if (retryResponse.status === 402) {
9921
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9922
+ throw new SafeStaleBlockhashError();
9923
+ }
9325
9924
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9326
9925
  }
9327
9926
  if (!retryResponse.ok) {