@blockrun/llm 3.13.2 → 3.13.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -145,7 +145,7 @@ var APIError = class extends BlockrunError {
145
145
  }
146
146
  };
147
147
 
148
- // node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+18bf4abc7e345_6te54dzikmquqgkdjdanfxg4a4/node_modules/@blockrun/router-core/dist/index.js
148
+ // node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+5d911879d7f1e_fbldkf3f3jmlwtwu53ftq2mj24/node_modules/@blockrun/router-core/dist/index.js
149
149
  function scoreTokenCount(estimatedTokens, thresholds) {
150
150
  if (estimatedTokens < thresholds.simple) {
151
151
  return { name: "tokenCount", score: -1, signal: `short (${estimatedTokens} tokens)` };
@@ -479,6 +479,21 @@ function applyPromotions(tierConfigs, promotions, profile, now = /* @__PURE__ */
479
479
  }
480
480
  return result;
481
481
  }
482
+ function applyUnavailableModels(tierConfigs, unavailableModels) {
483
+ if (!unavailableModels || unavailableModels.length === 0) return tierConfigs;
484
+ const dead = new Set(unavailableModels);
485
+ let result = tierConfigs;
486
+ for (const tier of Object.keys(tierConfigs)) {
487
+ const config = tierConfigs[tier];
488
+ const alive = [config.primary, ...config.fallback].filter((model) => !dead.has(model));
489
+ if (alive.length === 0 || alive[0] === config.primary && alive.length === config.fallback.length + 1) {
490
+ continue;
491
+ }
492
+ if (result === tierConfigs) result = { ...tierConfigs };
493
+ result[tier] = { primary: alive[0], fallback: alive.slice(1) };
494
+ }
495
+ return result;
496
+ }
482
497
  var RulesStrategy = class {
483
498
  name = "rules";
484
499
  route(prompt, systemPrompt, maxOutputTokens, options) {
@@ -530,6 +545,7 @@ ${value.slice(-(scanLimit - prefixLength))}`;
530
545
  profile = useAgenticTiers ? "agentic" : "auto";
531
546
  }
532
547
  tierConfigs = applyPromotions(tierConfigs, config.promotions, profile, options.now);
548
+ tierConfigs = applyUnavailableModels(tierConfigs, options.unavailableModels);
533
549
  const agenticScoreValue = ruleResult.agenticScore;
534
550
  if (estimatedTokens > config.overrides.maxTokensForceComplex) {
535
551
  const decision2 = selectModel(
@@ -603,14 +619,15 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
603
619
  supportsVision: true
604
620
  },
605
621
  "anthropic/claude-haiku-4.5": {
622
+ // override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
606
623
  contextWindow: 2e5,
607
- maxOutputTokens: 8192,
624
+ maxOutputTokens: 64e3,
608
625
  supportsTools: true,
609
626
  supportsVision: true
610
627
  },
611
- "anthropic/claude-opus-4.6": {
612
- contextWindow: 1e6,
613
- maxOutputTokens: 128e3,
628
+ "anthropic/claude-opus-4.5": {
629
+ contextWindow: 2e5,
630
+ maxOutputTokens: 64e3,
614
631
  supportsTools: true,
615
632
  supportsVision: true
616
633
  },
@@ -632,12 +649,19 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
632
649
  supportsTools: true,
633
650
  supportsVision: true
634
651
  },
635
- "anthropic/claude-sonnet-4.6": {
652
+ "anthropic/claude-sonnet-4.5": {
636
653
  contextWindow: 2e5,
637
654
  maxOutputTokens: 64e3,
638
655
  supportsTools: true,
639
656
  supportsVision: true
640
657
  },
658
+ "anthropic/claude-sonnet-4.6": {
659
+ // override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
660
+ contextWindow: 1e6,
661
+ maxOutputTokens: 128e3,
662
+ supportsTools: true,
663
+ supportsVision: true
664
+ },
641
665
  "anthropic/claude-sonnet-5": {
642
666
  contextWindow: 1e6,
643
667
  maxOutputTokens: 128e3,
@@ -645,14 +669,14 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
645
669
  supportsVision: true
646
670
  },
647
671
  "deepseek/deepseek-chat": {
648
- contextWindow: 1e6,
649
- maxOutputTokens: 8192,
672
+ contextWindow: 1048576,
673
+ maxOutputTokens: 65536,
650
674
  supportsTools: true,
651
675
  supportsVision: false
652
676
  },
653
677
  "deepseek/deepseek-reasoner": {
654
- contextWindow: 1e6,
655
- maxOutputTokens: 8192,
678
+ contextWindow: 1048576,
679
+ maxOutputTokens: 65536,
656
680
  supportsTools: true,
657
681
  supportsVision: false
658
682
  },
@@ -662,62 +686,38 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
662
686
  supportsTools: true,
663
687
  supportsVision: false
664
688
  },
665
- "free/deepseek-v4-flash": {
666
- contextWindow: 1e6,
667
- maxOutputTokens: 16384,
668
- supportsTools: false,
669
- supportsVision: false
670
- },
671
- "free/gpt-oss-120b": {
672
- contextWindow: 128e3,
673
- maxOutputTokens: 16384,
674
- supportsTools: false,
675
- supportsVision: false
676
- },
677
- "free/gpt-oss-20b": {
678
- contextWindow: 128e3,
679
- maxOutputTokens: 16384,
680
- supportsTools: false,
681
- supportsVision: false
682
- },
683
- "free/seed-oss-36b": {
684
- contextWindow: 131072,
685
- maxOutputTokens: 16384,
686
- supportsTools: false,
687
- supportsVision: false
688
- },
689
689
  "google/gemini-2.5-flash": {
690
- contextWindow: 1e6,
690
+ contextWindow: 1048576,
691
691
  maxOutputTokens: 65536,
692
692
  supportsTools: true,
693
693
  supportsVision: true
694
694
  },
695
695
  "google/gemini-2.5-flash-lite": {
696
- contextWindow: 1e6,
696
+ contextWindow: 1048576,
697
697
  maxOutputTokens: 65536,
698
698
  supportsTools: true,
699
699
  supportsVision: false
700
700
  },
701
701
  "google/gemini-2.5-pro": {
702
- contextWindow: 105e4,
702
+ contextWindow: 1048576,
703
703
  maxOutputTokens: 65536,
704
704
  supportsTools: true,
705
705
  supportsVision: true
706
706
  },
707
707
  "google/gemini-3-flash-preview": {
708
- contextWindow: 1e6,
708
+ contextWindow: 1048576,
709
709
  maxOutputTokens: 65536,
710
- supportsTools: false,
710
+ supportsTools: true,
711
711
  supportsVision: true
712
712
  },
713
713
  "google/gemini-3.1-flash-lite": {
714
- contextWindow: 1e6,
715
- maxOutputTokens: 8192,
714
+ contextWindow: 1048576,
715
+ maxOutputTokens: 65536,
716
716
  supportsTools: true,
717
717
  supportsVision: false
718
718
  },
719
719
  "google/gemini-3.1-pro": {
720
- contextWindow: 105e4,
720
+ contextWindow: 1048576,
721
721
  maxOutputTokens: 65536,
722
722
  supportsTools: true,
723
723
  supportsVision: true
@@ -728,23 +728,29 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
728
728
  supportsTools: true,
729
729
  supportsVision: true
730
730
  },
731
- "moonshot/kimi-k2.5": {
732
- contextWindow: 262144,
733
- maxOutputTokens: 16384,
731
+ "google/gemini-3.5-flash-lite": {
732
+ contextWindow: 1048576,
733
+ maxOutputTokens: 65536,
734
734
  supportsTools: true,
735
- supportsVision: true
735
+ supportsVision: false
736
736
  },
737
- "moonshot/kimi-k2.6": {
738
- contextWindow: 262144,
737
+ "google/gemini-3.6-flash": {
738
+ contextWindow: 1048576,
739
739
  maxOutputTokens: 65536,
740
740
  supportsTools: true,
741
741
  supportsVision: true
742
742
  },
743
- "moonshot/kimi-k2.7": {
744
- contextWindow: 262144,
743
+ "minimax/minimax-m2.7": {
744
+ contextWindow: 204800,
745
+ maxOutputTokens: 16384,
746
+ supportsTools: true,
747
+ supportsVision: false
748
+ },
749
+ "minimax/minimax-m3": {
750
+ contextWindow: 1048576,
745
751
  maxOutputTokens: 65536,
746
752
  supportsTools: true,
747
- supportsVision: true
753
+ supportsVision: false
748
754
  },
749
755
  "moonshot/kimi-k3": {
750
756
  contextWindow: 1048576,
@@ -752,7 +758,63 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
752
758
  supportsTools: true,
753
759
  supportsVision: true
754
760
  },
761
+ "nvidia/mistral-nemotron": {
762
+ // supportsTools: gateway unavailable at probe time — fails closed
763
+ contextWindow: 131072,
764
+ maxOutputTokens: 16384,
765
+ supportsTools: false,
766
+ supportsVision: false
767
+ },
768
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
769
+ contextWindow: 256e3,
770
+ maxOutputTokens: 16384,
771
+ supportsTools: false,
772
+ supportsVision: true
773
+ },
774
+ "nvidia/nemotron-nano-12b-v2-vl": {
775
+ // supportsTools: gateway unavailable at probe time — fails closed
776
+ contextWindow: 131072,
777
+ maxOutputTokens: 16384,
778
+ supportsTools: false,
779
+ supportsVision: true
780
+ },
781
+ "nvidia/nemotron-nano-9b-v2": {
782
+ contextWindow: 131072,
783
+ maxOutputTokens: 16384,
784
+ supportsTools: false,
785
+ supportsVision: false
786
+ },
787
+ "nvidia/step-3.7-flash": {
788
+ contextWindow: 131072,
789
+ maxOutputTokens: 16384,
790
+ supportsTools: false,
791
+ supportsVision: false
792
+ },
793
+ "openai/chat-latest": {
794
+ contextWindow: 128e3,
795
+ maxOutputTokens: 128e3,
796
+ supportsTools: true,
797
+ supportsVision: true
798
+ },
755
799
  "openai/gpt-4.1": {
800
+ contextWindow: 128e3,
801
+ maxOutputTokens: 32768,
802
+ supportsTools: true,
803
+ supportsVision: true
804
+ },
805
+ "openai/gpt-4.1-mini": {
806
+ contextWindow: 128e3,
807
+ maxOutputTokens: 32768,
808
+ supportsTools: true,
809
+ supportsVision: false
810
+ },
811
+ "openai/gpt-4.1-nano": {
812
+ contextWindow: 128e3,
813
+ maxOutputTokens: 32768,
814
+ supportsTools: true,
815
+ supportsVision: false
816
+ },
817
+ "openai/gpt-4o": {
756
818
  contextWindow: 128e3,
757
819
  maxOutputTokens: 16384,
758
820
  supportsTools: true,
@@ -766,17 +828,44 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
766
828
  },
767
829
  "openai/gpt-5-mini": {
768
830
  contextWindow: 2e5,
769
- maxOutputTokens: 65536,
831
+ maxOutputTokens: 128e3,
770
832
  supportsTools: true,
771
833
  supportsVision: false
772
834
  },
835
+ "openai/gpt-5.2": {
836
+ contextWindow: 4e5,
837
+ maxOutputTokens: 128e3,
838
+ supportsTools: true,
839
+ supportsVision: true
840
+ },
841
+ "openai/gpt-5.2-pro": {
842
+ // supportsTools: not probed — fails closed
843
+ contextWindow: 4e5,
844
+ maxOutputTokens: 128e3,
845
+ supportsTools: false,
846
+ supportsVision: true
847
+ },
848
+ "openai/gpt-5.3": {
849
+ // supportsTools: gateway unavailable at probe time — fails closed
850
+ contextWindow: 128e3,
851
+ maxOutputTokens: 128e3,
852
+ supportsTools: false,
853
+ supportsVision: true
854
+ },
773
855
  "openai/gpt-5.3-codex": {
856
+ // supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe measured an incident, not the model. Codex's function calling is established by the 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot claiming the model cannot call tools.
774
857
  contextWindow: 4e5,
775
858
  maxOutputTokens: 128e3,
776
859
  supportsTools: true,
777
860
  supportsVision: false
778
861
  },
779
862
  "openai/gpt-5.4": {
863
+ contextWindow: 105e4,
864
+ maxOutputTokens: 128e3,
865
+ supportsTools: true,
866
+ supportsVision: true
867
+ },
868
+ "openai/gpt-5.4-mini": {
780
869
  contextWindow: 4e5,
781
870
  maxOutputTokens: 128e3,
782
871
  supportsTools: true,
@@ -784,30 +873,92 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
784
873
  },
785
874
  "openai/gpt-5.4-nano": {
786
875
  contextWindow: 105e4,
787
- maxOutputTokens: 32768,
876
+ maxOutputTokens: 128e3,
788
877
  supportsTools: true,
789
878
  supportsVision: false
790
879
  },
880
+ "openai/gpt-5.4-pro": {
881
+ // supportsTools: not probed — fails closed
882
+ contextWindow: 105e4,
883
+ maxOutputTokens: 128e3,
884
+ supportsTools: false,
885
+ supportsVision: true
886
+ },
791
887
  "openai/gpt-5.5": {
792
888
  contextWindow: 105e4,
793
889
  maxOutputTokens: 128e3,
794
890
  supportsTools: true,
795
891
  supportsVision: true
796
892
  },
893
+ "openai/gpt-5.5-pro": {
894
+ // supportsTools: not probed — fails closed
895
+ contextWindow: 105e4,
896
+ maxOutputTokens: 128e3,
897
+ supportsTools: false,
898
+ supportsVision: true
899
+ },
900
+ "openai/gpt-5.6-luna": {
901
+ contextWindow: 105e4,
902
+ maxOutputTokens: 128e3,
903
+ supportsTools: true,
904
+ supportsVision: true
905
+ },
906
+ "openai/gpt-5.6-luna-pro": {
907
+ contextWindow: 105e4,
908
+ maxOutputTokens: 128e3,
909
+ supportsTools: false,
910
+ supportsVision: true
911
+ },
912
+ "openai/gpt-5.6-sol": {
913
+ contextWindow: 105e4,
914
+ maxOutputTokens: 128e3,
915
+ supportsTools: true,
916
+ supportsVision: true
917
+ },
918
+ "openai/gpt-5.6-sol-pro": {
919
+ contextWindow: 105e4,
920
+ maxOutputTokens: 128e3,
921
+ supportsTools: true,
922
+ supportsVision: true
923
+ },
797
924
  "openai/gpt-5.6-terra": {
798
925
  contextWindow: 105e4,
799
926
  maxOutputTokens: 128e3,
800
927
  supportsTools: true,
801
928
  supportsVision: true
802
929
  },
930
+ "openai/gpt-5.6-terra-pro": {
931
+ contextWindow: 105e4,
932
+ maxOutputTokens: 128e3,
933
+ supportsTools: true,
934
+ supportsVision: true
935
+ },
936
+ "openai/o1": {
937
+ contextWindow: 2e5,
938
+ maxOutputTokens: 1e5,
939
+ supportsTools: true,
940
+ supportsVision: false
941
+ },
803
942
  "openai/o3": {
804
943
  contextWindow: 2e5,
805
944
  maxOutputTokens: 1e5,
806
945
  supportsTools: true,
807
946
  supportsVision: false
808
947
  },
948
+ "openai/o3-mini": {
949
+ contextWindow: 128e3,
950
+ maxOutputTokens: 1e5,
951
+ supportsTools: true,
952
+ supportsVision: false
953
+ },
809
954
  "openai/o4-mini": {
810
955
  contextWindow: 128e3,
956
+ maxOutputTokens: 1e5,
957
+ supportsTools: true,
958
+ supportsVision: false
959
+ },
960
+ "qwen/qwen3.7-flash": {
961
+ contextWindow: 1e6,
811
962
  maxOutputTokens: 65536,
812
963
  supportsTools: true,
813
964
  supportsVision: false
@@ -818,299 +969,605 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
818
969
  supportsTools: true,
819
970
  supportsVision: false
820
971
  },
821
- "xai/grok-3-mini": {
822
- contextWindow: 131072,
823
- maxOutputTokens: 16384,
972
+ "qwen/qwen3.7-plus": {
973
+ contextWindow: 1e6,
974
+ maxOutputTokens: 131072,
824
975
  supportsTools: true,
825
976
  supportsVision: false
826
977
  },
827
- "xai/grok-4-0709": {
828
- contextWindow: 131072,
829
- maxOutputTokens: 16384,
978
+ "tencent/hy3": {
979
+ contextWindow: 262144,
980
+ maxOutputTokens: 128e3,
830
981
  supportsTools: true,
831
982
  supportsVision: false
832
983
  },
833
- "xai/grok-4-1-fast-non-reasoning": {
834
- contextWindow: 131072,
984
+ "xai/grok-4.3": {
985
+ contextWindow: 1e6,
835
986
  maxOutputTokens: 16384,
836
987
  supportsTools: true,
837
- supportsVision: false
988
+ supportsVision: true
838
989
  },
839
- "xai/grok-4-1-fast-reasoning": {
840
- contextWindow: 131072,
990
+ "xai/grok-4.5": {
991
+ contextWindow: 5e5,
841
992
  maxOutputTokens: 16384,
842
993
  supportsTools: true,
843
- supportsVision: false
994
+ supportsVision: true
844
995
  },
845
- "xai/grok-4-fast-non-reasoning": {
846
- contextWindow: 131072,
996
+ "xai/grok-build-0.1": {
997
+ contextWindow: 256e3,
847
998
  maxOutputTokens: 16384,
848
999
  supportsTools: true,
849
1000
  supportsVision: false
850
1001
  },
851
- "xai/grok-4-fast-reasoning": {
852
- contextWindow: 131072,
853
- maxOutputTokens: 16384,
854
- supportsTools: true,
855
- supportsVision: false
1002
+ "xiaomi/mimo-v2.5-pro": {
1003
+ contextWindow: 1048576,
1004
+ maxOutputTokens: 131072,
1005
+ supportsTools: true,
1006
+ supportsVision: false
1007
+ },
1008
+ "zai/glm-5": {
1009
+ contextWindow: 2e5,
1010
+ maxOutputTokens: 128e3,
1011
+ supportsTools: true,
1012
+ supportsVision: false
1013
+ },
1014
+ "zai/glm-5-turbo": {
1015
+ contextWindow: 2e5,
1016
+ maxOutputTokens: 128e3,
1017
+ supportsTools: true,
1018
+ supportsVision: false
1019
+ },
1020
+ "zai/glm-5.1": {
1021
+ contextWindow: 2e5,
1022
+ maxOutputTokens: 128e3,
1023
+ supportsTools: true,
1024
+ supportsVision: false
1025
+ },
1026
+ "zai/glm-5.2": {
1027
+ contextWindow: 1e6,
1028
+ maxOutputTokens: 131072,
1029
+ supportsTools: true,
1030
+ supportsVision: false
1031
+ },
1032
+ "zai/glm-5.3": {
1033
+ contextWindow: 1e6,
1034
+ maxOutputTokens: 131072,
1035
+ supportsTools: true,
1036
+ supportsVision: false
1037
+ },
1038
+ "zai/glm-5.3-flash": {
1039
+ contextWindow: 1e6,
1040
+ maxOutputTokens: 131072,
1041
+ supportsTools: true,
1042
+ supportsVision: true
1043
+ }
1044
+ });
1045
+ var model_profiles_generated_default = {
1046
+ "anthropic/claude-fable-5": {
1047
+ measuredAt: "2026-08-29T16:51:33Z",
1048
+ latencyMs: 9298.5,
1049
+ p95LatencyMs: 9873.4,
1050
+ outputTokensPerSecond: 55.17,
1051
+ errorRate: 0,
1052
+ samples: 3
1053
+ },
1054
+ "anthropic/claude-haiku-4.5": {
1055
+ measuredAt: "2026-08-29T16:51:33Z",
1056
+ latencyMs: 3157.4,
1057
+ p95LatencyMs: 3170.7,
1058
+ outputTokensPerSecond: 162.16,
1059
+ errorRate: 0,
1060
+ samples: 3
1061
+ },
1062
+ "anthropic/claude-opus-4.5": {
1063
+ measuredAt: "2026-08-29T16:51:33Z",
1064
+ latencyMs: 6497.7,
1065
+ p95LatencyMs: 6953.7,
1066
+ outputTokensPerSecond: 78.99,
1067
+ errorRate: 0,
1068
+ samples: 3
1069
+ },
1070
+ "anthropic/claude-opus-4.7": {
1071
+ measuredAt: "2026-08-29T16:51:33Z",
1072
+ latencyMs: 5316.5,
1073
+ p95LatencyMs: 6121.5,
1074
+ outputTokensPerSecond: 97.34,
1075
+ errorRate: 0,
1076
+ samples: 3
1077
+ },
1078
+ "anthropic/claude-opus-4.8": {
1079
+ measuredAt: "2026-08-29T16:51:33Z",
1080
+ latencyMs: 6216.1,
1081
+ p95LatencyMs: 6847.7,
1082
+ outputTokensPerSecond: 82.81,
1083
+ errorRate: 0,
1084
+ samples: 3
1085
+ },
1086
+ "anthropic/claude-opus-5": {
1087
+ measuredAt: "2026-08-29T16:51:33Z",
1088
+ latencyMs: 7309,
1089
+ p95LatencyMs: 7745.2,
1090
+ outputTokensPerSecond: 70.17,
1091
+ errorRate: 0,
1092
+ samples: 3
1093
+ },
1094
+ "anthropic/claude-sonnet-4.5": {
1095
+ measuredAt: "2026-08-29T16:51:33Z",
1096
+ latencyMs: 6330.4,
1097
+ p95LatencyMs: 6631.6,
1098
+ outputTokensPerSecond: 81.03,
1099
+ errorRate: 0,
1100
+ samples: 3
1101
+ },
1102
+ "anthropic/claude-sonnet-4.6": {
1103
+ measuredAt: "2026-08-29T16:51:33Z",
1104
+ latencyMs: 6508,
1105
+ p95LatencyMs: 6698.3,
1106
+ outputTokensPerSecond: 78.6,
1107
+ errorRate: 0,
1108
+ samples: 3
1109
+ },
1110
+ "anthropic/claude-sonnet-5": {
1111
+ measuredAt: "2026-08-29T16:51:33Z",
1112
+ latencyMs: 6165.4,
1113
+ p95LatencyMs: 6582.9,
1114
+ outputTokensPerSecond: 83.62,
1115
+ errorRate: 0,
1116
+ samples: 3
1117
+ },
1118
+ "deepseek/deepseek-chat": {
1119
+ measuredAt: "2026-08-29T16:51:33Z",
1120
+ latencyMs: 4351.4,
1121
+ p95LatencyMs: 4543.7,
1122
+ outputTokensPerSecond: 117.78,
1123
+ errorRate: 0,
1124
+ samples: 3
1125
+ },
1126
+ "deepseek/deepseek-reasoner": {
1127
+ measuredAt: "2026-08-29T16:51:33Z",
1128
+ latencyMs: 5201.2,
1129
+ p95LatencyMs: 6079.6,
1130
+ outputTokensPerSecond: 99.77,
1131
+ errorRate: 0,
1132
+ samples: 3
1133
+ },
1134
+ "deepseek/deepseek-v4-pro": {
1135
+ measuredAt: "2026-08-29T16:51:33Z",
1136
+ latencyMs: 8781.2,
1137
+ p95LatencyMs: 9881.1,
1138
+ outputTokensPerSecond: 58.98,
1139
+ errorRate: 0,
1140
+ samples: 3
1141
+ },
1142
+ "google/gemini-2.5-flash": {
1143
+ measuredAt: "2026-08-29T16:51:33Z",
1144
+ latencyMs: 5416.4,
1145
+ p95LatencyMs: 6442.8,
1146
+ outputTokensPerSecond: 213.07,
1147
+ errorRate: 0,
1148
+ samples: 3
1149
+ },
1150
+ "google/gemini-2.5-flash-lite": {
1151
+ measuredAt: "2026-08-29T16:51:33Z",
1152
+ latencyMs: 5002.6,
1153
+ p95LatencyMs: 5780.3,
1154
+ outputTokensPerSecond: 408.43,
1155
+ errorRate: 0,
1156
+ samples: 3
1157
+ },
1158
+ "google/gemini-2.5-pro": {
1159
+ measuredAt: "2026-08-29T16:51:33Z",
1160
+ latencyMs: 28169.5,
1161
+ p95LatencyMs: 29491.4,
1162
+ outputTokensPerSecond: 147.3,
1163
+ errorRate: 0,
1164
+ samples: 3
1165
+ },
1166
+ "google/gemini-3-flash-preview": {
1167
+ measuredAt: "2026-08-29T16:51:33Z",
1168
+ latencyMs: 4717.1,
1169
+ p95LatencyMs: 5037.1,
1170
+ outputTokensPerSecond: 198.71,
1171
+ errorRate: 0,
1172
+ samples: 3
1173
+ },
1174
+ "google/gemini-3.1-flash-lite": {
1175
+ measuredAt: "2026-08-29T16:51:33Z",
1176
+ latencyMs: 2855.8,
1177
+ p95LatencyMs: 3172.7,
1178
+ outputTokensPerSecond: 286.91,
1179
+ errorRate: 0,
1180
+ samples: 3
1181
+ },
1182
+ "google/gemini-3.1-pro": {
1183
+ measuredAt: "2026-08-29T16:59:54Z",
1184
+ latencyMs: 24194.1,
1185
+ p95LatencyMs: 27269.6,
1186
+ outputTokensPerSecond: 109.47,
1187
+ errorRate: 0,
1188
+ samples: 3
1189
+ },
1190
+ "google/gemini-3.5-flash": {
1191
+ measuredAt: "2026-08-29T16:51:33Z",
1192
+ latencyMs: 5320.6,
1193
+ p95LatencyMs: 5429.8,
1194
+ outputTokensPerSecond: 226.21,
1195
+ errorRate: 0,
1196
+ samples: 3
1197
+ },
1198
+ "google/gemini-3.5-flash-lite": {
1199
+ measuredAt: "2026-08-29T16:51:33Z",
1200
+ latencyMs: 3515.8,
1201
+ p95LatencyMs: 4363.4,
1202
+ outputTokensPerSecond: 248.9,
1203
+ errorRate: 0,
1204
+ samples: 3
1205
+ },
1206
+ "google/gemini-3.6-flash": {
1207
+ measuredAt: "2026-08-29T16:51:33Z",
1208
+ latencyMs: 13020,
1209
+ p95LatencyMs: 15383.1,
1210
+ outputTokensPerSecond: 187.87,
1211
+ errorRate: 0,
1212
+ samples: 3
1213
+ },
1214
+ "minimax/minimax-m2.7": {
1215
+ measuredAt: "2026-08-29T16:51:33Z",
1216
+ latencyMs: 8761.1,
1217
+ p95LatencyMs: 10199.3,
1218
+ outputTokensPerSecond: 59.18,
1219
+ errorRate: 0,
1220
+ samples: 3
1221
+ },
1222
+ "minimax/minimax-m3": {
1223
+ measuredAt: "2026-08-29T16:51:33Z",
1224
+ latencyMs: 11101.9,
1225
+ p95LatencyMs: 26087.1,
1226
+ outputTokensPerSecond: 101.12,
1227
+ errorRate: 0,
1228
+ samples: 3
1229
+ },
1230
+ "moonshot/kimi-k3": {
1231
+ measuredAt: "2026-08-29T16:51:33Z",
1232
+ latencyMs: 24498.9,
1233
+ p95LatencyMs: 40365.3,
1234
+ outputTokensPerSecond: 25.11,
1235
+ errorRate: 0,
1236
+ samples: 3
1237
+ },
1238
+ "nvidia/mistral-nemotron": {
1239
+ measuredAt: "2026-08-29T16:51:33Z",
1240
+ latencyMs: 7349.6,
1241
+ p95LatencyMs: 9932.3,
1242
+ outputTokensPerSecond: 79.48,
1243
+ errorRate: 0.3333,
1244
+ samples: 3
1245
+ },
1246
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
1247
+ measuredAt: "2026-08-29T16:59:54Z",
1248
+ latencyMs: 9324.6,
1249
+ p95LatencyMs: 12992,
1250
+ outputTokensPerSecond: 64.96,
1251
+ errorRate: 0.3333,
1252
+ samples: 3
1253
+ },
1254
+ "nvidia/nemotron-nano-12b-v2-vl": {
1255
+ measuredAt: "2026-08-29T16:51:33Z",
1256
+ latencyMs: 5846.9,
1257
+ p95LatencyMs: 5846.9,
1258
+ outputTokensPerSecond: 87.57,
1259
+ errorRate: 0.6667,
1260
+ samples: 3
1261
+ },
1262
+ "nvidia/nemotron-nano-9b-v2": {
1263
+ measuredAt: "2026-08-29T16:51:33Z",
1264
+ latencyMs: 5282.5,
1265
+ p95LatencyMs: 5282.5,
1266
+ outputTokensPerSecond: 96.92,
1267
+ errorRate: 0.6667,
1268
+ samples: 3
1269
+ },
1270
+ "nvidia/step-3.7-flash": {
1271
+ measuredAt: "2026-08-29T16:51:33Z",
1272
+ latencyMs: 4617.4,
1273
+ p95LatencyMs: 5237.4,
1274
+ outputTokensPerSecond: 112.92,
1275
+ errorRate: 0.3333,
1276
+ samples: 3
1277
+ },
1278
+ "openai/chat-latest": {
1279
+ measuredAt: "2026-08-29T16:51:33Z",
1280
+ latencyMs: 3690.9,
1281
+ p95LatencyMs: 4344,
1282
+ outputTokensPerSecond: 111.85,
1283
+ errorRate: 0,
1284
+ samples: 3
1285
+ },
1286
+ "openai/gpt-4.1": {
1287
+ measuredAt: "2026-08-29T16:51:33Z",
1288
+ latencyMs: 3527.9,
1289
+ p95LatencyMs: 3831.7,
1290
+ outputTokensPerSecond: 141.27,
1291
+ errorRate: 0,
1292
+ samples: 3
1293
+ },
1294
+ "openai/gpt-4.1-mini": {
1295
+ measuredAt: "2026-08-29T16:51:33Z",
1296
+ latencyMs: 4268.2,
1297
+ p95LatencyMs: 5101.5,
1298
+ outputTokensPerSecond: 103.42,
1299
+ errorRate: 0,
1300
+ samples: 3
856
1301
  },
857
- "xai/grok-4.5": {
858
- contextWindow: 5e5,
859
- maxOutputTokens: 16384,
860
- supportsTools: true,
861
- supportsVision: true
1302
+ "openai/gpt-4.1-nano": {
1303
+ measuredAt: "2026-08-29T16:51:33Z",
1304
+ latencyMs: 3088.3,
1305
+ p95LatencyMs: 3369.2,
1306
+ outputTokensPerSecond: 150.31,
1307
+ errorRate: 0,
1308
+ samples: 3
862
1309
  },
863
- "zai/glm-5.1": {
864
- contextWindow: 2e5,
865
- maxOutputTokens: 128e3,
866
- supportsTools: true,
867
- supportsVision: false
1310
+ "openai/gpt-4o": {
1311
+ measuredAt: "2026-08-29T16:51:33Z",
1312
+ latencyMs: 2995.2,
1313
+ p95LatencyMs: 3174.2,
1314
+ outputTokensPerSecond: 171.32,
1315
+ errorRate: 0,
1316
+ samples: 3
868
1317
  },
869
- "zai/glm-5.2": {
870
- contextWindow: 1e6,
871
- maxOutputTokens: 262144,
872
- supportsTools: true,
873
- supportsVision: false
874
- }
875
- });
876
- var model_profiles_generated_default = {
877
- "openai/gpt-5.5": {
878
- measuredAt: "2026-07-21T10:21:31Z",
879
- latencyMs: 6243.1,
880
- p95LatencyMs: 9865,
881
- outputTokensPerSecond: 12.53,
1318
+ "openai/gpt-4o-mini": {
1319
+ measuredAt: "2026-08-29T16:51:33Z",
1320
+ latencyMs: 4751.5,
1321
+ p95LatencyMs: 4930.4,
1322
+ outputTokensPerSecond: 107.84,
882
1323
  errorRate: 0,
883
1324
  samples: 3
884
1325
  },
885
- "openai/gpt-5.4-pro": {
886
- measuredAt: "2026-07-21T10:21:31Z",
887
- latencyMs: 13015.5,
888
- p95LatencyMs: 23976.4,
889
- outputTokensPerSecond: 6.42,
1326
+ "openai/gpt-5-mini": {
1327
+ measuredAt: "2026-08-29T16:51:33Z",
1328
+ latencyMs: 4558.1,
1329
+ p95LatencyMs: 5081.9,
1330
+ outputTokensPerSecond: 113.25,
890
1331
  errorRate: 0,
891
1332
  samples: 3
892
1333
  },
893
- "openai/gpt-5.4-mini": {
894
- measuredAt: "2026-07-21T10:21:31Z",
895
- latencyMs: 5550,
896
- p95LatencyMs: 6595.7,
897
- outputTokensPerSecond: 11.96,
898
- errorRate: 0.3333,
1334
+ "openai/gpt-5.2": {
1335
+ measuredAt: "2026-08-29T16:51:33Z",
1336
+ latencyMs: 5436.6,
1337
+ p95LatencyMs: 5928.8,
1338
+ outputTokensPerSecond: 95.47,
1339
+ errorRate: 0,
899
1340
  samples: 3
900
1341
  },
901
1342
  "openai/gpt-5.3-codex": {
902
- measuredAt: "2026-07-21T10:21:31Z",
903
- latencyMs: 4617.1,
904
- p95LatencyMs: 5800.7,
905
- outputTokensPerSecond: 12.48,
906
- errorRate: 0,
1343
+ measuredAt: "2026-08-29T16:59:54Z",
1344
+ latencyMs: 15290.4,
1345
+ p95LatencyMs: 15290.4,
1346
+ outputTokensPerSecond: 33.49,
1347
+ errorRate: 0.6667,
907
1348
  samples: 3
908
1349
  },
909
- "anthropic/claude-opus-4.8": {
910
- measuredAt: "2026-07-21T10:21:31Z",
911
- latencyMs: 3915.1,
912
- p95LatencyMs: 6130.8,
913
- outputTokensPerSecond: 16.33,
1350
+ "openai/gpt-5.4": {
1351
+ measuredAt: "2026-08-29T16:51:33Z",
1352
+ latencyMs: 5596,
1353
+ p95LatencyMs: 5919.4,
1354
+ outputTokensPerSecond: 91.67,
914
1355
  errorRate: 0,
915
1356
  samples: 3
916
1357
  },
917
- "anthropic/claude-opus-4.6": {
918
- measuredAt: "2026-07-21T10:21:31Z",
919
- latencyMs: 3765.5,
920
- p95LatencyMs: 4257.2,
921
- outputTokensPerSecond: 14.18,
1358
+ "openai/gpt-5.4-mini": {
1359
+ measuredAt: "2026-08-29T16:51:33Z",
1360
+ latencyMs: 3377.8,
1361
+ p95LatencyMs: 3646.8,
1362
+ outputTokensPerSecond: 138.08,
922
1363
  errorRate: 0,
923
1364
  samples: 3
924
1365
  },
925
- "anthropic/claude-sonnet-4.6": {
926
- measuredAt: "2026-07-21T10:21:31Z",
927
- latencyMs: 3860.6,
928
- p95LatencyMs: 5093.5,
929
- outputTokensPerSecond: 13.85,
1366
+ "openai/gpt-5.4-nano": {
1367
+ measuredAt: "2026-08-29T16:51:33Z",
1368
+ latencyMs: 4040.4,
1369
+ p95LatencyMs: 4205.9,
1370
+ outputTokensPerSecond: 118.52,
930
1371
  errorRate: 0,
931
1372
  samples: 3
932
1373
  },
933
- "anthropic/claude-haiku-4.5": {
934
- measuredAt: "2026-07-21T10:21:31Z",
935
- latencyMs: 2734.9,
936
- p95LatencyMs: 3181.6,
937
- outputTokensPerSecond: 19.58,
1374
+ "openai/gpt-5.5": {
1375
+ measuredAt: "2026-08-29T16:51:33Z",
1376
+ latencyMs: 6367.8,
1377
+ p95LatencyMs: 7330.7,
1378
+ outputTokensPerSecond: 81.29,
938
1379
  errorRate: 0,
939
1380
  samples: 3
940
1381
  },
941
- "google/gemini-3.1-pro": {
942
- measuredAt: "2026-07-21T10:21:31Z",
943
- latencyMs: 13935.7,
944
- p95LatencyMs: 26675.3,
945
- outputTokensPerSecond: 77.47,
1382
+ "openai/gpt-5.6-luna": {
1383
+ measuredAt: "2026-08-29T16:51:33Z",
1384
+ latencyMs: 6064.5,
1385
+ p95LatencyMs: 7347.3,
1386
+ outputTokensPerSecond: 87.93,
946
1387
  errorRate: 0,
947
1388
  samples: 3
948
1389
  },
949
- "google/gemini-3.5-flash": {
950
- measuredAt: "2026-07-21T10:21:31Z",
951
- latencyMs: 4608.7,
952
- p95LatencyMs: 8420.9,
953
- outputTokensPerSecond: 57.88,
1390
+ "openai/gpt-5.6-luna-pro": {
1391
+ measuredAt: "2026-08-29T16:59:54Z",
1392
+ latencyMs: 13914.9,
1393
+ p95LatencyMs: 13914.9,
1394
+ outputTokensPerSecond: 36.79,
1395
+ errorRate: 0.6667,
1396
+ samples: 3
1397
+ },
1398
+ "openai/gpt-5.6-sol": {
1399
+ measuredAt: "2026-08-29T16:51:33Z",
1400
+ latencyMs: 7720.2,
1401
+ p95LatencyMs: 9108.2,
1402
+ outputTokensPerSecond: 67.47,
954
1403
  errorRate: 0,
955
1404
  samples: 3
956
1405
  },
957
- "google/gemini-3.1-flash-lite": {
958
- measuredAt: "2026-07-21T10:21:31Z",
959
- latencyMs: 4619.7,
960
- p95LatencyMs: 9927.1,
961
- outputTokensPerSecond: 42.01,
1406
+ "openai/gpt-5.6-sol-pro": {
1407
+ measuredAt: "2026-08-29T16:51:33Z",
1408
+ latencyMs: 11442.7,
1409
+ p95LatencyMs: 13363.1,
1410
+ outputTokensPerSecond: 148.75,
962
1411
  errorRate: 0,
963
1412
  samples: 3
964
1413
  },
965
- "google/gemini-2.5-flash": {
966
- measuredAt: "2026-07-21T10:21:31Z",
967
- latencyMs: 5506.9,
968
- p95LatencyMs: 11462.5,
969
- outputTokensPerSecond: 65.19,
1414
+ "openai/gpt-5.6-terra": {
1415
+ measuredAt: "2026-08-29T16:51:33Z",
1416
+ latencyMs: 4941,
1417
+ p95LatencyMs: 5095.3,
1418
+ outputTokensPerSecond: 103.69,
970
1419
  errorRate: 0,
971
1420
  samples: 3
972
1421
  },
973
- "deepseek/deepseek-v4-pro": {
974
- measuredAt: "2026-07-21T10:21:31Z",
975
- latencyMs: 6044.8,
976
- p95LatencyMs: 10782.3,
977
- outputTokensPerSecond: 22.47,
1422
+ "openai/gpt-5.6-terra-pro": {
1423
+ measuredAt: "2026-08-29T16:51:33Z",
1424
+ latencyMs: 3574.1,
1425
+ p95LatencyMs: 4126.3,
1426
+ outputTokensPerSecond: 133.59,
978
1427
  errorRate: 0,
979
1428
  samples: 3
980
1429
  },
981
- "deepseek/deepseek-reasoner": {
982
- measuredAt: "2026-07-21T10:21:31Z",
983
- latencyMs: 4111.9,
984
- p95LatencyMs: 5305.7,
985
- outputTokensPerSecond: 16.46,
1430
+ "openai/o1": {
1431
+ measuredAt: "2026-08-29T16:51:33Z",
1432
+ latencyMs: 4324.9,
1433
+ p95LatencyMs: 5838.1,
1434
+ outputTokensPerSecond: 125.86,
986
1435
  errorRate: 0,
987
1436
  samples: 3
988
1437
  },
989
- "deepseek/deepseek-chat": {
990
- measuredAt: "2026-07-21T10:21:31Z",
991
- latencyMs: 2648.6,
992
- p95LatencyMs: 3524.1,
993
- outputTokensPerSecond: 16.73,
1438
+ "openai/o3": {
1439
+ measuredAt: "2026-08-29T16:51:33Z",
1440
+ latencyMs: 5463.4,
1441
+ p95LatencyMs: 5613.1,
1442
+ outputTokensPerSecond: 93.8,
994
1443
  errorRate: 0,
995
1444
  samples: 3
996
1445
  },
997
- "moonshot/kimi-k2.7": {
998
- measuredAt: "2026-07-21T10:21:31Z",
999
- latencyMs: 4295.4,
1000
- p95LatencyMs: 6153.8,
1001
- outputTokensPerSecond: 18.54,
1446
+ "openai/o3-mini": {
1447
+ measuredAt: "2026-08-29T16:51:33Z",
1448
+ latencyMs: 2912.7,
1449
+ p95LatencyMs: 3092.1,
1450
+ outputTokensPerSecond: 176.49,
1002
1451
  errorRate: 0,
1003
1452
  samples: 3
1004
1453
  },
1005
- "qwen/qwen3.7-max": {
1006
- measuredAt: "2026-07-21T10:21:31Z",
1007
- latencyMs: 30729.4,
1008
- p95LatencyMs: 39622,
1009
- outputTokensPerSecond: 36.89,
1010
- errorRate: 0.3333,
1454
+ "openai/o4-mini": {
1455
+ measuredAt: "2026-08-29T16:51:33Z",
1456
+ latencyMs: 4958.7,
1457
+ p95LatencyMs: 5313,
1458
+ outputTokensPerSecond: 103.81,
1459
+ errorRate: 0,
1011
1460
  samples: 3
1012
1461
  },
1013
- "xai/grok-4.3": {
1014
- measuredAt: "2026-07-21T10:21:31Z",
1015
- latencyMs: 6946.1,
1016
- p95LatencyMs: 9495.4,
1017
- outputTokensPerSecond: 65.3,
1462
+ "qwen/qwen3.7-flash": {
1463
+ measuredAt: "2026-08-29T16:51:33Z",
1464
+ latencyMs: 3385.5,
1465
+ p95LatencyMs: 4042.7,
1466
+ outputTokensPerSecond: 153.94,
1018
1467
  errorRate: 0,
1019
1468
  samples: 3
1020
1469
  },
1021
- "xai/grok-4.20-reasoning": {
1022
- measuredAt: "2026-07-21T10:21:31Z",
1023
- latencyMs: 3472.4,
1024
- p95LatencyMs: 5332.4,
1025
- outputTokensPerSecond: 13.27,
1470
+ "qwen/qwen3.7-max": {
1471
+ measuredAt: "2026-08-29T16:51:33Z",
1472
+ latencyMs: 9387.1,
1473
+ p95LatencyMs: 10490.2,
1474
+ outputTokensPerSecond: 54.92,
1026
1475
  errorRate: 0,
1027
1476
  samples: 3
1028
1477
  },
1029
- "xai/grok-4.20-non-reasoning": {
1030
- measuredAt: "2026-07-21T10:21:31Z",
1031
- latencyMs: 5174.4,
1032
- p95LatencyMs: 6081.7,
1033
- outputTokensPerSecond: 10.21,
1034
- errorRate: 0.3333,
1478
+ "qwen/qwen3.7-plus": {
1479
+ measuredAt: "2026-08-29T16:51:33Z",
1480
+ latencyMs: 9766.6,
1481
+ p95LatencyMs: 9798.2,
1482
+ outputTokensPerSecond: 52.42,
1483
+ errorRate: 0,
1035
1484
  samples: 3
1036
1485
  },
1037
- "xai/grok-4-1-fast-reasoning": {
1038
- measuredAt: "2026-07-21T10:21:31Z",
1039
- latencyMs: 13148.2,
1040
- p95LatencyMs: 19104.2,
1041
- outputTokensPerSecond: 4.28,
1486
+ "tencent/hy3": {
1487
+ measuredAt: "2026-08-29T16:51:33Z",
1488
+ latencyMs: 6062.3,
1489
+ p95LatencyMs: 7070.2,
1490
+ outputTokensPerSecond: 87.3,
1042
1491
  errorRate: 0,
1043
1492
  samples: 3
1044
1493
  },
1045
- "minimax/minimax-m3": {
1046
- measuredAt: "2026-07-21T10:21:31Z",
1047
- latencyMs: 3385,
1048
- p95LatencyMs: 4247.2,
1049
- outputTokensPerSecond: 15.16,
1494
+ "xai/grok-4.3": {
1495
+ measuredAt: "2026-08-29T16:51:33Z",
1496
+ latencyMs: 9467.7,
1497
+ p95LatencyMs: 10087.9,
1498
+ outputTokensPerSecond: 48.36,
1050
1499
  errorRate: 0,
1051
1500
  samples: 3
1052
1501
  },
1053
- "minimax/minimax-m2.7": {
1054
- measuredAt: "2026-07-21T10:21:31Z",
1055
- latencyMs: 4596.7,
1056
- p95LatencyMs: 6884.6,
1057
- outputTokensPerSecond: 17.03,
1502
+ "xai/grok-4.5": {
1503
+ measuredAt: "2026-08-29T16:51:33Z",
1504
+ latencyMs: 13564.8,
1505
+ p95LatencyMs: 17351.9,
1506
+ outputTokensPerSecond: 60.71,
1058
1507
  errorRate: 0,
1059
1508
  samples: 3
1060
1509
  },
1061
- "zai/glm-5.2": {
1062
- measuredAt: "2026-07-21T10:21:31Z",
1063
- latencyMs: 4406.3,
1064
- p95LatencyMs: 6139.7,
1065
- outputTokensPerSecond: 10.41,
1510
+ "xai/grok-build-0.1": {
1511
+ measuredAt: "2026-08-29T16:51:33Z",
1512
+ latencyMs: 16394.8,
1513
+ p95LatencyMs: 18035.4,
1514
+ outputTokensPerSecond: 96.86,
1066
1515
  errorRate: 0,
1067
1516
  samples: 3
1068
1517
  },
1069
- "zai/glm-5.1": {
1070
- measuredAt: "2026-07-21T10:21:31Z",
1071
- latencyMs: 7775.4,
1072
- p95LatencyMs: 9182.1,
1073
- outputTokensPerSecond: 6.08,
1518
+ "xiaomi/mimo-v2.5-pro": {
1519
+ measuredAt: "2026-08-29T16:51:33Z",
1520
+ latencyMs: 12070.7,
1521
+ p95LatencyMs: 12386.8,
1522
+ outputTokensPerSecond: 42.44,
1074
1523
  errorRate: 0,
1075
1524
  samples: 3
1076
1525
  },
1077
1526
  "zai/glm-5": {
1078
- measuredAt: "2026-07-21T10:21:31Z",
1079
- latencyMs: 4159.4,
1080
- p95LatencyMs: 4992.7,
1081
- outputTokensPerSecond: 10.28,
1527
+ measuredAt: "2026-08-29T16:51:33Z",
1528
+ latencyMs: 6839.7,
1529
+ p95LatencyMs: 7261.4,
1530
+ outputTokensPerSecond: 75.16,
1531
+ errorRate: 0,
1532
+ samples: 3
1533
+ },
1534
+ "zai/glm-5-turbo": {
1535
+ measuredAt: "2026-08-29T16:51:33Z",
1536
+ latencyMs: 55348.5,
1537
+ p95LatencyMs: 114086.6,
1538
+ outputTokensPerSecond: 14.64,
1082
1539
  errorRate: 0,
1083
1540
  samples: 3
1084
1541
  },
1085
- "free/qwen3-coder-480b": {
1086
- measuredAt: "2026-07-21T10:21:31Z",
1087
- latencyMs: 2063.9,
1088
- p95LatencyMs: 3646.3,
1089
- outputTokensPerSecond: 39.8,
1542
+ "zai/glm-5.1": {
1543
+ measuredAt: "2026-08-29T16:51:33Z",
1544
+ latencyMs: 15658.4,
1545
+ p95LatencyMs: 17307.1,
1546
+ outputTokensPerSecond: 32.9,
1090
1547
  errorRate: 0,
1091
1548
  samples: 3
1092
1549
  },
1093
- "free/mistral-large-3-675b": {
1094
- measuredAt: "2026-07-21T10:21:31Z",
1095
- latencyMs: 3147.5,
1096
- p95LatencyMs: 5555.3,
1097
- outputTokensPerSecond: 27.76,
1550
+ "zai/glm-5.2": {
1551
+ measuredAt: "2026-08-29T16:51:33Z",
1552
+ latencyMs: 10308.5,
1553
+ p95LatencyMs: 15127.6,
1554
+ outputTokensPerSecond: 54.87,
1098
1555
  errorRate: 0,
1099
1556
  samples: 3
1100
1557
  },
1101
- "free/nemotron-3-nano-omni-30b-a3b-reasoning": {
1102
- measuredAt: "2026-07-21T10:21:31Z",
1103
- latencyMs: 6508.4,
1104
- p95LatencyMs: 14252.7,
1105
- outputTokensPerSecond: 68.26,
1558
+ "zai/glm-5.3": {
1559
+ measuredAt: "2026-08-29T16:51:33Z",
1560
+ latencyMs: 7272.4,
1561
+ p95LatencyMs: 7998.1,
1562
+ outputTokensPerSecond: 71.09,
1106
1563
  errorRate: 0,
1107
1564
  samples: 3
1108
1565
  },
1109
- "free/glm-4.7": {
1110
- measuredAt: "2026-07-21T10:21:31Z",
1111
- latencyMs: 2014.8,
1112
- p95LatencyMs: 3039.9,
1113
- outputTokensPerSecond: 39.92,
1566
+ "zai/glm-5.3-flash": {
1567
+ measuredAt: "2026-08-29T16:51:33Z",
1568
+ latencyMs: 10545.3,
1569
+ p95LatencyMs: 11672.4,
1570
+ outputTokensPerSecond: 49.01,
1114
1571
  errorRate: 0,
1115
1572
  samples: 3
1116
1573
  }
@@ -1124,11 +1581,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1124
1581
  latencyMs: 2305,
1125
1582
  outputTokensPerSecond: 140.6
1126
1583
  },
1127
- "anthropic/claude-opus-4.6": {
1128
- measuredAt: "2026-03-16T13:50:48Z",
1129
- latencyMs: 2139,
1130
- outputTokensPerSecond: 119.7
1131
- },
1132
1584
  "anthropic/claude-sonnet-4.6": {
1133
1585
  measuredAt: "2026-03-16T13:50:48Z",
1134
1586
  latencyMs: 2110,
@@ -1162,11 +1614,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1162
1614
  latencyMs: 1609,
1163
1615
  outputTokensPerSecond: 167.2
1164
1616
  },
1165
- "moonshot/kimi-k2.5": {
1166
- measuredAt: "2026-03-16T13:50:48Z",
1167
- latencyMs: 1646,
1168
- outputTokensPerSecond: 155.7
1169
- },
1170
1617
  "openai/gpt-4o-mini": {
1171
1618
  measuredAt: "2026-03-16T13:50:48Z",
1172
1619
  latencyMs: 2764,
@@ -1176,18 +1623,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1176
1623
  measuredAt: "2026-03-16T13:50:48Z",
1177
1624
  latencyMs: 7935,
1178
1625
  outputTokensPerSecond: 32.3
1179
- },
1180
- "xai/grok-4-1-fast-non-reasoning": {
1181
- measuredAt: "2026-03-16T13:50:48Z",
1182
- latencyMs: 1244,
1183
- outputTokensPerSecond: 205.8,
1184
- intelligenceIndex: 41
1185
- },
1186
- "xai/grok-4-1-fast-reasoning": {
1187
- measuredAt: "2026-03-16T13:50:48Z",
1188
- latencyMs: 1454,
1189
- outputTokensPerSecond: 176.2,
1190
- intelligenceIndex: 41
1191
1626
  }
1192
1627
  });
1193
1628
  function inferToolRequirement(prompt, _systemPrompt, toolChoice) {
@@ -1680,7 +2115,7 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1680
2115
  match(["gpt-5.3-codex"], 1),
1681
2116
  match(["claude-sonnet-4.6"], 0.94),
1682
2117
  match(["glm-5.2"], 0.9),
1683
- match(["kimi-k2.7", "deepseek-v4-pro"], 0.86)
2118
+ match(["deepseek-v4-pro"], 0.86)
1684
2119
  );
1685
2120
  case "reasoning":
1686
2121
  return Math.max(
@@ -1704,14 +2139,13 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1704
2139
  base,
1705
2140
  match(["gemini-3.5-flash"], 1),
1706
2141
  match(["grok-4.5"], 0.93),
1707
- match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9),
1708
- match(["kimi-k2.7"], 0.84)
2142
+ match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
1709
2143
  );
1710
2144
  case "vision":
1711
2145
  return Math.max(
1712
2146
  base,
1713
2147
  match(["gemini-3.1-pro"], 0.96),
1714
- match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k2.7", "grok-4.3"], 0.9)
2148
+ match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9)
1715
2149
  );
1716
2150
  case "long_context":
1717
2151
  return Math.max(
@@ -1723,17 +2157,18 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1723
2157
  );
1724
2158
  case "extraction": {
1725
2159
  const kimiExtractionAffinity = language === "zh" ? 1 : 0.9;
2160
+ const otherExtractionAffinity = language === "zh" ? 0.88 : 0.9;
1726
2161
  return Math.max(
1727
2162
  base,
1728
- match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], 0.9),
1729
- match(["claude-sonnet-5", "claude-sonnet-4.6"], 0.9),
1730
- match(["kimi-k3", "kimi-k2.7"], kimiExtractionAffinity)
2163
+ match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], otherExtractionAffinity),
2164
+ match(["claude-sonnet-5", "claude-sonnet-4.6"], otherExtractionAffinity),
2165
+ match(["kimi-k3"], kimiExtractionAffinity)
1731
2166
  );
1732
2167
  }
1733
2168
  default:
1734
2169
  return Math.max(
1735
2170
  base,
1736
- match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3", "kimi-k2.7"], 0.86)
2171
+ match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)
1737
2172
  );
1738
2173
  }
1739
2174
  }
@@ -1792,6 +2227,9 @@ function evidenceCandidates(task) {
1792
2227
  "deepseek/deepseek-v4-pro"
1793
2228
  ];
1794
2229
  }
2230
+ if (task === "extraction") {
2231
+ return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"];
2232
+ }
1795
2233
  if (task === "reasoning_math") {
1796
2234
  return [
1797
2235
  "google/gemini-3.5-flash",
@@ -1848,9 +2286,12 @@ var PortfolioStrategy = class {
1848
2286
  const targetTier = (features.taskType === "reasoning_mcq" || features.taskType === "reasoning_math") && (base.tier === "SIMPLE" || base.tier === "MEDIUM") ? "REASONING" : base.tier;
1849
2287
  const tierConfig = tierConfigs[targetTier];
1850
2288
  const configuredCandidates = tierConfig ? getFallbackChain(targetTier, tierConfigs) : [];
2289
+ const unavailable = new Set(options.unavailableModels ?? []);
1851
2290
  const chain = [
1852
2291
  .../* @__PURE__ */ new Set([...configuredCandidates, ...evidenceCandidates(features.taskType)])
1853
- ].filter((model2) => typeof model2 === "string" && model2.length > 0);
2292
+ ].filter(
2293
+ (model2) => typeof model2 === "string" && model2.length > 0 && !unavailable.has(model2)
2294
+ );
1854
2295
  const eligible = chain.filter(
1855
2296
  (model2) => isEligible(model2, features, maxOutputTokens, options)
1856
2297
  );
@@ -1927,13 +2368,7 @@ var PortfolioStrategy = class {
1927
2368
  ...eligibleCandidates.filter(
1928
2369
  (model2) => !scoredModels.includes(model2) && !webResearchFallbackOrder.includes(model2)
1929
2370
  )
1930
- ] : features.taskType === "tool_agent" || features.taskType === "tool_agent_parallel" && features.agentDomain !== "other" ? [
1931
- ...scoredModels,
1932
- ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
1933
- ] : [
1934
- ...scoredModels,
1935
- ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
1936
- ];
2371
+ ] : [...scoredModels, ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))];
1937
2372
  const model = ranked[0] ?? base.model;
1938
2373
  const selectedTierConfigs = {
1939
2374
  ...tierConfigs,
@@ -1970,7 +2405,7 @@ var PortfolioStrategy = class {
1970
2405
  }
1971
2406
  };
1972
2407
  var DEFAULT_ROUTING_CONFIG = {
1973
- version: "3.4",
2408
+ version: "3.5",
1974
2409
  strategy: "portfolio",
1975
2410
  portfolio: {
1976
2411
  auto: {
@@ -3024,185 +3459,249 @@ var DEFAULT_ROUTING_CONFIG = {
3024
3459
  // Below this confidence → ambiguous (null tier)
3025
3460
  confidenceThreshold: 0.7
3026
3461
  },
3462
+ // ─── Tier chains ───
3463
+ //
3464
+ // Catalog refresh 2026-08-29 (V3.5). Every chain below names only models
3465
+ // the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the
3466
+ // gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast
3467
+ // and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview,
3468
+ // the whole `free/*` namespace — were removed everywhere, including fallback
3469
+ // rungs, so a routed model is always one a user can find on blockrun.ai/models.
3470
+ //
3471
+ // Primaries moved only where portfolio.ts already carries calibration
3472
+ // evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for
3473
+ // agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no
3474
+ // trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash,
3475
+ // gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs;
3476
+ // promotion waits for a calibration run, because version recency is not a
3477
+ // quality signal.
3478
+ //
3479
+ // Latency figures in comments are the 2026-08-29 gateway probe
3480
+ // (model-profiles.generated.json); prices are the catalog list.
3027
3481
  // Auto (balanced) tier configs - current default smart routing
3028
- // Benchmark-tuned 2026-03-16: balancing quality (retention) + latency
3029
3482
  tiers: {
3030
3483
  SIMPLE: {
3031
3484
  primary: "google/gemini-2.5-flash",
3032
- // 1,238ms, IQ 20, 60% retention (best) — fast AND quality
3485
+ // $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer
3033
3486
  fallback: [
3034
3487
  "google/gemini-3-flash-preview",
3035
- // 1,398ms, IQ 46 — smarter fallback
3488
+ // $0.50/$3 — GPQA 5/6 in the 2026-07 calibration
3489
+ "google/gemini-3.5-flash-lite",
3490
+ // $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation
3036
3491
  "deepseek/deepseek-chat",
3037
- // V4 Flash chat ($0.20/$0.40, 1M ctx) — repriced 2026-04-24
3038
- "moonshot/kimi-k2.5",
3039
- // 1,646ms, IQ 47, strong quality
3492
+ // $0.14/$0.28, 1M ctx
3040
3493
  "google/gemini-3.1-flash-lite",
3041
- // $0.25/$1.50, 1M context — newest flash-lite
3042
- "google/gemini-2.5-flash-lite",
3043
- // 1,353ms, $0.10/$0.40
3494
+ // $0.25/$1.50, 1M ctx
3495
+ "openai/gpt-5.6-luna",
3496
+ // $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30)
3044
3497
  "openai/gpt-5.4-nano",
3045
- // $0.20/$1.25, 1M context
3046
- "xai/grok-4-fast-non-reasoning",
3047
- // 1,143ms, $0.20/$0.50 — fast fallback
3048
- "free/gpt-oss-120b"
3049
- // 1,252ms, FREE fallback (hidden from /v1/models but direct calls work)
3498
+ // $0.20/$1.25, 1M ctx
3499
+ "google/gemini-2.5-flash-lite",
3500
+ // $0.10/$0.40
3501
+ "nvidia/step-3.7-flash"
3502
+ // FREE backstop — NVIDIA free tier (probed 2026-08-21)
3050
3503
  ]
3051
3504
  },
3052
3505
  MEDIUM: {
3053
- primary: "moonshot/kimi-k2.7",
3054
- // $0.95/$4.00, 256K ctx, multi-modal + reasoning — Moonshot flagship; promoted from K2.6 (2026-06-14) after BlockRun added K2.7 + hid K2.6. Same price as K2.6.
3506
+ // Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the
3507
+ // calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts).
3508
+ primary: "google/gemini-3.5-flash",
3509
+ // $1.50/$9, 1M ctx, vision + tools
3055
3510
  fallback: [
3056
- "moonshot/kimi-k2.6",
3057
- // identical-cost in-family hot swap (K2.6 still routable)
3058
- "moonshot/kimi-k2.5",
3059
- // $0.60/$3.00 — graceful-degradation backstop
3511
+ "google/gemini-3.6-flash",
3512
+ // $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration
3513
+ "zai/glm-5.3-flash",
3514
+ // $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27
3515
+ "openai/gpt-5.6-terra",
3516
+ // $2/$12, 1M ctx — GPT-5.6 balanced tier
3060
3517
  "google/gemini-3-flash-preview",
3061
- // 1,398ms, IQ 46 — nearly same IQ, faster + cheaper
3518
+ // $0.50/$3
3062
3519
  "deepseek/deepseek-chat",
3063
- // 1,431ms, IQ 32, 41% retention
3520
+ // $0.14/$0.28
3064
3521
  "google/gemini-2.5-flash",
3065
- // 1,238ms, 60% retention
3522
+ // $0.30/$2.50
3523
+ "minimax/minimax-m3",
3524
+ // $0.30/$1.20, 1M ctx
3066
3525
  "google/gemini-3.1-flash-lite",
3067
- // $0.25/$1.50, 1M context
3068
- "google/gemini-2.5-flash-lite",
3069
- // 1,353ms, $0.10/$0.40
3070
- "xai/grok-4-1-fast-non-reasoning",
3071
- // 1,244ms, fast fallback
3072
- "xai/grok-3-mini"
3073
- // 1,202ms, $0.30/$0.50
3526
+ // $0.25/$1.50
3527
+ "openai/gpt-5.6-luna",
3528
+ // $0.20/$1.20
3529
+ "google/gemini-2.5-flash-lite"
3530
+ // $0.10/$0.40
3074
3531
  ]
3075
3532
  },
3076
3533
  COMPLEX: {
3077
3534
  primary: "google/gemini-3.1-pro",
3078
- // 1,609ms, IQ 57 — fast flagship quality
3535
+ // $2/$12 — proven long-context flagship (portfolio.ts long_context lead)
3079
3536
  fallback: [
3080
- "google/gemini-3-flash-preview",
3081
- // 1,398ms, IQ 46 — fast + smart
3082
- "xai/grok-4-0709",
3083
- // 1,348ms, IQ 41
3084
- "google/gemini-2.5-pro",
3085
- // 1,294ms
3537
+ "google/gemini-3.6-flash",
3538
+ // $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here)
3539
+ "google/gemini-3.5-flash",
3540
+ // $1.50/$9 — calibrated
3086
3541
  "anthropic/claude-sonnet-5",
3087
- // near-Opus quality at Sonnet cost, 1M ctx
3542
+ // $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated
3543
+ "xai/grok-4.5",
3544
+ // $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden)
3545
+ "google/gemini-2.5-pro",
3546
+ // $1.25/$10
3088
3547
  "anthropic/claude-sonnet-4.6",
3089
- // 2,110ms, IQ 52 — quality fallback
3090
- "deepseek/deepseek-chat",
3091
- // 1,431ms, IQ 32
3092
- "google/gemini-2.5-flash",
3093
- // 1,238ms, IQ 20 — cheap last resort
3548
+ // $3/$15
3094
3549
  "openai/gpt-5.6-terra",
3095
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3550
+ // $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202)
3096
3551
  "openai/gpt-5.5",
3097
- // Prior OpenAI flagship — 1M+ ctx, native agent + computer use; benchmark TBD
3098
- "openai/gpt-5.4"
3099
- // 6,213ms, IQ 57 — previous flagship, benchmarked
3552
+ // $5/$30 — prior OpenAI flagship
3553
+ "openai/gpt-5.4",
3554
+ // $2.50/$15 — previous flagship, benchmarked
3555
+ "zai/glm-5.3",
3556
+ // $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19
3557
+ "moonshot/kimi-k3",
3558
+ // $3/$15, 1M ctx — Moonshot flagship (K2.7 successor)
3559
+ "deepseek/deepseek-v4-pro",
3560
+ // $0.435/$0.87 — strongest open-weight reasoner
3561
+ "deepseek/deepseek-chat",
3562
+ // $0.14/$0.28 — cheap last resort
3563
+ "google/gemini-2.5-flash"
3564
+ // $0.30/$2.50
3100
3565
  ]
3101
3566
  },
3102
3567
  REASONING: {
3103
- primary: "xai/grok-4-1-fast-reasoning",
3104
- // 1,454ms, $0.20/$0.50
3568
+ // Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek
3569
+ // Reasoner is the cheapest listed reasoner at the same 1M context.
3570
+ primary: "deepseek/deepseek-reasoner",
3571
+ // $0.14/$0.28, 1M ctx
3105
3572
  fallback: [
3106
- "xai/grok-4-fast-reasoning",
3107
- // 1,298ms, $0.20/$0.50
3108
- "deepseek/deepseek-reasoner",
3109
- // V4 Flash thinking ($0.20/$0.40, 1M ctx)
3110
3573
  "deepseek/deepseek-v4-pro",
3111
- // V4 Pro flagship ($0.50/$1.00 promo through 2026-05-31, list $2/$4) — strongest open-weight reasoner
3574
+ // $0.435/$0.87 — calibrated reasoning band 0.95
3575
+ "xai/grok-4.3",
3576
+ // $1.50/$4, 1M ctx — xAI reasoning model, vision
3577
+ "qwen/qwen3.7-plus",
3578
+ // $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed)
3579
+ "google/gemini-3.5-flash",
3580
+ // $1.50/$9 — MGSM 5/5
3112
3581
  "openai/o4-mini",
3113
- // 2,328ms ($1.10/$4.40)
3582
+ // $1.10/$4.40
3114
3583
  "openai/o3"
3115
- // 2,862ms
3584
+ // $2/$8
3116
3585
  ]
3117
3586
  }
3118
3587
  },
3119
3588
  // Eco tier configs - absolute cheapest (blockrun/eco)
3120
3589
  ecoTiers: {
3121
3590
  SIMPLE: {
3122
- primary: "free/gpt-oss-120b",
3123
- // FREE! $0.00/$0.00 — heavy user default
3591
+ primary: "nvidia/step-3.7-flash",
3592
+ // FREE — NVIDIA free tier flagship
3124
3593
  fallback: [
3125
- "free/gpt-oss-20b",
3126
- // FREE — smaller, faster
3127
- // deepseek-v4-flash and seed-oss-36b sat here until NVIDIA EOL'd them
3128
- // (410; 2026-08-12 and 2026-08-03 respectively). gpt-oss-120b/20b already
3129
- // head this chain, so the rungs are dropped, not retargeted.
3130
- "google/gemini-3.1-flash-lite",
3131
- // $0.25/$1.50 — newest flash-lite
3132
- "openai/gpt-5.4-nano",
3133
- // $0.20/$1.25 — fast nano
3594
+ "nvidia/nemotron-nano-9b-v2",
3595
+ // FREE — compact + fast, high-volume light tasks
3596
+ // The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash
3597
+ // 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400
3598
+ // 2026-08-21). Each retirement retargets the two free rungs to the
3599
+ // current free tier; the paid rungs below never move.
3134
3600
  "google/gemini-2.5-flash-lite",
3135
- // $0.10/$0.40
3136
- "xai/grok-4-fast-non-reasoning"
3137
- // $0.20/$0.50
3601
+ // $0.10/$0.40 — cheapest paid rung
3602
+ "zai/glm-5.3-flash",
3603
+ // $0.15/$0.50, 1M ctx, vision + tools
3604
+ "openai/gpt-5.6-luna",
3605
+ // $0.20/$1.20, 1M ctx
3606
+ "openai/gpt-5.4-nano",
3607
+ // $0.20/$1.25
3608
+ "google/gemini-3.1-flash-lite"
3609
+ // $0.25/$1.50
3138
3610
  ]
3139
3611
  },
3140
3612
  MEDIUM: {
3141
- primary: "google/gemini-3.1-flash-lite",
3142
- // $0.25/$1.50 — newest flash-lite
3613
+ primary: "zai/glm-5.3-flash",
3614
+ // $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model
3143
3615
  fallback: [
3616
+ "deepseek/deepseek-chat",
3617
+ // $0.14/$0.28
3618
+ "google/gemini-3.1-flash-lite",
3619
+ // $0.25/$1.50
3620
+ "openai/gpt-5.6-luna",
3621
+ // $0.20/$1.20
3144
3622
  "openai/gpt-5.4-nano",
3145
3623
  // $0.20/$1.25
3146
3624
  "google/gemini-2.5-flash-lite",
3147
3625
  // $0.10/$0.40
3148
- "xai/grok-4-fast-non-reasoning",
3149
3626
  "google/gemini-2.5-flash"
3627
+ // $0.30/$2.50
3150
3628
  ]
3151
3629
  },
3152
3630
  COMPLEX: {
3153
- primary: "google/gemini-3.1-flash-lite",
3154
- // $0.25/$1.50
3631
+ primary: "zai/glm-5.3-flash",
3632
+ // $0.15/$0.50, 1M ctx
3155
3633
  fallback: [
3156
- "google/gemini-2.5-flash-lite",
3157
- "xai/grok-4-0709",
3158
- "google/gemini-2.5-flash",
3159
- "deepseek/deepseek-chat"
3634
+ "deepseek/deepseek-chat",
3635
+ // $0.14/$0.28, 1M ctx
3636
+ "minimax/minimax-m3",
3637
+ // $0.30/$1.20, 1M ctx
3638
+ "deepseek/deepseek-v4-pro",
3639
+ // $0.435/$0.87
3640
+ "google/gemini-3.1-flash-lite",
3641
+ // $0.25/$1.50
3642
+ "google/gemini-2.5-flash"
3643
+ // $0.30/$2.50
3160
3644
  ]
3161
3645
  },
3162
3646
  REASONING: {
3163
- primary: "xai/grok-4-1-fast-reasoning",
3164
- // $0.20/$0.50
3647
+ primary: "deepseek/deepseek-reasoner",
3648
+ // $0.14/$0.28, 1M ctx — cheapest listed reasoner
3165
3649
  fallback: [
3166
- "xai/grok-4-fast-reasoning",
3167
- "deepseek/deepseek-reasoner",
3168
- // V4 Flash thinking — $0.20/$0.40
3169
- "deepseek/deepseek-v4-pro"
3170
- // V4 Pro flagship — $0.50/$1.00 promo, post-promo $2/$4
3650
+ "deepseek/deepseek-v4-pro",
3651
+ // $0.435/$0.87
3652
+ "qwen/qwen3.7-plus",
3653
+ // $0.32/$1.28 — reasoning
3654
+ "minimax/minimax-m3",
3655
+ // $0.30/$1.20 — reasoning + coding
3656
+ "zai/glm-5.3-flash"
3657
+ // $0.15/$0.50 — reasoning tokens alongside content
3171
3658
  ]
3172
3659
  }
3173
3660
  },
3174
3661
  // Premium tier configs - best quality (blockrun/premium)
3175
- // codex=complex coding, kimi=simple coding, sonnet=reasoning/instructions, opus=architecture/PM/audits
3662
+ // codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits
3176
3663
  premiumTiers: {
3177
3664
  SIMPLE: {
3178
- primary: "moonshot/kimi-k2.7",
3179
- // $0.95/$4.00 - Moonshot flagship (256K ctx, multi-modal + reasoning); promoted from K2.6 (2026-06-14), same price
3665
+ // Was moonshot/kimi-k2.7 (hidden 2026-08).
3666
+ primary: "google/gemini-3.5-flash",
3667
+ // $1.50/$9, 1M ctx, vision + tools — calibrated
3180
3668
  fallback: [
3181
- "moonshot/kimi-k2.6",
3182
- // identical-cost in-family hot swap (K2.6 still routable)
3183
- "moonshot/kimi-k2.5",
3184
- // $0.60/$3.00 - proven reliable backstop when Moonshot direct API falters
3185
- "google/gemini-2.5-flash",
3186
- // 60% retention, fast growth
3669
+ "google/gemini-3.6-flash",
3670
+ // $1.50/$7.50 — newest Flash
3187
3671
  "anthropic/claude-haiku-4.5",
3188
- "google/gemini-2.5-flash-lite",
3672
+ // $1/$5
3673
+ "zai/glm-5.3",
3674
+ // $1.40/$4.40, 1M ctx
3675
+ "google/gemini-2.5-flash",
3676
+ // $0.30/$2.50
3677
+ "google/gemini-3.5-flash-lite",
3678
+ // $0.30/$2.50
3189
3679
  "deepseek/deepseek-chat"
3680
+ // $0.14/$0.28
3190
3681
  ]
3191
3682
  },
3192
3683
  MEDIUM: {
3193
3684
  primary: "openai/gpt-5.3-codex",
3194
- // $1.75/$14 - 400K context, 128K output, replaces 5.2
3685
+ // $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts)
3195
3686
  fallback: [
3196
- "moonshot/kimi-k2.7",
3197
- // Moonshot flagship
3198
- "moonshot/kimi-k2.6",
3199
- "moonshot/kimi-k2.5",
3200
- "google/gemini-2.5-flash",
3201
- // 60% retention, good coding capability
3202
- "google/gemini-2.5-pro",
3203
- "xai/grok-4-0709",
3204
3687
  "anthropic/claude-sonnet-5",
3205
- "anthropic/claude-sonnet-4.6"
3688
+ // $3/$15 — code_agent band 0.98
3689
+ "moonshot/kimi-k3",
3690
+ // $3/$15, 1M ctx — Moonshot flagship
3691
+ "zai/glm-5.3",
3692
+ // $1.40/$4.40 — long-horizon coding
3693
+ "google/gemini-3.6-flash",
3694
+ // $1.50/$7.50
3695
+ "google/gemini-3.5-flash",
3696
+ // $1.50/$9
3697
+ "google/gemini-2.5-pro",
3698
+ // $1.25/$10
3699
+ "xai/grok-4.5",
3700
+ // $2.50/$9
3701
+ "anthropic/claude-sonnet-4.6",
3702
+ // $3/$15
3703
+ "openai/gpt-5.6-terra"
3704
+ // $2/$12
3206
3705
  ]
3207
3706
  },
3208
3707
  COMPLEX: {
@@ -3212,8 +3711,8 @@ var DEFAULT_ROUTING_CONFIG = {
3212
3711
  // Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking)
3213
3712
  // Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is
3214
3713
  // also prone to "high demand" 503s (correlated failure — everyone falls
3215
- // back to Google at the same time). Prefer xAI Grok → Moonshot → OpenAI
3216
- // flagship → DeepSeek → NVIDIA free instead.
3714
+ // back to Google at the same time). Prefer in-family → xAI → Moonshot →
3715
+ // OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead.
3217
3716
  fallback: [
3218
3717
  "anthropic/claude-opus-5",
3219
3718
  // in-family hot swap first (half the price, 1M ctx + adaptive thinking)
@@ -3221,52 +3720,54 @@ var DEFAULT_ROUTING_CONFIG = {
3221
3720
  // in-family hot swap (identical cost to 5)
3222
3721
  "anthropic/claude-opus-4.7",
3223
3722
  // in-family hot swap (identical cost to 4.8)
3224
- "anthropic/claude-opus-4.6",
3225
- // in-family hot swap
3226
3723
  "anthropic/claude-sonnet-5",
3227
3724
  // Sonnet-tier drop-down, near-Opus quality
3228
3725
  "anthropic/claude-sonnet-4.6",
3229
3726
  "xai/grok-4.5",
3230
- // xAI flagship — 503-resistant, direct-xAI SKU (added 2026-07-14)
3231
- "xai/grok-4-0709",
3232
- // 503-resistant flagship
3233
- "moonshot/kimi-k2.7",
3727
+ // xAI flagship — 503-resistant, direct-xAI SKU
3728
+ "moonshot/kimi-k3",
3234
3729
  // Moonshot flagship, independent infra
3235
- "moonshot/kimi-k2.6",
3236
- "moonshot/kimi-k2.5",
3237
3730
  "openai/gpt-5.6-terra",
3238
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3731
+ // GPT-5.6 balanced tier — stable (Sol excluded: #202)
3239
3732
  "openai/gpt-5.5",
3240
3733
  // Prior OpenAI flagship — 1M+ ctx, native agent + computer use
3241
3734
  "openai/gpt-5.4",
3242
3735
  // Previous flagship (slow but stable, benchmarked at 6,213ms)
3243
3736
  "openai/gpt-5.3-codex",
3737
+ "zai/glm-5.3",
3738
+ // Z.AI flagship, 1M ctx
3739
+ "deepseek/deepseek-v4-pro",
3740
+ // strongest open-weight reasoner
3244
3741
  "deepseek/deepseek-chat",
3245
3742
  // Cheap, reliable
3246
- "free/gpt-oss-120b"
3247
- // NVIDIA free ultimate backstop (was seed-oss-36b; EOL'd 2026-08-03)
3743
+ "nvidia/step-3.7-flash"
3744
+ // NVIDIA free ultimate backstop
3248
3745
  ]
3249
3746
  },
3250
3747
  REASONING: {
3251
- primary: "anthropic/claude-sonnet-4.6",
3252
- // 2,110ms, $3/$15 - best for reasoning/instructions
3748
+ // Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both,
3749
+ // plus Sonnet 5's tau2/BrowseComp trajectory evidence).
3750
+ primary: "anthropic/claude-sonnet-5",
3751
+ // $3/$15, 1M ctx, adaptive thinking
3253
3752
  fallback: [
3254
- "anthropic/claude-sonnet-5",
3255
- // in-family hot swap — same cost, adaptive thinking, 1M ctx
3753
+ "anthropic/claude-sonnet-4.6",
3754
+ // in-family hot swap — same cost
3256
3755
  "anthropic/claude-opus-5",
3257
3756
  // Newest flagship Opus w/ adaptive thinking
3258
3757
  "anthropic/claude-opus-4.8",
3259
3758
  // Prior flagship Opus — identical cost to 5
3260
3759
  "anthropic/claude-opus-4.7",
3261
3760
  // Flagship Opus w/ adaptive thinking
3262
- "anthropic/claude-opus-4.6",
3263
- // 2,139ms
3264
- "xai/grok-4-1-fast-reasoning",
3265
- // 1,454ms, cheap fast reasoning
3761
+ "xai/grok-4.5",
3762
+ // reasoning band 0.94
3763
+ "deepseek/deepseek-v4-pro",
3764
+ // reasoning band 0.95
3765
+ "xai/grok-4.3",
3766
+ // $1.50/$4 — xAI reasoning model
3266
3767
  "openai/o4-mini",
3267
- // 2,328ms ($1.10/$4.40)
3768
+ // $1.10/$4.40
3268
3769
  "openai/o3"
3269
- // 2,862ms
3770
+ // $2/$8
3270
3771
  ]
3271
3772
  }
3272
3773
  },
@@ -3276,101 +3777,102 @@ var DEFAULT_ROUTING_CONFIG = {
3276
3777
  primary: "openai/gpt-4o-mini",
3277
3778
  // $0.15/$0.60 - best tool compliance at lowest cost
3278
3779
  fallback: [
3279
- "moonshot/kimi-k2.5",
3280
- // 1,646ms, strong tool use quality
3780
+ "openai/gpt-5.6-luna",
3781
+ // $0.20/$1.20 — lightweight agentic tier of GPT-5.6
3782
+ "zai/glm-5.3-flash",
3783
+ // $0.15/$0.50 — tool calls verified live 2026-08-27
3281
3784
  "anthropic/claude-haiku-4.5",
3282
- // 2,305ms
3283
- "xai/grok-4-1-fast-non-reasoning"
3284
- // 1,244ms, fast fallback
3785
+ // $1/$5
3786
+ "google/gemini-2.5-flash"
3787
+ // $0.30/$2.50
3285
3788
  ]
3286
3789
  },
3287
3790
  MEDIUM: {
3288
- primary: "moonshot/kimi-k2.7",
3289
- // $0.95/$4.00 — Moonshot flagship, strong tool use; promoted from K2.6 (2026-06-14) after BlockRun added K2.7 + hid K2.6. Same price.
3791
+ // Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the
3792
+ // Terminal-Bench and tau2 trajectory evidence in portfolio.ts.
3793
+ primary: "openai/gpt-5-mini",
3794
+ // $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline
3290
3795
  fallback: [
3291
- "moonshot/kimi-k2.6",
3292
- // identical-cost in-family hot swap (K2.6 still routable)
3293
- "moonshot/kimi-k2.5",
3294
- // $0.60/$3.00 — graceful-degradation backstop
3295
- "xai/grok-4-1-fast-non-reasoning",
3296
- // 1,244ms, fast fallback
3796
+ "google/gemini-3.5-flash",
3797
+ // $1.50/$9 — tool_agent band 0.88
3798
+ "zai/glm-5.3-flash",
3799
+ // $0.15/$0.50 — tools verified
3800
+ "openai/gpt-5.6-terra",
3801
+ // $2/$12
3297
3802
  "openai/gpt-4o-mini",
3298
- // 2,764ms, reliable tool calling
3803
+ // $0.15/$0.60 — reliable tool calling
3299
3804
  "anthropic/claude-haiku-4.5",
3300
- // 2,305ms
3301
- "deepseek/deepseek-chat"
3302
- // 1,431ms
3805
+ // $1/$5
3806
+ "deepseek/deepseek-chat",
3807
+ // $0.14/$0.28
3808
+ "moonshot/kimi-k3"
3809
+ // $3/$15 — tool_agent band 0.85
3303
3810
  ]
3304
3811
  },
3305
3812
  COMPLEX: {
3306
- primary: "anthropic/claude-sonnet-4.6",
3307
- // 2,110ms — best agentic quality
3813
+ // Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0,
3814
+ // Terminal-Bench safety band lead (portfolio.ts).
3815
+ primary: "anthropic/claude-sonnet-5",
3816
+ // $3/$15 — best agentic quality per trajectory evidence
3308
3817
  // Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s
3309
3818
  // correlate with Anthropic outages (everyone falls back together).
3310
3819
  // Prefer 503-resistant providers first.
3311
3820
  fallback: [
3312
- "anthropic/claude-sonnet-5",
3313
- // in-family hot swap — same cost, near-Opus agentic quality
3821
+ "anthropic/claude-sonnet-4.6",
3822
+ // in-family hot swap — same cost
3314
3823
  "anthropic/claude-opus-5",
3315
3824
  // Newest flagship Opus — in-family hot swap
3316
3825
  "anthropic/claude-opus-4.8",
3317
3826
  // Prior flagship Opus — identical cost to 5
3318
3827
  "anthropic/claude-opus-4.7",
3319
3828
  // Flagship Opus — in-family hot swap
3320
- "anthropic/claude-opus-4.6",
3321
- // 2,139ms
3322
- "xai/grok-4-0709",
3323
- // 1,348ms — strong tool use, independent infra
3324
- "moonshot/kimi-k2.7",
3325
- // Moonshot flagship — strong tool use, independent infra
3326
- "moonshot/kimi-k2.5",
3327
- // cost-stability backstop
3829
+ "xai/grok-4.5",
3830
+ // xAI flagship — strong tool use, independent infra
3831
+ "moonshot/kimi-k3",
3832
+ // Moonshot flagship — independent infra
3328
3833
  "openai/gpt-5.6-terra",
3329
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3834
+ // GPT-5.6 balanced tier — stable (Sol excluded: #202)
3330
3835
  "openai/gpt-5.5",
3331
3836
  // Prior flagship — native agent + computer use (exactly the agentic-tier use case)
3332
3837
  "openai/gpt-5.4",
3333
- // Previous flagship — 6,213ms, reliable
3838
+ // Previous flagship — reliable
3839
+ "openai/gpt-5.3-codex",
3840
+ // code_agent lead
3841
+ "zai/glm-5.3",
3842
+ // long-horizon coding
3843
+ "deepseek/deepseek-v4-pro",
3844
+ // retail high-risk 3/3
3334
3845
  "deepseek/deepseek-chat",
3335
- // 1,431ms — cheap, reliable
3336
- "free/gpt-oss-120b"
3337
- // NVIDIA free ultimate backstop (was seed-oss-36b; EOL'd 2026-08-03)
3846
+ // cheap, reliable
3847
+ "nvidia/step-3.7-flash"
3848
+ // NVIDIA free ultimate backstop
3338
3849
  ]
3339
3850
  },
3340
3851
  REASONING: {
3341
- primary: "anthropic/claude-sonnet-4.6",
3342
- // 2,110ms — strong tool use + reasoning
3852
+ primary: "anthropic/claude-sonnet-5",
3853
+ // $3/$15 — strong tool use + adaptive thinking
3343
3854
  fallback: [
3344
- "anthropic/claude-sonnet-5",
3345
- // in-family hot swap — same cost, adaptive thinking
3855
+ "anthropic/claude-sonnet-4.6",
3856
+ // in-family hot swap — same cost
3346
3857
  "anthropic/claude-opus-5",
3347
3858
  // Newest flagship Opus w/ adaptive thinking
3348
3859
  "anthropic/claude-opus-4.8",
3349
3860
  // Prior flagship Opus — identical cost to 5
3350
3861
  "anthropic/claude-opus-4.7",
3351
3862
  // Flagship Opus w/ adaptive thinking
3352
- "anthropic/claude-opus-4.6",
3353
- // 2,139ms
3354
- "xai/grok-4-1-fast-reasoning",
3355
- // 1,454ms
3863
+ "xai/grok-4.5",
3864
+ // reasoning band 0.94
3865
+ "deepseek/deepseek-v4-pro",
3866
+ // reasoning band 0.95
3356
3867
  "deepseek/deepseek-reasoner"
3357
- // 1,454ms
3868
+ // $0.14/$0.28
3358
3869
  ]
3359
3870
  }
3360
3871
  },
3361
- // Time-windowed promotions — auto-applied when active, ignored when expired
3362
- promotions: [
3363
- {
3364
- name: "GLM-5.1 Launch Promo ($0.001 flat)",
3365
- startDate: "2026-04-01",
3366
- endDate: "2026-05-01",
3367
- tierOverrides: {
3368
- SIMPLE: { primary: "zai/glm-5.1" }
3369
- },
3370
- profiles: ["auto"]
3371
- // only auto profile — eco stays free, premium stays premium
3372
- }
3373
- ],
3872
+ // Time-windowed promotions — auto-applied when active, ignored when expired.
3873
+ // The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and
3874
+ // has expired; the list is kept empty so the mechanism stays wired.
3875
+ promotions: [],
3374
3876
  overrides: {
3375
3877
  maxTokensForceComplex: 1e5,
3376
3878
  structuredOutputMinTier: "MEDIUM",
@@ -3712,7 +4214,7 @@ async function createSolanaPaymentPayload(secretKey, fromAddress, recipient, amo
3712
4214
  }
3713
4215
  return null;
3714
4216
  };
3715
- let entry = await getBlockhashEntry(connection, rpcUrl, false);
4217
+ let entry = await getBlockhashEntry(connection, rpcUrl, options.forceFreshBlockhash ?? false);
3716
4218
  let serializedTx = findDistinctTx(entry);
3717
4219
  if (serializedTx === null) {
3718
4220
  entry = await getBlockhashEntry(connection, rpcUrl, true);
@@ -3963,7 +4465,7 @@ function getCostSummary() {
3963
4465
  }
3964
4466
 
3965
4467
  // src/version.ts
3966
- var SDK_VERSION = "3.13.2";
4468
+ var SDK_VERSION = "3.13.5";
3967
4469
  var USER_AGENT = `blockrun-ts/${SDK_VERSION}`;
3968
4470
 
3969
4471
  // src/client.ts
@@ -8617,6 +9119,68 @@ async function getOrCreateSolanaWallet() {
8617
9119
  var SOLANA_API_URL = "https://sol.blockrun.ai/api";
8618
9120
  var DEFAULT_MAX_TOKENS2 = 1024;
8619
9121
  var DEFAULT_TIMEOUT14 = 6e4;
9122
+ var STALE_BLOCKHASH_RETRY_BACKOFFS_MS = [500, 2e3];
9123
+ var MAX_PAYMENT_FAILURE_BYTES = 64 * 1024;
9124
+ var SafeStaleBlockhashError = class extends PaymentError {
9125
+ constructor() {
9126
+ super("Payment verification used an expired Solana blockhash; retrying with a fresh quote.");
9127
+ this.name = "SafeStaleBlockhashError";
9128
+ }
9129
+ };
9130
+ function normalizePaymentSignal(value) {
9131
+ return typeof value === "string" ? value.toLowerCase().replace(/[_\-\s:]/g, "") : "";
9132
+ }
9133
+ async function readPaymentFailureBody(response) {
9134
+ const reader = response.body?.getReader();
9135
+ if (!reader) return "";
9136
+ const decoder = new TextDecoder();
9137
+ let total = 0;
9138
+ let text = "";
9139
+ try {
9140
+ for (; ; ) {
9141
+ const { done, value } = await reader.read();
9142
+ if (done) return text + decoder.decode();
9143
+ total += value.byteLength;
9144
+ if (total > MAX_PAYMENT_FAILURE_BYTES) {
9145
+ void reader.cancel();
9146
+ return null;
9147
+ }
9148
+ text += decoder.decode(value, { stream: true });
9149
+ }
9150
+ } catch {
9151
+ return null;
9152
+ }
9153
+ }
9154
+ async function isSafeStaleBlockhashResponse(response) {
9155
+ const length = Number(response.headers.get("content-length") || "0");
9156
+ if (Number.isFinite(length) && length > MAX_PAYMENT_FAILURE_BYTES) return false;
9157
+ const text = await readPaymentFailureBody(response);
9158
+ if (text === null) return false;
9159
+ let body;
9160
+ try {
9161
+ const parsed = JSON.parse(text);
9162
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
9163
+ body = parsed;
9164
+ } catch {
9165
+ return false;
9166
+ }
9167
+ const nested = body.error && typeof body.error === "object" && !Array.isArray(body.error) ? body.error : void 0;
9168
+ const errorLabel = typeof body.error === "string" ? body.error : "";
9169
+ const code = normalizePaymentSignal(body.code ?? nested?.code);
9170
+ const reason = normalizePaymentSignal(body.reason);
9171
+ const detail = normalizePaymentSignal(body.invalidMessage);
9172
+ const message = normalizePaymentSignal(nested?.message ?? body.message);
9173
+ const label = normalizePaymentSignal(errorLabel);
9174
+ if (code.includes("settlementfailed") || label.includes("settlementfailed") || message.includes("settlementfailed")) return false;
9175
+ const verifyPhase = code === "paymentinvalid" || label.includes("verificationfailed") || message.includes("verificationfailed");
9176
+ if (!verifyPhase) return false;
9177
+ return code === "paymentblockhashstale" || detail.includes("blockhashnotfound") || detail.includes("blockheightexceeded") || reason === "expiredsignature" || message.includes("expiredsignature");
9178
+ }
9179
+ async function waitForStaleRetry(attempt) {
9180
+ await new Promise(
9181
+ (resolve) => setTimeout(resolve, STALE_BLOCKHASH_RETRY_BACKOFFS_MS[attempt])
9182
+ );
9183
+ }
8620
9184
  var DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc";
8621
9185
  function resolveRpcConfig(rpcUrl, rpcHeaders) {
8622
9186
  const env = typeof process !== "undefined" && process.env ? process.env : {};
@@ -9063,26 +9627,34 @@ var SolanaLLMClient = class {
9063
9627
  }
9064
9628
  async requestWithPayment(endpoint, body) {
9065
9629
  const url = `${this.apiUrl}${endpoint}`;
9066
- const response = await this.fetchWithTimeout(url, {
9067
- method: "POST",
9068
- headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9069
- body: JSON.stringify(body)
9070
- });
9071
- if (response.status === 402) {
9072
- return this.handlePaymentAndRetry(url, body, response);
9073
- }
9074
- if (!response.ok) {
9075
- let errorBody;
9076
- try {
9077
- errorBody = await response.json();
9078
- } catch {
9079
- errorBody = { error: "Request failed" };
9630
+ for (let staleRetries = 0; ; ) {
9631
+ const response = await this.fetchWithTimeout(url, {
9632
+ method: "POST",
9633
+ headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9634
+ body: JSON.stringify(body)
9635
+ });
9636
+ if (response.status === 402) {
9637
+ try {
9638
+ return await this.handlePaymentAndRetry(url, body, response, staleRetries > 0);
9639
+ } catch (error) {
9640
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9641
+ await waitForStaleRetry(staleRetries++);
9642
+ continue;
9643
+ }
9080
9644
  }
9081
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9645
+ if (!response.ok) {
9646
+ let errorBody;
9647
+ try {
9648
+ errorBody = await response.json();
9649
+ } catch {
9650
+ errorBody = { error: "Request failed" };
9651
+ }
9652
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9653
+ }
9654
+ return response.json();
9082
9655
  }
9083
- return response.json();
9084
9656
  }
9085
- async handlePaymentAndRetry(url, body, response) {
9657
+ async handlePaymentAndRetry(url, body, response, forceFreshBlockhash = false) {
9086
9658
  let paymentHeader = response.headers.get("payment-required");
9087
9659
  if (!paymentHeader) {
9088
9660
  try {
@@ -9124,7 +9696,8 @@ var SolanaLLMClient = class {
9124
9696
  extra: details.extra,
9125
9697
  extensions,
9126
9698
  rpcUrl: this.rpcUrl,
9127
- rpcHeaders: this.rpcHeaders
9699
+ rpcHeaders: this.rpcHeaders,
9700
+ forceFreshBlockhash
9128
9701
  }
9129
9702
  );
9130
9703
  const retryResponse = await this.fetchWithTimeout(url, {
@@ -9137,6 +9710,9 @@ var SolanaLLMClient = class {
9137
9710
  body: JSON.stringify(body)
9138
9711
  });
9139
9712
  if (retryResponse.status === 402) {
9713
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9714
+ throw new SafeStaleBlockhashError();
9715
+ }
9140
9716
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9141
9717
  }
9142
9718
  if (!retryResponse.ok) {
@@ -9155,26 +9731,34 @@ var SolanaLLMClient = class {
9155
9731
  }
9156
9732
  async requestWithPaymentRaw(endpoint, body) {
9157
9733
  const url = `${this.apiUrl}${endpoint}`;
9158
- const response = await this.fetchWithTimeout(url, {
9159
- method: "POST",
9160
- headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9161
- body: JSON.stringify(body)
9162
- });
9163
- if (response.status === 402) {
9164
- return this.handlePaymentAndRetryRaw(url, body, response);
9165
- }
9166
- if (!response.ok) {
9167
- let errorBody;
9168
- try {
9169
- errorBody = await response.json();
9170
- } catch {
9171
- errorBody = { error: "Request failed" };
9734
+ for (let staleRetries = 0; ; ) {
9735
+ const response = await this.fetchWithTimeout(url, {
9736
+ method: "POST",
9737
+ headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9738
+ body: JSON.stringify(body)
9739
+ });
9740
+ if (response.status === 402) {
9741
+ try {
9742
+ return await this.handlePaymentAndRetryRaw(url, body, response, staleRetries > 0);
9743
+ } catch (error) {
9744
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9745
+ await waitForStaleRetry(staleRetries++);
9746
+ continue;
9747
+ }
9172
9748
  }
9173
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9749
+ if (!response.ok) {
9750
+ let errorBody;
9751
+ try {
9752
+ errorBody = await response.json();
9753
+ } catch {
9754
+ errorBody = { error: "Request failed" };
9755
+ }
9756
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9757
+ }
9758
+ return response.json();
9174
9759
  }
9175
- return response.json();
9176
9760
  }
9177
- async handlePaymentAndRetryRaw(url, body, response) {
9761
+ async handlePaymentAndRetryRaw(url, body, response, forceFreshBlockhash = false) {
9178
9762
  let paymentHeader = response.headers.get("payment-required");
9179
9763
  if (!paymentHeader) {
9180
9764
  try {
@@ -9216,7 +9800,8 @@ var SolanaLLMClient = class {
9216
9800
  extra: details.extra,
9217
9801
  extensions,
9218
9802
  rpcUrl: this.rpcUrl,
9219
- rpcHeaders: this.rpcHeaders
9803
+ rpcHeaders: this.rpcHeaders,
9804
+ forceFreshBlockhash
9220
9805
  }
9221
9806
  );
9222
9807
  const retryResponse = await this.fetchWithTimeout(url, {
@@ -9229,6 +9814,9 @@ var SolanaLLMClient = class {
9229
9814
  body: JSON.stringify(body)
9230
9815
  });
9231
9816
  if (retryResponse.status === 402) {
9817
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9818
+ throw new SafeStaleBlockhashError();
9819
+ }
9232
9820
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9233
9821
  }
9234
9822
  if (!retryResponse.ok) {
@@ -9248,25 +9836,33 @@ var SolanaLLMClient = class {
9248
9836
  async getWithPaymentRaw(endpoint, params) {
9249
9837
  const query = params ? "?" + new URLSearchParams(params).toString() : "";
9250
9838
  const url = `${this.apiUrl}${endpoint}${query}`;
9251
- const response = await this.fetchWithTimeout(url, {
9252
- method: "GET",
9253
- headers: { "User-Agent": USER_AGENT }
9254
- });
9255
- if (response.status === 402) {
9256
- return this.handleGetPaymentAndRetryRaw(url, endpoint, params, response);
9257
- }
9258
- if (!response.ok) {
9259
- let errorBody;
9260
- try {
9261
- errorBody = await response.json();
9262
- } catch {
9263
- errorBody = { error: "Request failed" };
9839
+ for (let staleRetries = 0; ; ) {
9840
+ const response = await this.fetchWithTimeout(url, {
9841
+ method: "GET",
9842
+ headers: { "User-Agent": USER_AGENT }
9843
+ });
9844
+ if (response.status === 402) {
9845
+ try {
9846
+ return await this.handleGetPaymentAndRetryRaw(url, endpoint, params, response, staleRetries > 0);
9847
+ } catch (error) {
9848
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9849
+ await waitForStaleRetry(staleRetries++);
9850
+ continue;
9851
+ }
9264
9852
  }
9265
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9853
+ if (!response.ok) {
9854
+ let errorBody;
9855
+ try {
9856
+ errorBody = await response.json();
9857
+ } catch {
9858
+ errorBody = { error: "Request failed" };
9859
+ }
9860
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9861
+ }
9862
+ return response.json();
9266
9863
  }
9267
- return response.json();
9268
9864
  }
9269
- async handleGetPaymentAndRetryRaw(url, endpoint, params, response) {
9865
+ async handleGetPaymentAndRetryRaw(url, endpoint, params, response, forceFreshBlockhash = false) {
9270
9866
  let paymentHeader = response.headers.get("payment-required");
9271
9867
  if (!paymentHeader) {
9272
9868
  try {
@@ -9308,7 +9904,8 @@ var SolanaLLMClient = class {
9308
9904
  extra: details.extra,
9309
9905
  extensions,
9310
9906
  rpcUrl: this.rpcUrl,
9311
- rpcHeaders: this.rpcHeaders
9907
+ rpcHeaders: this.rpcHeaders,
9908
+ forceFreshBlockhash
9312
9909
  }
9313
9910
  );
9314
9911
  const query = params ? "?" + new URLSearchParams(params).toString() : "";
@@ -9321,6 +9918,9 @@ var SolanaLLMClient = class {
9321
9918
  }
9322
9919
  });
9323
9920
  if (retryResponse.status === 402) {
9921
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9922
+ throw new SafeStaleBlockhashError();
9923
+ }
9324
9924
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9325
9925
  }
9326
9926
  if (!retryResponse.ok) {