@blockrun/llm 3.13.1 → 3.13.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -31,7 +31,7 @@ var APIError = class extends BlockrunError {
31
31
  }
32
32
  };
33
33
 
34
- // node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+d4308049348e1_gni7wlazdyzf7iqax5nhqdpoqi/node_modules/@blockrun/router-core/dist/index.js
34
+ // node_modules/.pnpm/@blockrun+router-core@https+++codeload.github.com+BlockRunAI+router-core+tar.gz+5d911879d7f1e_fbldkf3f3jmlwtwu53ftq2mj24/node_modules/@blockrun/router-core/dist/index.js
35
35
  function scoreTokenCount(estimatedTokens, thresholds) {
36
36
  if (estimatedTokens < thresholds.simple) {
37
37
  return { name: "tokenCount", score: -1, signal: `short (${estimatedTokens} tokens)` };
@@ -365,6 +365,21 @@ function applyPromotions(tierConfigs, promotions, profile, now = /* @__PURE__ */
365
365
  }
366
366
  return result;
367
367
  }
368
+ function applyUnavailableModels(tierConfigs, unavailableModels) {
369
+ if (!unavailableModels || unavailableModels.length === 0) return tierConfigs;
370
+ const dead = new Set(unavailableModels);
371
+ let result = tierConfigs;
372
+ for (const tier of Object.keys(tierConfigs)) {
373
+ const config = tierConfigs[tier];
374
+ const alive = [config.primary, ...config.fallback].filter((model) => !dead.has(model));
375
+ if (alive.length === 0 || alive[0] === config.primary && alive.length === config.fallback.length + 1) {
376
+ continue;
377
+ }
378
+ if (result === tierConfigs) result = { ...tierConfigs };
379
+ result[tier] = { primary: alive[0], fallback: alive.slice(1) };
380
+ }
381
+ return result;
382
+ }
368
383
  var RulesStrategy = class {
369
384
  name = "rules";
370
385
  route(prompt, systemPrompt, maxOutputTokens, options) {
@@ -416,6 +431,7 @@ ${value.slice(-(scanLimit - prefixLength))}`;
416
431
  profile = useAgenticTiers ? "agentic" : "auto";
417
432
  }
418
433
  tierConfigs = applyPromotions(tierConfigs, config.promotions, profile, options.now);
434
+ tierConfigs = applyUnavailableModels(tierConfigs, options.unavailableModels);
419
435
  const agenticScoreValue = ruleResult.agenticScore;
420
436
  if (estimatedTokens > config.overrides.maxTokensForceComplex) {
421
437
  const decision2 = selectModel(
@@ -489,14 +505,15 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
489
505
  supportsVision: true
490
506
  },
491
507
  "anthropic/claude-haiku-4.5": {
508
+ // override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
492
509
  contextWindow: 2e5,
493
- maxOutputTokens: 8192,
510
+ maxOutputTokens: 64e3,
494
511
  supportsTools: true,
495
512
  supportsVision: true
496
513
  },
497
- "anthropic/claude-opus-4.6": {
498
- contextWindow: 1e6,
499
- maxOutputTokens: 128e3,
514
+ "anthropic/claude-opus-4.5": {
515
+ contextWindow: 2e5,
516
+ maxOutputTokens: 64e3,
500
517
  supportsTools: true,
501
518
  supportsVision: true
502
519
  },
@@ -518,12 +535,19 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
518
535
  supportsTools: true,
519
536
  supportsVision: true
520
537
  },
521
- "anthropic/claude-sonnet-4.6": {
538
+ "anthropic/claude-sonnet-4.5": {
522
539
  contextWindow: 2e5,
523
540
  maxOutputTokens: 64e3,
524
541
  supportsTools: true,
525
542
  supportsVision: true
526
543
  },
544
+ "anthropic/claude-sonnet-4.6": {
545
+ // override: The public catalog's `categories` omit "vision" for this Anthropic model even though the gateway accepts image input for it (the prior hand-maintained snapshot had it, and Anthropic's model card lists it). Without this the vision filter would silently drop it — reported against the catalog; remove once the categories carry vision.
546
+ contextWindow: 1e6,
547
+ maxOutputTokens: 128e3,
548
+ supportsTools: true,
549
+ supportsVision: true
550
+ },
527
551
  "anthropic/claude-sonnet-5": {
528
552
  contextWindow: 1e6,
529
553
  maxOutputTokens: 128e3,
@@ -531,14 +555,14 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
531
555
  supportsVision: true
532
556
  },
533
557
  "deepseek/deepseek-chat": {
534
- contextWindow: 1e6,
535
- maxOutputTokens: 8192,
558
+ contextWindow: 1048576,
559
+ maxOutputTokens: 65536,
536
560
  supportsTools: true,
537
561
  supportsVision: false
538
562
  },
539
563
  "deepseek/deepseek-reasoner": {
540
- contextWindow: 1e6,
541
- maxOutputTokens: 8192,
564
+ contextWindow: 1048576,
565
+ maxOutputTokens: 65536,
542
566
  supportsTools: true,
543
567
  supportsVision: false
544
568
  },
@@ -548,62 +572,38 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
548
572
  supportsTools: true,
549
573
  supportsVision: false
550
574
  },
551
- "free/deepseek-v4-flash": {
552
- contextWindow: 1e6,
553
- maxOutputTokens: 16384,
554
- supportsTools: false,
555
- supportsVision: false
556
- },
557
- "free/gpt-oss-120b": {
558
- contextWindow: 128e3,
559
- maxOutputTokens: 16384,
560
- supportsTools: false,
561
- supportsVision: false
562
- },
563
- "free/gpt-oss-20b": {
564
- contextWindow: 128e3,
565
- maxOutputTokens: 16384,
566
- supportsTools: false,
567
- supportsVision: false
568
- },
569
- "free/seed-oss-36b": {
570
- contextWindow: 131072,
571
- maxOutputTokens: 16384,
572
- supportsTools: false,
573
- supportsVision: false
574
- },
575
575
  "google/gemini-2.5-flash": {
576
- contextWindow: 1e6,
576
+ contextWindow: 1048576,
577
577
  maxOutputTokens: 65536,
578
578
  supportsTools: true,
579
579
  supportsVision: true
580
580
  },
581
581
  "google/gemini-2.5-flash-lite": {
582
- contextWindow: 1e6,
582
+ contextWindow: 1048576,
583
583
  maxOutputTokens: 65536,
584
584
  supportsTools: true,
585
585
  supportsVision: false
586
586
  },
587
587
  "google/gemini-2.5-pro": {
588
- contextWindow: 105e4,
588
+ contextWindow: 1048576,
589
589
  maxOutputTokens: 65536,
590
590
  supportsTools: true,
591
591
  supportsVision: true
592
592
  },
593
593
  "google/gemini-3-flash-preview": {
594
- contextWindow: 1e6,
594
+ contextWindow: 1048576,
595
595
  maxOutputTokens: 65536,
596
- supportsTools: false,
596
+ supportsTools: true,
597
597
  supportsVision: true
598
598
  },
599
599
  "google/gemini-3.1-flash-lite": {
600
- contextWindow: 1e6,
601
- maxOutputTokens: 8192,
600
+ contextWindow: 1048576,
601
+ maxOutputTokens: 65536,
602
602
  supportsTools: true,
603
603
  supportsVision: false
604
604
  },
605
605
  "google/gemini-3.1-pro": {
606
- contextWindow: 105e4,
606
+ contextWindow: 1048576,
607
607
  maxOutputTokens: 65536,
608
608
  supportsTools: true,
609
609
  supportsVision: true
@@ -614,23 +614,29 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
614
614
  supportsTools: true,
615
615
  supportsVision: true
616
616
  },
617
- "moonshot/kimi-k2.5": {
618
- contextWindow: 262144,
619
- maxOutputTokens: 16384,
617
+ "google/gemini-3.5-flash-lite": {
618
+ contextWindow: 1048576,
619
+ maxOutputTokens: 65536,
620
620
  supportsTools: true,
621
- supportsVision: true
621
+ supportsVision: false
622
622
  },
623
- "moonshot/kimi-k2.6": {
624
- contextWindow: 262144,
623
+ "google/gemini-3.6-flash": {
624
+ contextWindow: 1048576,
625
625
  maxOutputTokens: 65536,
626
626
  supportsTools: true,
627
627
  supportsVision: true
628
628
  },
629
- "moonshot/kimi-k2.7": {
630
- contextWindow: 262144,
629
+ "minimax/minimax-m2.7": {
630
+ contextWindow: 204800,
631
+ maxOutputTokens: 16384,
632
+ supportsTools: true,
633
+ supportsVision: false
634
+ },
635
+ "minimax/minimax-m3": {
636
+ contextWindow: 1048576,
631
637
  maxOutputTokens: 65536,
632
638
  supportsTools: true,
633
- supportsVision: true
639
+ supportsVision: false
634
640
  },
635
641
  "moonshot/kimi-k3": {
636
642
  contextWindow: 1048576,
@@ -638,7 +644,63 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
638
644
  supportsTools: true,
639
645
  supportsVision: true
640
646
  },
647
+ "nvidia/mistral-nemotron": {
648
+ // supportsTools: gateway unavailable at probe time — fails closed
649
+ contextWindow: 131072,
650
+ maxOutputTokens: 16384,
651
+ supportsTools: false,
652
+ supportsVision: false
653
+ },
654
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
655
+ contextWindow: 256e3,
656
+ maxOutputTokens: 16384,
657
+ supportsTools: false,
658
+ supportsVision: true
659
+ },
660
+ "nvidia/nemotron-nano-12b-v2-vl": {
661
+ // supportsTools: gateway unavailable at probe time — fails closed
662
+ contextWindow: 131072,
663
+ maxOutputTokens: 16384,
664
+ supportsTools: false,
665
+ supportsVision: true
666
+ },
667
+ "nvidia/nemotron-nano-9b-v2": {
668
+ contextWindow: 131072,
669
+ maxOutputTokens: 16384,
670
+ supportsTools: false,
671
+ supportsVision: false
672
+ },
673
+ "nvidia/step-3.7-flash": {
674
+ contextWindow: 131072,
675
+ maxOutputTokens: 16384,
676
+ supportsTools: false,
677
+ supportsVision: false
678
+ },
679
+ "openai/chat-latest": {
680
+ contextWindow: 128e3,
681
+ maxOutputTokens: 128e3,
682
+ supportsTools: true,
683
+ supportsVision: true
684
+ },
641
685
  "openai/gpt-4.1": {
686
+ contextWindow: 128e3,
687
+ maxOutputTokens: 32768,
688
+ supportsTools: true,
689
+ supportsVision: true
690
+ },
691
+ "openai/gpt-4.1-mini": {
692
+ contextWindow: 128e3,
693
+ maxOutputTokens: 32768,
694
+ supportsTools: true,
695
+ supportsVision: false
696
+ },
697
+ "openai/gpt-4.1-nano": {
698
+ contextWindow: 128e3,
699
+ maxOutputTokens: 32768,
700
+ supportsTools: true,
701
+ supportsVision: false
702
+ },
703
+ "openai/gpt-4o": {
642
704
  contextWindow: 128e3,
643
705
  maxOutputTokens: 16384,
644
706
  supportsTools: true,
@@ -652,17 +714,44 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
652
714
  },
653
715
  "openai/gpt-5-mini": {
654
716
  contextWindow: 2e5,
655
- maxOutputTokens: 65536,
717
+ maxOutputTokens: 128e3,
656
718
  supportsTools: true,
657
719
  supportsVision: false
658
720
  },
721
+ "openai/gpt-5.2": {
722
+ contextWindow: 4e5,
723
+ maxOutputTokens: 128e3,
724
+ supportsTools: true,
725
+ supportsVision: true
726
+ },
727
+ "openai/gpt-5.2-pro": {
728
+ // supportsTools: not probed — fails closed
729
+ contextWindow: 4e5,
730
+ maxOutputTokens: 128e3,
731
+ supportsTools: false,
732
+ supportsVision: true
733
+ },
734
+ "openai/gpt-5.3": {
735
+ // supportsTools: gateway unavailable at probe time — fails closed
736
+ contextWindow: 128e3,
737
+ maxOutputTokens: 128e3,
738
+ supportsTools: false,
739
+ supportsVision: true
740
+ },
659
741
  "openai/gpt-5.3-codex": {
742
+ // supportsTools: gateway unavailable at probe time — fails closed; override: 2026-08-29 probe: every request (6 plain + 3 tool attempts) returned a gateway 500, so the probe measured an incident, not the model. Codex's function calling is established by the 2026-07 Terminal-Bench / tau2 calibration trajectories in portfolio.ts. Hosts observing the 500s should drop it with RouterOptions.unavailableModels rather than this snapshot claiming the model cannot call tools.
660
743
  contextWindow: 4e5,
661
744
  maxOutputTokens: 128e3,
662
745
  supportsTools: true,
663
746
  supportsVision: false
664
747
  },
665
748
  "openai/gpt-5.4": {
749
+ contextWindow: 105e4,
750
+ maxOutputTokens: 128e3,
751
+ supportsTools: true,
752
+ supportsVision: true
753
+ },
754
+ "openai/gpt-5.4-mini": {
666
755
  contextWindow: 4e5,
667
756
  maxOutputTokens: 128e3,
668
757
  supportsTools: true,
@@ -670,30 +759,92 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
670
759
  },
671
760
  "openai/gpt-5.4-nano": {
672
761
  contextWindow: 105e4,
673
- maxOutputTokens: 32768,
762
+ maxOutputTokens: 128e3,
674
763
  supportsTools: true,
675
764
  supportsVision: false
676
765
  },
766
+ "openai/gpt-5.4-pro": {
767
+ // supportsTools: not probed — fails closed
768
+ contextWindow: 105e4,
769
+ maxOutputTokens: 128e3,
770
+ supportsTools: false,
771
+ supportsVision: true
772
+ },
677
773
  "openai/gpt-5.5": {
678
774
  contextWindow: 105e4,
679
775
  maxOutputTokens: 128e3,
680
776
  supportsTools: true,
681
777
  supportsVision: true
682
778
  },
779
+ "openai/gpt-5.5-pro": {
780
+ // supportsTools: not probed — fails closed
781
+ contextWindow: 105e4,
782
+ maxOutputTokens: 128e3,
783
+ supportsTools: false,
784
+ supportsVision: true
785
+ },
786
+ "openai/gpt-5.6-luna": {
787
+ contextWindow: 105e4,
788
+ maxOutputTokens: 128e3,
789
+ supportsTools: true,
790
+ supportsVision: true
791
+ },
792
+ "openai/gpt-5.6-luna-pro": {
793
+ contextWindow: 105e4,
794
+ maxOutputTokens: 128e3,
795
+ supportsTools: false,
796
+ supportsVision: true
797
+ },
798
+ "openai/gpt-5.6-sol": {
799
+ contextWindow: 105e4,
800
+ maxOutputTokens: 128e3,
801
+ supportsTools: true,
802
+ supportsVision: true
803
+ },
804
+ "openai/gpt-5.6-sol-pro": {
805
+ contextWindow: 105e4,
806
+ maxOutputTokens: 128e3,
807
+ supportsTools: true,
808
+ supportsVision: true
809
+ },
683
810
  "openai/gpt-5.6-terra": {
684
811
  contextWindow: 105e4,
685
812
  maxOutputTokens: 128e3,
686
813
  supportsTools: true,
687
814
  supportsVision: true
688
815
  },
816
+ "openai/gpt-5.6-terra-pro": {
817
+ contextWindow: 105e4,
818
+ maxOutputTokens: 128e3,
819
+ supportsTools: true,
820
+ supportsVision: true
821
+ },
822
+ "openai/o1": {
823
+ contextWindow: 2e5,
824
+ maxOutputTokens: 1e5,
825
+ supportsTools: true,
826
+ supportsVision: false
827
+ },
689
828
  "openai/o3": {
690
829
  contextWindow: 2e5,
691
830
  maxOutputTokens: 1e5,
692
831
  supportsTools: true,
693
832
  supportsVision: false
694
833
  },
834
+ "openai/o3-mini": {
835
+ contextWindow: 128e3,
836
+ maxOutputTokens: 1e5,
837
+ supportsTools: true,
838
+ supportsVision: false
839
+ },
695
840
  "openai/o4-mini": {
696
841
  contextWindow: 128e3,
842
+ maxOutputTokens: 1e5,
843
+ supportsTools: true,
844
+ supportsVision: false
845
+ },
846
+ "qwen/qwen3.7-flash": {
847
+ contextWindow: 1e6,
697
848
  maxOutputTokens: 65536,
698
849
  supportsTools: true,
699
850
  supportsVision: false
@@ -704,299 +855,605 @@ var DEFAULT_MODEL_CAPABILITIES = Object.freeze({
704
855
  supportsTools: true,
705
856
  supportsVision: false
706
857
  },
707
- "xai/grok-3-mini": {
708
- contextWindow: 131072,
709
- maxOutputTokens: 16384,
858
+ "qwen/qwen3.7-plus": {
859
+ contextWindow: 1e6,
860
+ maxOutputTokens: 131072,
710
861
  supportsTools: true,
711
862
  supportsVision: false
712
863
  },
713
- "xai/grok-4-0709": {
714
- contextWindow: 131072,
715
- maxOutputTokens: 16384,
864
+ "tencent/hy3": {
865
+ contextWindow: 262144,
866
+ maxOutputTokens: 128e3,
716
867
  supportsTools: true,
717
868
  supportsVision: false
718
869
  },
719
- "xai/grok-4-1-fast-non-reasoning": {
720
- contextWindow: 131072,
870
+ "xai/grok-4.3": {
871
+ contextWindow: 1e6,
721
872
  maxOutputTokens: 16384,
722
873
  supportsTools: true,
723
- supportsVision: false
874
+ supportsVision: true
724
875
  },
725
- "xai/grok-4-1-fast-reasoning": {
726
- contextWindow: 131072,
876
+ "xai/grok-4.5": {
877
+ contextWindow: 5e5,
727
878
  maxOutputTokens: 16384,
728
879
  supportsTools: true,
729
- supportsVision: false
880
+ supportsVision: true
730
881
  },
731
- "xai/grok-4-fast-non-reasoning": {
732
- contextWindow: 131072,
882
+ "xai/grok-build-0.1": {
883
+ contextWindow: 256e3,
733
884
  maxOutputTokens: 16384,
734
885
  supportsTools: true,
735
886
  supportsVision: false
736
887
  },
737
- "xai/grok-4-fast-reasoning": {
738
- contextWindow: 131072,
739
- maxOutputTokens: 16384,
740
- supportsTools: true,
741
- supportsVision: false
888
+ "xiaomi/mimo-v2.5-pro": {
889
+ contextWindow: 1048576,
890
+ maxOutputTokens: 131072,
891
+ supportsTools: true,
892
+ supportsVision: false
893
+ },
894
+ "zai/glm-5": {
895
+ contextWindow: 2e5,
896
+ maxOutputTokens: 128e3,
897
+ supportsTools: true,
898
+ supportsVision: false
899
+ },
900
+ "zai/glm-5-turbo": {
901
+ contextWindow: 2e5,
902
+ maxOutputTokens: 128e3,
903
+ supportsTools: true,
904
+ supportsVision: false
905
+ },
906
+ "zai/glm-5.1": {
907
+ contextWindow: 2e5,
908
+ maxOutputTokens: 128e3,
909
+ supportsTools: true,
910
+ supportsVision: false
911
+ },
912
+ "zai/glm-5.2": {
913
+ contextWindow: 1e6,
914
+ maxOutputTokens: 131072,
915
+ supportsTools: true,
916
+ supportsVision: false
917
+ },
918
+ "zai/glm-5.3": {
919
+ contextWindow: 1e6,
920
+ maxOutputTokens: 131072,
921
+ supportsTools: true,
922
+ supportsVision: false
923
+ },
924
+ "zai/glm-5.3-flash": {
925
+ contextWindow: 1e6,
926
+ maxOutputTokens: 131072,
927
+ supportsTools: true,
928
+ supportsVision: true
929
+ }
930
+ });
931
+ var model_profiles_generated_default = {
932
+ "anthropic/claude-fable-5": {
933
+ measuredAt: "2026-08-29T16:51:33Z",
934
+ latencyMs: 9298.5,
935
+ p95LatencyMs: 9873.4,
936
+ outputTokensPerSecond: 55.17,
937
+ errorRate: 0,
938
+ samples: 3
939
+ },
940
+ "anthropic/claude-haiku-4.5": {
941
+ measuredAt: "2026-08-29T16:51:33Z",
942
+ latencyMs: 3157.4,
943
+ p95LatencyMs: 3170.7,
944
+ outputTokensPerSecond: 162.16,
945
+ errorRate: 0,
946
+ samples: 3
947
+ },
948
+ "anthropic/claude-opus-4.5": {
949
+ measuredAt: "2026-08-29T16:51:33Z",
950
+ latencyMs: 6497.7,
951
+ p95LatencyMs: 6953.7,
952
+ outputTokensPerSecond: 78.99,
953
+ errorRate: 0,
954
+ samples: 3
955
+ },
956
+ "anthropic/claude-opus-4.7": {
957
+ measuredAt: "2026-08-29T16:51:33Z",
958
+ latencyMs: 5316.5,
959
+ p95LatencyMs: 6121.5,
960
+ outputTokensPerSecond: 97.34,
961
+ errorRate: 0,
962
+ samples: 3
963
+ },
964
+ "anthropic/claude-opus-4.8": {
965
+ measuredAt: "2026-08-29T16:51:33Z",
966
+ latencyMs: 6216.1,
967
+ p95LatencyMs: 6847.7,
968
+ outputTokensPerSecond: 82.81,
969
+ errorRate: 0,
970
+ samples: 3
971
+ },
972
+ "anthropic/claude-opus-5": {
973
+ measuredAt: "2026-08-29T16:51:33Z",
974
+ latencyMs: 7309,
975
+ p95LatencyMs: 7745.2,
976
+ outputTokensPerSecond: 70.17,
977
+ errorRate: 0,
978
+ samples: 3
979
+ },
980
+ "anthropic/claude-sonnet-4.5": {
981
+ measuredAt: "2026-08-29T16:51:33Z",
982
+ latencyMs: 6330.4,
983
+ p95LatencyMs: 6631.6,
984
+ outputTokensPerSecond: 81.03,
985
+ errorRate: 0,
986
+ samples: 3
987
+ },
988
+ "anthropic/claude-sonnet-4.6": {
989
+ measuredAt: "2026-08-29T16:51:33Z",
990
+ latencyMs: 6508,
991
+ p95LatencyMs: 6698.3,
992
+ outputTokensPerSecond: 78.6,
993
+ errorRate: 0,
994
+ samples: 3
995
+ },
996
+ "anthropic/claude-sonnet-5": {
997
+ measuredAt: "2026-08-29T16:51:33Z",
998
+ latencyMs: 6165.4,
999
+ p95LatencyMs: 6582.9,
1000
+ outputTokensPerSecond: 83.62,
1001
+ errorRate: 0,
1002
+ samples: 3
1003
+ },
1004
+ "deepseek/deepseek-chat": {
1005
+ measuredAt: "2026-08-29T16:51:33Z",
1006
+ latencyMs: 4351.4,
1007
+ p95LatencyMs: 4543.7,
1008
+ outputTokensPerSecond: 117.78,
1009
+ errorRate: 0,
1010
+ samples: 3
1011
+ },
1012
+ "deepseek/deepseek-reasoner": {
1013
+ measuredAt: "2026-08-29T16:51:33Z",
1014
+ latencyMs: 5201.2,
1015
+ p95LatencyMs: 6079.6,
1016
+ outputTokensPerSecond: 99.77,
1017
+ errorRate: 0,
1018
+ samples: 3
1019
+ },
1020
+ "deepseek/deepseek-v4-pro": {
1021
+ measuredAt: "2026-08-29T16:51:33Z",
1022
+ latencyMs: 8781.2,
1023
+ p95LatencyMs: 9881.1,
1024
+ outputTokensPerSecond: 58.98,
1025
+ errorRate: 0,
1026
+ samples: 3
1027
+ },
1028
+ "google/gemini-2.5-flash": {
1029
+ measuredAt: "2026-08-29T16:51:33Z",
1030
+ latencyMs: 5416.4,
1031
+ p95LatencyMs: 6442.8,
1032
+ outputTokensPerSecond: 213.07,
1033
+ errorRate: 0,
1034
+ samples: 3
1035
+ },
1036
+ "google/gemini-2.5-flash-lite": {
1037
+ measuredAt: "2026-08-29T16:51:33Z",
1038
+ latencyMs: 5002.6,
1039
+ p95LatencyMs: 5780.3,
1040
+ outputTokensPerSecond: 408.43,
1041
+ errorRate: 0,
1042
+ samples: 3
1043
+ },
1044
+ "google/gemini-2.5-pro": {
1045
+ measuredAt: "2026-08-29T16:51:33Z",
1046
+ latencyMs: 28169.5,
1047
+ p95LatencyMs: 29491.4,
1048
+ outputTokensPerSecond: 147.3,
1049
+ errorRate: 0,
1050
+ samples: 3
1051
+ },
1052
+ "google/gemini-3-flash-preview": {
1053
+ measuredAt: "2026-08-29T16:51:33Z",
1054
+ latencyMs: 4717.1,
1055
+ p95LatencyMs: 5037.1,
1056
+ outputTokensPerSecond: 198.71,
1057
+ errorRate: 0,
1058
+ samples: 3
1059
+ },
1060
+ "google/gemini-3.1-flash-lite": {
1061
+ measuredAt: "2026-08-29T16:51:33Z",
1062
+ latencyMs: 2855.8,
1063
+ p95LatencyMs: 3172.7,
1064
+ outputTokensPerSecond: 286.91,
1065
+ errorRate: 0,
1066
+ samples: 3
1067
+ },
1068
+ "google/gemini-3.1-pro": {
1069
+ measuredAt: "2026-08-29T16:59:54Z",
1070
+ latencyMs: 24194.1,
1071
+ p95LatencyMs: 27269.6,
1072
+ outputTokensPerSecond: 109.47,
1073
+ errorRate: 0,
1074
+ samples: 3
1075
+ },
1076
+ "google/gemini-3.5-flash": {
1077
+ measuredAt: "2026-08-29T16:51:33Z",
1078
+ latencyMs: 5320.6,
1079
+ p95LatencyMs: 5429.8,
1080
+ outputTokensPerSecond: 226.21,
1081
+ errorRate: 0,
1082
+ samples: 3
1083
+ },
1084
+ "google/gemini-3.5-flash-lite": {
1085
+ measuredAt: "2026-08-29T16:51:33Z",
1086
+ latencyMs: 3515.8,
1087
+ p95LatencyMs: 4363.4,
1088
+ outputTokensPerSecond: 248.9,
1089
+ errorRate: 0,
1090
+ samples: 3
1091
+ },
1092
+ "google/gemini-3.6-flash": {
1093
+ measuredAt: "2026-08-29T16:51:33Z",
1094
+ latencyMs: 13020,
1095
+ p95LatencyMs: 15383.1,
1096
+ outputTokensPerSecond: 187.87,
1097
+ errorRate: 0,
1098
+ samples: 3
1099
+ },
1100
+ "minimax/minimax-m2.7": {
1101
+ measuredAt: "2026-08-29T16:51:33Z",
1102
+ latencyMs: 8761.1,
1103
+ p95LatencyMs: 10199.3,
1104
+ outputTokensPerSecond: 59.18,
1105
+ errorRate: 0,
1106
+ samples: 3
1107
+ },
1108
+ "minimax/minimax-m3": {
1109
+ measuredAt: "2026-08-29T16:51:33Z",
1110
+ latencyMs: 11101.9,
1111
+ p95LatencyMs: 26087.1,
1112
+ outputTokensPerSecond: 101.12,
1113
+ errorRate: 0,
1114
+ samples: 3
1115
+ },
1116
+ "moonshot/kimi-k3": {
1117
+ measuredAt: "2026-08-29T16:51:33Z",
1118
+ latencyMs: 24498.9,
1119
+ p95LatencyMs: 40365.3,
1120
+ outputTokensPerSecond: 25.11,
1121
+ errorRate: 0,
1122
+ samples: 3
1123
+ },
1124
+ "nvidia/mistral-nemotron": {
1125
+ measuredAt: "2026-08-29T16:51:33Z",
1126
+ latencyMs: 7349.6,
1127
+ p95LatencyMs: 9932.3,
1128
+ outputTokensPerSecond: 79.48,
1129
+ errorRate: 0.3333,
1130
+ samples: 3
1131
+ },
1132
+ "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
1133
+ measuredAt: "2026-08-29T16:59:54Z",
1134
+ latencyMs: 9324.6,
1135
+ p95LatencyMs: 12992,
1136
+ outputTokensPerSecond: 64.96,
1137
+ errorRate: 0.3333,
1138
+ samples: 3
1139
+ },
1140
+ "nvidia/nemotron-nano-12b-v2-vl": {
1141
+ measuredAt: "2026-08-29T16:51:33Z",
1142
+ latencyMs: 5846.9,
1143
+ p95LatencyMs: 5846.9,
1144
+ outputTokensPerSecond: 87.57,
1145
+ errorRate: 0.6667,
1146
+ samples: 3
1147
+ },
1148
+ "nvidia/nemotron-nano-9b-v2": {
1149
+ measuredAt: "2026-08-29T16:51:33Z",
1150
+ latencyMs: 5282.5,
1151
+ p95LatencyMs: 5282.5,
1152
+ outputTokensPerSecond: 96.92,
1153
+ errorRate: 0.6667,
1154
+ samples: 3
1155
+ },
1156
+ "nvidia/step-3.7-flash": {
1157
+ measuredAt: "2026-08-29T16:51:33Z",
1158
+ latencyMs: 4617.4,
1159
+ p95LatencyMs: 5237.4,
1160
+ outputTokensPerSecond: 112.92,
1161
+ errorRate: 0.3333,
1162
+ samples: 3
1163
+ },
1164
+ "openai/chat-latest": {
1165
+ measuredAt: "2026-08-29T16:51:33Z",
1166
+ latencyMs: 3690.9,
1167
+ p95LatencyMs: 4344,
1168
+ outputTokensPerSecond: 111.85,
1169
+ errorRate: 0,
1170
+ samples: 3
1171
+ },
1172
+ "openai/gpt-4.1": {
1173
+ measuredAt: "2026-08-29T16:51:33Z",
1174
+ latencyMs: 3527.9,
1175
+ p95LatencyMs: 3831.7,
1176
+ outputTokensPerSecond: 141.27,
1177
+ errorRate: 0,
1178
+ samples: 3
1179
+ },
1180
+ "openai/gpt-4.1-mini": {
1181
+ measuredAt: "2026-08-29T16:51:33Z",
1182
+ latencyMs: 4268.2,
1183
+ p95LatencyMs: 5101.5,
1184
+ outputTokensPerSecond: 103.42,
1185
+ errorRate: 0,
1186
+ samples: 3
742
1187
  },
743
- "xai/grok-4.5": {
744
- contextWindow: 5e5,
745
- maxOutputTokens: 16384,
746
- supportsTools: true,
747
- supportsVision: true
1188
+ "openai/gpt-4.1-nano": {
1189
+ measuredAt: "2026-08-29T16:51:33Z",
1190
+ latencyMs: 3088.3,
1191
+ p95LatencyMs: 3369.2,
1192
+ outputTokensPerSecond: 150.31,
1193
+ errorRate: 0,
1194
+ samples: 3
748
1195
  },
749
- "zai/glm-5.1": {
750
- contextWindow: 2e5,
751
- maxOutputTokens: 128e3,
752
- supportsTools: true,
753
- supportsVision: false
1196
+ "openai/gpt-4o": {
1197
+ measuredAt: "2026-08-29T16:51:33Z",
1198
+ latencyMs: 2995.2,
1199
+ p95LatencyMs: 3174.2,
1200
+ outputTokensPerSecond: 171.32,
1201
+ errorRate: 0,
1202
+ samples: 3
754
1203
  },
755
- "zai/glm-5.2": {
756
- contextWindow: 1e6,
757
- maxOutputTokens: 262144,
758
- supportsTools: true,
759
- supportsVision: false
760
- }
761
- });
762
- var model_profiles_generated_default = {
763
- "openai/gpt-5.5": {
764
- measuredAt: "2026-07-21T10:21:31Z",
765
- latencyMs: 6243.1,
766
- p95LatencyMs: 9865,
767
- outputTokensPerSecond: 12.53,
1204
+ "openai/gpt-4o-mini": {
1205
+ measuredAt: "2026-08-29T16:51:33Z",
1206
+ latencyMs: 4751.5,
1207
+ p95LatencyMs: 4930.4,
1208
+ outputTokensPerSecond: 107.84,
768
1209
  errorRate: 0,
769
1210
  samples: 3
770
1211
  },
771
- "openai/gpt-5.4-pro": {
772
- measuredAt: "2026-07-21T10:21:31Z",
773
- latencyMs: 13015.5,
774
- p95LatencyMs: 23976.4,
775
- outputTokensPerSecond: 6.42,
1212
+ "openai/gpt-5-mini": {
1213
+ measuredAt: "2026-08-29T16:51:33Z",
1214
+ latencyMs: 4558.1,
1215
+ p95LatencyMs: 5081.9,
1216
+ outputTokensPerSecond: 113.25,
776
1217
  errorRate: 0,
777
1218
  samples: 3
778
1219
  },
779
- "openai/gpt-5.4-mini": {
780
- measuredAt: "2026-07-21T10:21:31Z",
781
- latencyMs: 5550,
782
- p95LatencyMs: 6595.7,
783
- outputTokensPerSecond: 11.96,
784
- errorRate: 0.3333,
1220
+ "openai/gpt-5.2": {
1221
+ measuredAt: "2026-08-29T16:51:33Z",
1222
+ latencyMs: 5436.6,
1223
+ p95LatencyMs: 5928.8,
1224
+ outputTokensPerSecond: 95.47,
1225
+ errorRate: 0,
785
1226
  samples: 3
786
1227
  },
787
1228
  "openai/gpt-5.3-codex": {
788
- measuredAt: "2026-07-21T10:21:31Z",
789
- latencyMs: 4617.1,
790
- p95LatencyMs: 5800.7,
791
- outputTokensPerSecond: 12.48,
792
- errorRate: 0,
1229
+ measuredAt: "2026-08-29T16:59:54Z",
1230
+ latencyMs: 15290.4,
1231
+ p95LatencyMs: 15290.4,
1232
+ outputTokensPerSecond: 33.49,
1233
+ errorRate: 0.6667,
793
1234
  samples: 3
794
1235
  },
795
- "anthropic/claude-opus-4.8": {
796
- measuredAt: "2026-07-21T10:21:31Z",
797
- latencyMs: 3915.1,
798
- p95LatencyMs: 6130.8,
799
- outputTokensPerSecond: 16.33,
1236
+ "openai/gpt-5.4": {
1237
+ measuredAt: "2026-08-29T16:51:33Z",
1238
+ latencyMs: 5596,
1239
+ p95LatencyMs: 5919.4,
1240
+ outputTokensPerSecond: 91.67,
800
1241
  errorRate: 0,
801
1242
  samples: 3
802
1243
  },
803
- "anthropic/claude-opus-4.6": {
804
- measuredAt: "2026-07-21T10:21:31Z",
805
- latencyMs: 3765.5,
806
- p95LatencyMs: 4257.2,
807
- outputTokensPerSecond: 14.18,
1244
+ "openai/gpt-5.4-mini": {
1245
+ measuredAt: "2026-08-29T16:51:33Z",
1246
+ latencyMs: 3377.8,
1247
+ p95LatencyMs: 3646.8,
1248
+ outputTokensPerSecond: 138.08,
808
1249
  errorRate: 0,
809
1250
  samples: 3
810
1251
  },
811
- "anthropic/claude-sonnet-4.6": {
812
- measuredAt: "2026-07-21T10:21:31Z",
813
- latencyMs: 3860.6,
814
- p95LatencyMs: 5093.5,
815
- outputTokensPerSecond: 13.85,
1252
+ "openai/gpt-5.4-nano": {
1253
+ measuredAt: "2026-08-29T16:51:33Z",
1254
+ latencyMs: 4040.4,
1255
+ p95LatencyMs: 4205.9,
1256
+ outputTokensPerSecond: 118.52,
816
1257
  errorRate: 0,
817
1258
  samples: 3
818
1259
  },
819
- "anthropic/claude-haiku-4.5": {
820
- measuredAt: "2026-07-21T10:21:31Z",
821
- latencyMs: 2734.9,
822
- p95LatencyMs: 3181.6,
823
- outputTokensPerSecond: 19.58,
1260
+ "openai/gpt-5.5": {
1261
+ measuredAt: "2026-08-29T16:51:33Z",
1262
+ latencyMs: 6367.8,
1263
+ p95LatencyMs: 7330.7,
1264
+ outputTokensPerSecond: 81.29,
824
1265
  errorRate: 0,
825
1266
  samples: 3
826
1267
  },
827
- "google/gemini-3.1-pro": {
828
- measuredAt: "2026-07-21T10:21:31Z",
829
- latencyMs: 13935.7,
830
- p95LatencyMs: 26675.3,
831
- outputTokensPerSecond: 77.47,
1268
+ "openai/gpt-5.6-luna": {
1269
+ measuredAt: "2026-08-29T16:51:33Z",
1270
+ latencyMs: 6064.5,
1271
+ p95LatencyMs: 7347.3,
1272
+ outputTokensPerSecond: 87.93,
832
1273
  errorRate: 0,
833
1274
  samples: 3
834
1275
  },
835
- "google/gemini-3.5-flash": {
836
- measuredAt: "2026-07-21T10:21:31Z",
837
- latencyMs: 4608.7,
838
- p95LatencyMs: 8420.9,
839
- outputTokensPerSecond: 57.88,
1276
+ "openai/gpt-5.6-luna-pro": {
1277
+ measuredAt: "2026-08-29T16:59:54Z",
1278
+ latencyMs: 13914.9,
1279
+ p95LatencyMs: 13914.9,
1280
+ outputTokensPerSecond: 36.79,
1281
+ errorRate: 0.6667,
1282
+ samples: 3
1283
+ },
1284
+ "openai/gpt-5.6-sol": {
1285
+ measuredAt: "2026-08-29T16:51:33Z",
1286
+ latencyMs: 7720.2,
1287
+ p95LatencyMs: 9108.2,
1288
+ outputTokensPerSecond: 67.47,
840
1289
  errorRate: 0,
841
1290
  samples: 3
842
1291
  },
843
- "google/gemini-3.1-flash-lite": {
844
- measuredAt: "2026-07-21T10:21:31Z",
845
- latencyMs: 4619.7,
846
- p95LatencyMs: 9927.1,
847
- outputTokensPerSecond: 42.01,
1292
+ "openai/gpt-5.6-sol-pro": {
1293
+ measuredAt: "2026-08-29T16:51:33Z",
1294
+ latencyMs: 11442.7,
1295
+ p95LatencyMs: 13363.1,
1296
+ outputTokensPerSecond: 148.75,
848
1297
  errorRate: 0,
849
1298
  samples: 3
850
1299
  },
851
- "google/gemini-2.5-flash": {
852
- measuredAt: "2026-07-21T10:21:31Z",
853
- latencyMs: 5506.9,
854
- p95LatencyMs: 11462.5,
855
- outputTokensPerSecond: 65.19,
1300
+ "openai/gpt-5.6-terra": {
1301
+ measuredAt: "2026-08-29T16:51:33Z",
1302
+ latencyMs: 4941,
1303
+ p95LatencyMs: 5095.3,
1304
+ outputTokensPerSecond: 103.69,
856
1305
  errorRate: 0,
857
1306
  samples: 3
858
1307
  },
859
- "deepseek/deepseek-v4-pro": {
860
- measuredAt: "2026-07-21T10:21:31Z",
861
- latencyMs: 6044.8,
862
- p95LatencyMs: 10782.3,
863
- outputTokensPerSecond: 22.47,
1308
+ "openai/gpt-5.6-terra-pro": {
1309
+ measuredAt: "2026-08-29T16:51:33Z",
1310
+ latencyMs: 3574.1,
1311
+ p95LatencyMs: 4126.3,
1312
+ outputTokensPerSecond: 133.59,
864
1313
  errorRate: 0,
865
1314
  samples: 3
866
1315
  },
867
- "deepseek/deepseek-reasoner": {
868
- measuredAt: "2026-07-21T10:21:31Z",
869
- latencyMs: 4111.9,
870
- p95LatencyMs: 5305.7,
871
- outputTokensPerSecond: 16.46,
1316
+ "openai/o1": {
1317
+ measuredAt: "2026-08-29T16:51:33Z",
1318
+ latencyMs: 4324.9,
1319
+ p95LatencyMs: 5838.1,
1320
+ outputTokensPerSecond: 125.86,
872
1321
  errorRate: 0,
873
1322
  samples: 3
874
1323
  },
875
- "deepseek/deepseek-chat": {
876
- measuredAt: "2026-07-21T10:21:31Z",
877
- latencyMs: 2648.6,
878
- p95LatencyMs: 3524.1,
879
- outputTokensPerSecond: 16.73,
1324
+ "openai/o3": {
1325
+ measuredAt: "2026-08-29T16:51:33Z",
1326
+ latencyMs: 5463.4,
1327
+ p95LatencyMs: 5613.1,
1328
+ outputTokensPerSecond: 93.8,
880
1329
  errorRate: 0,
881
1330
  samples: 3
882
1331
  },
883
- "moonshot/kimi-k2.7": {
884
- measuredAt: "2026-07-21T10:21:31Z",
885
- latencyMs: 4295.4,
886
- p95LatencyMs: 6153.8,
887
- outputTokensPerSecond: 18.54,
1332
+ "openai/o3-mini": {
1333
+ measuredAt: "2026-08-29T16:51:33Z",
1334
+ latencyMs: 2912.7,
1335
+ p95LatencyMs: 3092.1,
1336
+ outputTokensPerSecond: 176.49,
888
1337
  errorRate: 0,
889
1338
  samples: 3
890
1339
  },
891
- "qwen/qwen3.7-max": {
892
- measuredAt: "2026-07-21T10:21:31Z",
893
- latencyMs: 30729.4,
894
- p95LatencyMs: 39622,
895
- outputTokensPerSecond: 36.89,
896
- errorRate: 0.3333,
1340
+ "openai/o4-mini": {
1341
+ measuredAt: "2026-08-29T16:51:33Z",
1342
+ latencyMs: 4958.7,
1343
+ p95LatencyMs: 5313,
1344
+ outputTokensPerSecond: 103.81,
1345
+ errorRate: 0,
897
1346
  samples: 3
898
1347
  },
899
- "xai/grok-4.3": {
900
- measuredAt: "2026-07-21T10:21:31Z",
901
- latencyMs: 6946.1,
902
- p95LatencyMs: 9495.4,
903
- outputTokensPerSecond: 65.3,
1348
+ "qwen/qwen3.7-flash": {
1349
+ measuredAt: "2026-08-29T16:51:33Z",
1350
+ latencyMs: 3385.5,
1351
+ p95LatencyMs: 4042.7,
1352
+ outputTokensPerSecond: 153.94,
904
1353
  errorRate: 0,
905
1354
  samples: 3
906
1355
  },
907
- "xai/grok-4.20-reasoning": {
908
- measuredAt: "2026-07-21T10:21:31Z",
909
- latencyMs: 3472.4,
910
- p95LatencyMs: 5332.4,
911
- outputTokensPerSecond: 13.27,
1356
+ "qwen/qwen3.7-max": {
1357
+ measuredAt: "2026-08-29T16:51:33Z",
1358
+ latencyMs: 9387.1,
1359
+ p95LatencyMs: 10490.2,
1360
+ outputTokensPerSecond: 54.92,
912
1361
  errorRate: 0,
913
1362
  samples: 3
914
1363
  },
915
- "xai/grok-4.20-non-reasoning": {
916
- measuredAt: "2026-07-21T10:21:31Z",
917
- latencyMs: 5174.4,
918
- p95LatencyMs: 6081.7,
919
- outputTokensPerSecond: 10.21,
920
- errorRate: 0.3333,
1364
+ "qwen/qwen3.7-plus": {
1365
+ measuredAt: "2026-08-29T16:51:33Z",
1366
+ latencyMs: 9766.6,
1367
+ p95LatencyMs: 9798.2,
1368
+ outputTokensPerSecond: 52.42,
1369
+ errorRate: 0,
921
1370
  samples: 3
922
1371
  },
923
- "xai/grok-4-1-fast-reasoning": {
924
- measuredAt: "2026-07-21T10:21:31Z",
925
- latencyMs: 13148.2,
926
- p95LatencyMs: 19104.2,
927
- outputTokensPerSecond: 4.28,
1372
+ "tencent/hy3": {
1373
+ measuredAt: "2026-08-29T16:51:33Z",
1374
+ latencyMs: 6062.3,
1375
+ p95LatencyMs: 7070.2,
1376
+ outputTokensPerSecond: 87.3,
928
1377
  errorRate: 0,
929
1378
  samples: 3
930
1379
  },
931
- "minimax/minimax-m3": {
932
- measuredAt: "2026-07-21T10:21:31Z",
933
- latencyMs: 3385,
934
- p95LatencyMs: 4247.2,
935
- outputTokensPerSecond: 15.16,
1380
+ "xai/grok-4.3": {
1381
+ measuredAt: "2026-08-29T16:51:33Z",
1382
+ latencyMs: 9467.7,
1383
+ p95LatencyMs: 10087.9,
1384
+ outputTokensPerSecond: 48.36,
936
1385
  errorRate: 0,
937
1386
  samples: 3
938
1387
  },
939
- "minimax/minimax-m2.7": {
940
- measuredAt: "2026-07-21T10:21:31Z",
941
- latencyMs: 4596.7,
942
- p95LatencyMs: 6884.6,
943
- outputTokensPerSecond: 17.03,
1388
+ "xai/grok-4.5": {
1389
+ measuredAt: "2026-08-29T16:51:33Z",
1390
+ latencyMs: 13564.8,
1391
+ p95LatencyMs: 17351.9,
1392
+ outputTokensPerSecond: 60.71,
944
1393
  errorRate: 0,
945
1394
  samples: 3
946
1395
  },
947
- "zai/glm-5.2": {
948
- measuredAt: "2026-07-21T10:21:31Z",
949
- latencyMs: 4406.3,
950
- p95LatencyMs: 6139.7,
951
- outputTokensPerSecond: 10.41,
1396
+ "xai/grok-build-0.1": {
1397
+ measuredAt: "2026-08-29T16:51:33Z",
1398
+ latencyMs: 16394.8,
1399
+ p95LatencyMs: 18035.4,
1400
+ outputTokensPerSecond: 96.86,
952
1401
  errorRate: 0,
953
1402
  samples: 3
954
1403
  },
955
- "zai/glm-5.1": {
956
- measuredAt: "2026-07-21T10:21:31Z",
957
- latencyMs: 7775.4,
958
- p95LatencyMs: 9182.1,
959
- outputTokensPerSecond: 6.08,
1404
+ "xiaomi/mimo-v2.5-pro": {
1405
+ measuredAt: "2026-08-29T16:51:33Z",
1406
+ latencyMs: 12070.7,
1407
+ p95LatencyMs: 12386.8,
1408
+ outputTokensPerSecond: 42.44,
960
1409
  errorRate: 0,
961
1410
  samples: 3
962
1411
  },
963
1412
  "zai/glm-5": {
964
- measuredAt: "2026-07-21T10:21:31Z",
965
- latencyMs: 4159.4,
966
- p95LatencyMs: 4992.7,
967
- outputTokensPerSecond: 10.28,
1413
+ measuredAt: "2026-08-29T16:51:33Z",
1414
+ latencyMs: 6839.7,
1415
+ p95LatencyMs: 7261.4,
1416
+ outputTokensPerSecond: 75.16,
1417
+ errorRate: 0,
1418
+ samples: 3
1419
+ },
1420
+ "zai/glm-5-turbo": {
1421
+ measuredAt: "2026-08-29T16:51:33Z",
1422
+ latencyMs: 55348.5,
1423
+ p95LatencyMs: 114086.6,
1424
+ outputTokensPerSecond: 14.64,
968
1425
  errorRate: 0,
969
1426
  samples: 3
970
1427
  },
971
- "free/qwen3-coder-480b": {
972
- measuredAt: "2026-07-21T10:21:31Z",
973
- latencyMs: 2063.9,
974
- p95LatencyMs: 3646.3,
975
- outputTokensPerSecond: 39.8,
1428
+ "zai/glm-5.1": {
1429
+ measuredAt: "2026-08-29T16:51:33Z",
1430
+ latencyMs: 15658.4,
1431
+ p95LatencyMs: 17307.1,
1432
+ outputTokensPerSecond: 32.9,
976
1433
  errorRate: 0,
977
1434
  samples: 3
978
1435
  },
979
- "free/mistral-large-3-675b": {
980
- measuredAt: "2026-07-21T10:21:31Z",
981
- latencyMs: 3147.5,
982
- p95LatencyMs: 5555.3,
983
- outputTokensPerSecond: 27.76,
1436
+ "zai/glm-5.2": {
1437
+ measuredAt: "2026-08-29T16:51:33Z",
1438
+ latencyMs: 10308.5,
1439
+ p95LatencyMs: 15127.6,
1440
+ outputTokensPerSecond: 54.87,
984
1441
  errorRate: 0,
985
1442
  samples: 3
986
1443
  },
987
- "free/nemotron-3-nano-omni-30b-a3b-reasoning": {
988
- measuredAt: "2026-07-21T10:21:31Z",
989
- latencyMs: 6508.4,
990
- p95LatencyMs: 14252.7,
991
- outputTokensPerSecond: 68.26,
1444
+ "zai/glm-5.3": {
1445
+ measuredAt: "2026-08-29T16:51:33Z",
1446
+ latencyMs: 7272.4,
1447
+ p95LatencyMs: 7998.1,
1448
+ outputTokensPerSecond: 71.09,
992
1449
  errorRate: 0,
993
1450
  samples: 3
994
1451
  },
995
- "free/glm-4.7": {
996
- measuredAt: "2026-07-21T10:21:31Z",
997
- latencyMs: 2014.8,
998
- p95LatencyMs: 3039.9,
999
- outputTokensPerSecond: 39.92,
1452
+ "zai/glm-5.3-flash": {
1453
+ measuredAt: "2026-08-29T16:51:33Z",
1454
+ latencyMs: 10545.3,
1455
+ p95LatencyMs: 11672.4,
1456
+ outputTokensPerSecond: 49.01,
1000
1457
  errorRate: 0,
1001
1458
  samples: 3
1002
1459
  }
@@ -1010,11 +1467,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1010
1467
  latencyMs: 2305,
1011
1468
  outputTokensPerSecond: 140.6
1012
1469
  },
1013
- "anthropic/claude-opus-4.6": {
1014
- measuredAt: "2026-03-16T13:50:48Z",
1015
- latencyMs: 2139,
1016
- outputTokensPerSecond: 119.7
1017
- },
1018
1470
  "anthropic/claude-sonnet-4.6": {
1019
1471
  measuredAt: "2026-03-16T13:50:48Z",
1020
1472
  latencyMs: 2110,
@@ -1048,11 +1500,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1048
1500
  latencyMs: 1609,
1049
1501
  outputTokensPerSecond: 167.2
1050
1502
  },
1051
- "moonshot/kimi-k2.5": {
1052
- measuredAt: "2026-03-16T13:50:48Z",
1053
- latencyMs: 1646,
1054
- outputTokensPerSecond: 155.7
1055
- },
1056
1503
  "openai/gpt-4o-mini": {
1057
1504
  measuredAt: "2026-03-16T13:50:48Z",
1058
1505
  latencyMs: 2764,
@@ -1062,18 +1509,6 @@ var HISTORICAL_MODEL_PROFILES = Object.freeze({
1062
1509
  measuredAt: "2026-03-16T13:50:48Z",
1063
1510
  latencyMs: 7935,
1064
1511
  outputTokensPerSecond: 32.3
1065
- },
1066
- "xai/grok-4-1-fast-non-reasoning": {
1067
- measuredAt: "2026-03-16T13:50:48Z",
1068
- latencyMs: 1244,
1069
- outputTokensPerSecond: 205.8,
1070
- intelligenceIndex: 41
1071
- },
1072
- "xai/grok-4-1-fast-reasoning": {
1073
- measuredAt: "2026-03-16T13:50:48Z",
1074
- latencyMs: 1454,
1075
- outputTokensPerSecond: 176.2,
1076
- intelligenceIndex: 41
1077
1512
  }
1078
1513
  });
1079
1514
  function inferToolRequirement(prompt, _systemPrompt, toolChoice) {
@@ -1566,7 +2001,7 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1566
2001
  match(["gpt-5.3-codex"], 1),
1567
2002
  match(["claude-sonnet-4.6"], 0.94),
1568
2003
  match(["glm-5.2"], 0.9),
1569
- match(["kimi-k2.7", "deepseek-v4-pro"], 0.86)
2004
+ match(["deepseek-v4-pro"], 0.86)
1570
2005
  );
1571
2006
  case "reasoning":
1572
2007
  return Math.max(
@@ -1590,14 +2025,13 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1590
2025
  base,
1591
2026
  match(["gemini-3.5-flash"], 1),
1592
2027
  match(["grok-4.5"], 0.93),
1593
- match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9),
1594
- match(["kimi-k2.7"], 0.84)
2028
+ match(["claude-sonnet-5", "deepseek-v4-pro", "kimi-k3"], 0.9)
1595
2029
  );
1596
2030
  case "vision":
1597
2031
  return Math.max(
1598
2032
  base,
1599
2033
  match(["gemini-3.1-pro"], 0.96),
1600
- match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k2.7", "grok-4.3"], 0.9)
2034
+ match(["qwen3.7-max", "claude-sonnet-4.6", "kimi-k3", "grok-4.3"], 0.9)
1601
2035
  );
1602
2036
  case "long_context":
1603
2037
  return Math.max(
@@ -1609,17 +2043,18 @@ function affinity(modelId, task, language = "other", agentDomain = "other", deep
1609
2043
  );
1610
2044
  case "extraction": {
1611
2045
  const kimiExtractionAffinity = language === "zh" ? 1 : 0.9;
2046
+ const otherExtractionAffinity = language === "zh" ? 0.88 : 0.9;
1612
2047
  return Math.max(
1613
2048
  base,
1614
- match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], 0.9),
1615
- match(["claude-sonnet-5", "claude-sonnet-4.6"], 0.9),
1616
- match(["kimi-k3", "kimi-k2.7"], kimiExtractionAffinity)
2049
+ match(["gemini-3.5-flash", "gemini-2.5-flash", "gpt-4o-mini"], otherExtractionAffinity),
2050
+ match(["claude-sonnet-5", "claude-sonnet-4.6"], otherExtractionAffinity),
2051
+ match(["kimi-k3"], kimiExtractionAffinity)
1617
2052
  );
1618
2053
  }
1619
2054
  default:
1620
2055
  return Math.max(
1621
2056
  base,
1622
- match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3", "kimi-k2.7"], 0.86)
2057
+ match(["gemini-3.5-flash", "gemini-2.5-flash", "kimi-k3"], 0.86)
1623
2058
  );
1624
2059
  }
1625
2060
  }
@@ -1678,6 +2113,9 @@ function evidenceCandidates(task) {
1678
2113
  "deepseek/deepseek-v4-pro"
1679
2114
  ];
1680
2115
  }
2116
+ if (task === "extraction") {
2117
+ return ["moonshot/kimi-k3", "google/gemini-3.5-flash", "anthropic/claude-sonnet-5"];
2118
+ }
1681
2119
  if (task === "reasoning_math") {
1682
2120
  return [
1683
2121
  "google/gemini-3.5-flash",
@@ -1734,9 +2172,12 @@ var PortfolioStrategy = class {
1734
2172
  const targetTier = (features.taskType === "reasoning_mcq" || features.taskType === "reasoning_math") && (base.tier === "SIMPLE" || base.tier === "MEDIUM") ? "REASONING" : base.tier;
1735
2173
  const tierConfig = tierConfigs[targetTier];
1736
2174
  const configuredCandidates = tierConfig ? getFallbackChain(targetTier, tierConfigs) : [];
2175
+ const unavailable = new Set(options.unavailableModels ?? []);
1737
2176
  const chain = [
1738
2177
  .../* @__PURE__ */ new Set([...configuredCandidates, ...evidenceCandidates(features.taskType)])
1739
- ].filter((model2) => typeof model2 === "string" && model2.length > 0);
2178
+ ].filter(
2179
+ (model2) => typeof model2 === "string" && model2.length > 0 && !unavailable.has(model2)
2180
+ );
1740
2181
  const eligible = chain.filter(
1741
2182
  (model2) => isEligible(model2, features, maxOutputTokens, options)
1742
2183
  );
@@ -1813,13 +2254,7 @@ var PortfolioStrategy = class {
1813
2254
  ...eligibleCandidates.filter(
1814
2255
  (model2) => !scoredModels.includes(model2) && !webResearchFallbackOrder.includes(model2)
1815
2256
  )
1816
- ] : features.taskType === "tool_agent" || features.taskType === "tool_agent_parallel" && features.agentDomain !== "other" ? [
1817
- ...scoredModels,
1818
- ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
1819
- ] : [
1820
- ...scoredModels,
1821
- ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))
1822
- ];
2257
+ ] : [...scoredModels, ...eligibleCandidates.filter((model2) => !scoredModels.includes(model2))];
1823
2258
  const model = ranked[0] ?? base.model;
1824
2259
  const selectedTierConfigs = {
1825
2260
  ...tierConfigs,
@@ -1856,7 +2291,7 @@ var PortfolioStrategy = class {
1856
2291
  }
1857
2292
  };
1858
2293
  var DEFAULT_ROUTING_CONFIG = {
1859
- version: "3.4",
2294
+ version: "3.5",
1860
2295
  strategy: "portfolio",
1861
2296
  portfolio: {
1862
2297
  auto: {
@@ -2910,186 +3345,249 @@ var DEFAULT_ROUTING_CONFIG = {
2910
3345
  // Below this confidence → ambiguous (null tier)
2911
3346
  confidenceThreshold: 0.7
2912
3347
  },
3348
+ // ─── Tier chains ───
3349
+ //
3350
+ // Catalog refresh 2026-08-29 (V3.5). Every chain below names only models
3351
+ // the public catalog lists (GET https://blockrun.ai/api/v1/models). Ids the
3352
+ // gateway withholds (`hidden: true`) — kimi-k2.5/k2.6/k2.7, the grok-4-fast
3353
+ // and grok-4-1-fast pairs, grok-4-0709, claude-opus-4.6, gemini-3-pro-preview,
3354
+ // the whole `free/*` namespace — were removed everywhere, including fallback
3355
+ // rungs, so a routed model is always one a user can find on blockrun.ai/models.
3356
+ //
3357
+ // Primaries moved only where portfolio.ts already carries calibration
3358
+ // evidence for the successor (Sonnet 5 over Sonnet 4.6, GPT-5 Mini for
3359
+ // agentic MEDIUM, Gemini 3.5 Flash where Kimi K2.7 was). Newcomers with no
3360
+ // trajectory evidence yet (gemini-3.6-flash, glm-5.3, glm-5.3-flash,
3361
+ // gpt-5.6-luna, grok-4.3, minimax-m3, qwen3.7-plus) enter as fallback rungs;
3362
+ // promotion waits for a calibration run, because version recency is not a
3363
+ // quality signal.
3364
+ //
3365
+ // Latency figures in comments are the 2026-08-29 gateway probe
3366
+ // (model-profiles.generated.json); prices are the catalog list.
2913
3367
  // Auto (balanced) tier configs - current default smart routing
2914
- // Benchmark-tuned 2026-03-16: balancing quality (retention) + latency
2915
3368
  tiers: {
2916
3369
  SIMPLE: {
2917
3370
  primary: "google/gemini-2.5-flash",
2918
- // 1,238ms, IQ 20, 60% retention (best) — fast AND quality
3371
+ // $0.30/$2.50 — 60% retention (best) in the 2026-03 run; still the fastest quality answer
2919
3372
  fallback: [
2920
3373
  "google/gemini-3-flash-preview",
2921
- // 1,398ms, IQ 46 — smarter fallback
3374
+ // $0.50/$3 — GPQA 5/6 in the 2026-07 calibration
3375
+ "google/gemini-3.5-flash-lite",
3376
+ // $0.30/$2.50, 1M ctx, thinking mode — same price as 2.5 Flash, newer generation
2922
3377
  "deepseek/deepseek-chat",
2923
- // V4 Flash chat ($0.20/$0.40, 1M ctx) — repriced 2026-04-24
2924
- "moonshot/kimi-k2.5",
2925
- // 1,646ms, IQ 47, strong quality
3378
+ // $0.14/$0.28, 1M ctx
2926
3379
  "google/gemini-3.1-flash-lite",
2927
- // $0.25/$1.50, 1M context — newest flash-lite
2928
- "google/gemini-2.5-flash-lite",
2929
- // 1,353ms, $0.10/$0.40
3380
+ // $0.25/$1.50, 1M ctx
3381
+ "openai/gpt-5.6-luna",
3382
+ // $0.20/$1.20, 1M ctx — GPT-5.6 cost tier (cut 2026-07-30)
2930
3383
  "openai/gpt-5.4-nano",
2931
- // $0.20/$1.25, 1M context
2932
- "xai/grok-4-fast-non-reasoning",
2933
- // 1,143ms, $0.20/$0.50 — fast fallback
2934
- "free/gpt-oss-120b"
2935
- // 1,252ms, FREE fallback (hidden from /v1/models but direct calls work)
3384
+ // $0.20/$1.25, 1M ctx
3385
+ "google/gemini-2.5-flash-lite",
3386
+ // $0.10/$0.40
3387
+ "nvidia/step-3.7-flash"
3388
+ // FREE backstop — NVIDIA free tier (probed 2026-08-21)
2936
3389
  ]
2937
3390
  },
2938
3391
  MEDIUM: {
2939
- primary: "moonshot/kimi-k2.7",
2940
- // $0.95/$4.00, 256K ctx, multi-modal + reasoning — Moonshot flagship; promoted from K2.6 (2026-06-14) after BlockRun added K2.7 + hid K2.6. Same price as K2.6.
3392
+ // Was moonshot/kimi-k2.7 (hidden 2026-08). Gemini 3.5 Flash is the
3393
+ // calibrated successor: MGSM 5/5, GPQA 4/6, extraction band (portfolio.ts).
3394
+ primary: "google/gemini-3.5-flash",
3395
+ // $1.50/$9, 1M ctx, vision + tools
2941
3396
  fallback: [
2942
- "moonshot/kimi-k2.6",
2943
- // identical-cost in-family hot swap (K2.6 still routable)
2944
- "moonshot/kimi-k2.5",
2945
- // $0.60/$3.00 — graceful-degradation backstop
3397
+ "google/gemini-3.6-flash",
3398
+ // $1.50/$7.50 — newest Flash, output 17% cheaper than 3.5; awaiting calibration
3399
+ "zai/glm-5.3-flash",
3400
+ // $0.15/$0.50, 1M ctx, vision + tools verified live 2026-08-27
3401
+ "openai/gpt-5.6-terra",
3402
+ // $2/$12, 1M ctx — GPT-5.6 balanced tier
2946
3403
  "google/gemini-3-flash-preview",
2947
- // 1,398ms, IQ 46 — nearly same IQ, faster + cheaper
3404
+ // $0.50/$3
2948
3405
  "deepseek/deepseek-chat",
2949
- // 1,431ms, IQ 32, 41% retention
3406
+ // $0.14/$0.28
2950
3407
  "google/gemini-2.5-flash",
2951
- // 1,238ms, 60% retention
3408
+ // $0.30/$2.50
3409
+ "minimax/minimax-m3",
3410
+ // $0.30/$1.20, 1M ctx
2952
3411
  "google/gemini-3.1-flash-lite",
2953
- // $0.25/$1.50, 1M context
2954
- "google/gemini-2.5-flash-lite",
2955
- // 1,353ms, $0.10/$0.40
2956
- "xai/grok-4-1-fast-non-reasoning",
2957
- // 1,244ms, fast fallback
2958
- "xai/grok-3-mini"
2959
- // 1,202ms, $0.30/$0.50
3412
+ // $0.25/$1.50
3413
+ "openai/gpt-5.6-luna",
3414
+ // $0.20/$1.20
3415
+ "google/gemini-2.5-flash-lite"
3416
+ // $0.10/$0.40
2960
3417
  ]
2961
3418
  },
2962
3419
  COMPLEX: {
2963
3420
  primary: "google/gemini-3.1-pro",
2964
- // 1,609ms, IQ 57 — fast flagship quality
3421
+ // $2/$12 — proven long-context flagship (portfolio.ts long_context lead)
2965
3422
  fallback: [
2966
- "google/gemini-3-flash-preview",
2967
- // 1,398ms, IQ 46 — fast + smart
2968
- "xai/grok-4-0709",
2969
- // 1,348ms, IQ 41
2970
- "google/gemini-2.5-pro",
2971
- // 1,294ms
3423
+ "google/gemini-3.6-flash",
3424
+ // $1.50/$7.50 — Pro-level quality at Flash price (Google's claim; uncalibrated here)
3425
+ "google/gemini-3.5-flash",
3426
+ // $1.50/$9 — calibrated
2972
3427
  "anthropic/claude-sonnet-5",
2973
- // near-Opus quality at Sonnet cost, 1M ctx
3428
+ // $3/$15 — near-Opus quality, tau2 + Terminal-Bench calibrated
3429
+ "xai/grok-4.5",
3430
+ // $2.50/$9 — 503-resistant, independent infra (was grok-4-0709, now hidden)
3431
+ "google/gemini-2.5-pro",
3432
+ // $1.25/$10
2974
3433
  "anthropic/claude-sonnet-4.6",
2975
- // 2,110ms, IQ 52 — quality fallback
2976
- "deepseek/deepseek-chat",
2977
- // 1,431ms, IQ 32
2978
- "google/gemini-2.5-flash",
2979
- // 1,238ms, IQ 20 — cheap last resort
3434
+ // $3/$15
2980
3435
  "openai/gpt-5.6-terra",
2981
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3436
+ // $2/$12 — GPT-5.6 balanced tier (Sol excluded: #202)
2982
3437
  "openai/gpt-5.5",
2983
- // Prior OpenAI flagship — 1M+ ctx, native agent + computer use; benchmark TBD
2984
- "openai/gpt-5.4"
2985
- // 6,213ms, IQ 57 — previous flagship, benchmarked
3438
+ // $5/$30 — prior OpenAI flagship
3439
+ "openai/gpt-5.4",
3440
+ // $2.50/$15 — previous flagship, benchmarked
3441
+ "zai/glm-5.3",
3442
+ // $1.40/$4.40, 1M ctx, always-on thinking — verified live 2026-08-19
3443
+ "moonshot/kimi-k3",
3444
+ // $3/$15, 1M ctx — Moonshot flagship (K2.7 successor)
3445
+ "deepseek/deepseek-v4-pro",
3446
+ // $0.435/$0.87 — strongest open-weight reasoner
3447
+ "deepseek/deepseek-chat",
3448
+ // $0.14/$0.28 — cheap last resort
3449
+ "google/gemini-2.5-flash"
3450
+ // $0.30/$2.50
2986
3451
  ]
2987
3452
  },
2988
3453
  REASONING: {
2989
- primary: "xai/grok-4-1-fast-reasoning",
2990
- // 1,454ms, $0.20/$0.50
3454
+ // Was xai/grok-4-1-fast-reasoning ($0.20/$0.50, hidden 2026-08). DeepSeek
3455
+ // Reasoner is the cheapest listed reasoner at the same 1M context.
3456
+ primary: "deepseek/deepseek-reasoner",
3457
+ // $0.14/$0.28, 1M ctx
2991
3458
  fallback: [
2992
- "xai/grok-4-fast-reasoning",
2993
- // 1,298ms, $0.20/$0.50
2994
- "deepseek/deepseek-reasoner",
2995
- // V4 Flash thinking ($0.20/$0.40, 1M ctx)
2996
3459
  "deepseek/deepseek-v4-pro",
2997
- // V4 Pro flagship ($0.50/$1.00 promo through 2026-05-31, list $2/$4) — strongest open-weight reasoner
3460
+ // $0.435/$0.87 — calibrated reasoning band 0.95
3461
+ "xai/grok-4.3",
3462
+ // $1.50/$4, 1M ctx — xAI reasoning model, vision
3463
+ "qwen/qwen3.7-plus",
3464
+ // $0.32/$1.28, 1M ctx — reasoning; needs a generous max_tokens (thinking is billed)
3465
+ "google/gemini-3.5-flash",
3466
+ // $1.50/$9 — MGSM 5/5
2998
3467
  "openai/o4-mini",
2999
- // 2,328ms ($1.10/$4.40)
3468
+ // $1.10/$4.40
3000
3469
  "openai/o3"
3001
- // 2,862ms
3470
+ // $2/$8
3002
3471
  ]
3003
3472
  }
3004
3473
  },
3005
3474
  // Eco tier configs - absolute cheapest (blockrun/eco)
3006
3475
  ecoTiers: {
3007
3476
  SIMPLE: {
3008
- primary: "free/gpt-oss-120b",
3009
- // FREE! $0.00/$0.00 — heavy user default
3477
+ primary: "nvidia/step-3.7-flash",
3478
+ // FREE — NVIDIA free tier flagship
3010
3479
  fallback: [
3011
- "free/gpt-oss-20b",
3012
- // FREE — smaller, faster
3013
- "free/deepseek-v4-flash",
3014
- // FREE — 1M ctx; slow (~10 tok/s, 07-28 probe) but completes
3015
- // seed-oss-36b sat here as the free coder until it EOL'd 2026-08-03 (HTTP 410).
3016
- // gpt-oss-120b/20b already head this chain, so the rung is dropped, not retargeted.
3017
- "google/gemini-3.1-flash-lite",
3018
- // $0.25/$1.50 — newest flash-lite
3019
- "openai/gpt-5.4-nano",
3020
- // $0.20/$1.25 — fast nano
3480
+ "nvidia/nemotron-nano-9b-v2",
3481
+ // FREE — compact + fast, high-volume light tasks
3482
+ // The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash
3483
+ // 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400
3484
+ // 2026-08-21). Each retirement retargets the two free rungs to the
3485
+ // current free tier; the paid rungs below never move.
3021
3486
  "google/gemini-2.5-flash-lite",
3022
- // $0.10/$0.40
3023
- "xai/grok-4-fast-non-reasoning"
3024
- // $0.20/$0.50
3487
+ // $0.10/$0.40 — cheapest paid rung
3488
+ "zai/glm-5.3-flash",
3489
+ // $0.15/$0.50, 1M ctx, vision + tools
3490
+ "openai/gpt-5.6-luna",
3491
+ // $0.20/$1.20, 1M ctx
3492
+ "openai/gpt-5.4-nano",
3493
+ // $0.20/$1.25
3494
+ "google/gemini-3.1-flash-lite"
3495
+ // $0.25/$1.50
3025
3496
  ]
3026
3497
  },
3027
3498
  MEDIUM: {
3028
- primary: "google/gemini-3.1-flash-lite",
3029
- // $0.25/$1.50 — newest flash-lite
3499
+ primary: "zai/glm-5.3-flash",
3500
+ // $0.15/$0.50, 1M ctx, vision + tools verified live — cheapest full-capability model
3030
3501
  fallback: [
3502
+ "deepseek/deepseek-chat",
3503
+ // $0.14/$0.28
3504
+ "google/gemini-3.1-flash-lite",
3505
+ // $0.25/$1.50
3506
+ "openai/gpt-5.6-luna",
3507
+ // $0.20/$1.20
3031
3508
  "openai/gpt-5.4-nano",
3032
3509
  // $0.20/$1.25
3033
3510
  "google/gemini-2.5-flash-lite",
3034
3511
  // $0.10/$0.40
3035
- "xai/grok-4-fast-non-reasoning",
3036
3512
  "google/gemini-2.5-flash"
3513
+ // $0.30/$2.50
3037
3514
  ]
3038
3515
  },
3039
3516
  COMPLEX: {
3040
- primary: "google/gemini-3.1-flash-lite",
3041
- // $0.25/$1.50
3517
+ primary: "zai/glm-5.3-flash",
3518
+ // $0.15/$0.50, 1M ctx
3042
3519
  fallback: [
3043
- "google/gemini-2.5-flash-lite",
3044
- "xai/grok-4-0709",
3045
- "google/gemini-2.5-flash",
3046
- "deepseek/deepseek-chat"
3520
+ "deepseek/deepseek-chat",
3521
+ // $0.14/$0.28, 1M ctx
3522
+ "minimax/minimax-m3",
3523
+ // $0.30/$1.20, 1M ctx
3524
+ "deepseek/deepseek-v4-pro",
3525
+ // $0.435/$0.87
3526
+ "google/gemini-3.1-flash-lite",
3527
+ // $0.25/$1.50
3528
+ "google/gemini-2.5-flash"
3529
+ // $0.30/$2.50
3047
3530
  ]
3048
3531
  },
3049
3532
  REASONING: {
3050
- primary: "xai/grok-4-1-fast-reasoning",
3051
- // $0.20/$0.50
3533
+ primary: "deepseek/deepseek-reasoner",
3534
+ // $0.14/$0.28, 1M ctx — cheapest listed reasoner
3052
3535
  fallback: [
3053
- "xai/grok-4-fast-reasoning",
3054
- "deepseek/deepseek-reasoner",
3055
- // V4 Flash thinking — $0.20/$0.40
3056
- "deepseek/deepseek-v4-pro"
3057
- // V4 Pro flagship — $0.50/$1.00 promo, post-promo $2/$4
3536
+ "deepseek/deepseek-v4-pro",
3537
+ // $0.435/$0.87
3538
+ "qwen/qwen3.7-plus",
3539
+ // $0.32/$1.28 — reasoning
3540
+ "minimax/minimax-m3",
3541
+ // $0.30/$1.20 — reasoning + coding
3542
+ "zai/glm-5.3-flash"
3543
+ // $0.15/$0.50 — reasoning tokens alongside content
3058
3544
  ]
3059
3545
  }
3060
3546
  },
3061
3547
  // Premium tier configs - best quality (blockrun/premium)
3062
- // codex=complex coding, kimi=simple coding, sonnet=reasoning/instructions, opus=architecture/PM/audits
3548
+ // codex=complex coding, flash=simple coding, sonnet=reasoning/instructions, fable/opus=architecture/PM/audits
3063
3549
  premiumTiers: {
3064
3550
  SIMPLE: {
3065
- primary: "moonshot/kimi-k2.7",
3066
- // $0.95/$4.00 - Moonshot flagship (256K ctx, multi-modal + reasoning); promoted from K2.6 (2026-06-14), same price
3551
+ // Was moonshot/kimi-k2.7 (hidden 2026-08).
3552
+ primary: "google/gemini-3.5-flash",
3553
+ // $1.50/$9, 1M ctx, vision + tools — calibrated
3067
3554
  fallback: [
3068
- "moonshot/kimi-k2.6",
3069
- // identical-cost in-family hot swap (K2.6 still routable)
3070
- "moonshot/kimi-k2.5",
3071
- // $0.60/$3.00 - proven reliable backstop when Moonshot direct API falters
3072
- "google/gemini-2.5-flash",
3073
- // 60% retention, fast growth
3555
+ "google/gemini-3.6-flash",
3556
+ // $1.50/$7.50 — newest Flash
3074
3557
  "anthropic/claude-haiku-4.5",
3075
- "google/gemini-2.5-flash-lite",
3558
+ // $1/$5
3559
+ "zai/glm-5.3",
3560
+ // $1.40/$4.40, 1M ctx
3561
+ "google/gemini-2.5-flash",
3562
+ // $0.30/$2.50
3563
+ "google/gemini-3.5-flash-lite",
3564
+ // $0.30/$2.50
3076
3565
  "deepseek/deepseek-chat"
3566
+ // $0.14/$0.28
3077
3567
  ]
3078
3568
  },
3079
3569
  MEDIUM: {
3080
3570
  primary: "openai/gpt-5.3-codex",
3081
- // $1.75/$14 - 400K context, 128K output, replaces 5.2
3571
+ // $1.75/$14 - 400K context, 128K output — code_edit/debug lead (portfolio.ts)
3082
3572
  fallback: [
3083
- "moonshot/kimi-k2.7",
3084
- // Moonshot flagship
3085
- "moonshot/kimi-k2.6",
3086
- "moonshot/kimi-k2.5",
3087
- "google/gemini-2.5-flash",
3088
- // 60% retention, good coding capability
3089
- "google/gemini-2.5-pro",
3090
- "xai/grok-4-0709",
3091
3573
  "anthropic/claude-sonnet-5",
3092
- "anthropic/claude-sonnet-4.6"
3574
+ // $3/$15 — code_agent band 0.98
3575
+ "moonshot/kimi-k3",
3576
+ // $3/$15, 1M ctx — Moonshot flagship
3577
+ "zai/glm-5.3",
3578
+ // $1.40/$4.40 — long-horizon coding
3579
+ "google/gemini-3.6-flash",
3580
+ // $1.50/$7.50
3581
+ "google/gemini-3.5-flash",
3582
+ // $1.50/$9
3583
+ "google/gemini-2.5-pro",
3584
+ // $1.25/$10
3585
+ "xai/grok-4.5",
3586
+ // $2.50/$9
3587
+ "anthropic/claude-sonnet-4.6",
3588
+ // $3/$15
3589
+ "openai/gpt-5.6-terra"
3590
+ // $2/$12
3093
3591
  ]
3094
3592
  },
3095
3593
  COMPLEX: {
@@ -3099,8 +3597,8 @@ var DEFAULT_ROUTING_CONFIG = {
3099
3597
  // Best quality for complex tasks — Mythos-class flagship above Opus ($10/$50, 1M ctx, always-on thinking)
3100
3598
  // Fallback chain de-Gemini'd 2026-04-22: when Anthropic 503s, Gemini is
3101
3599
  // also prone to "high demand" 503s (correlated failure — everyone falls
3102
- // back to Google at the same time). Prefer xAI Grok → Moonshot → OpenAI
3103
- // flagship → DeepSeek → NVIDIA free instead.
3600
+ // back to Google at the same time). Prefer in-family → xAI → Moonshot →
3601
+ // OpenAI flagship → Z.AI → DeepSeek → NVIDIA free instead.
3104
3602
  fallback: [
3105
3603
  "anthropic/claude-opus-5",
3106
3604
  // in-family hot swap first (half the price, 1M ctx + adaptive thinking)
@@ -3108,52 +3606,54 @@ var DEFAULT_ROUTING_CONFIG = {
3108
3606
  // in-family hot swap (identical cost to 5)
3109
3607
  "anthropic/claude-opus-4.7",
3110
3608
  // in-family hot swap (identical cost to 4.8)
3111
- "anthropic/claude-opus-4.6",
3112
- // in-family hot swap
3113
3609
  "anthropic/claude-sonnet-5",
3114
3610
  // Sonnet-tier drop-down, near-Opus quality
3115
3611
  "anthropic/claude-sonnet-4.6",
3116
3612
  "xai/grok-4.5",
3117
- // xAI flagship — 503-resistant, direct-xAI SKU (added 2026-07-14)
3118
- "xai/grok-4-0709",
3119
- // 503-resistant flagship
3120
- "moonshot/kimi-k2.7",
3613
+ // xAI flagship — 503-resistant, direct-xAI SKU
3614
+ "moonshot/kimi-k3",
3121
3615
  // Moonshot flagship, independent infra
3122
- "moonshot/kimi-k2.6",
3123
- "moonshot/kimi-k2.5",
3124
3616
  "openai/gpt-5.6-terra",
3125
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3617
+ // GPT-5.6 balanced tier — stable (Sol excluded: #202)
3126
3618
  "openai/gpt-5.5",
3127
3619
  // Prior OpenAI flagship — 1M+ ctx, native agent + computer use
3128
3620
  "openai/gpt-5.4",
3129
3621
  // Previous flagship (slow but stable, benchmarked at 6,213ms)
3130
3622
  "openai/gpt-5.3-codex",
3623
+ "zai/glm-5.3",
3624
+ // Z.AI flagship, 1M ctx
3625
+ "deepseek/deepseek-v4-pro",
3626
+ // strongest open-weight reasoner
3131
3627
  "deepseek/deepseek-chat",
3132
3628
  // Cheap, reliable
3133
- "free/gpt-oss-120b"
3134
- // NVIDIA free ultimate backstop (was seed-oss-36b; EOL'd 2026-08-03)
3629
+ "nvidia/step-3.7-flash"
3630
+ // NVIDIA free ultimate backstop
3135
3631
  ]
3136
3632
  },
3137
3633
  REASONING: {
3138
- primary: "anthropic/claude-sonnet-4.6",
3139
- // 2,110ms, $3/$15 - best for reasoning/instructions
3634
+ // Sonnet 5 promoted over Sonnet 4.6 (same price; reasoning band 0.98 for both,
3635
+ // plus Sonnet 5's tau2/BrowseComp trajectory evidence).
3636
+ primary: "anthropic/claude-sonnet-5",
3637
+ // $3/$15, 1M ctx, adaptive thinking
3140
3638
  fallback: [
3141
- "anthropic/claude-sonnet-5",
3142
- // in-family hot swap — same cost, adaptive thinking, 1M ctx
3639
+ "anthropic/claude-sonnet-4.6",
3640
+ // in-family hot swap — same cost
3143
3641
  "anthropic/claude-opus-5",
3144
3642
  // Newest flagship Opus w/ adaptive thinking
3145
3643
  "anthropic/claude-opus-4.8",
3146
3644
  // Prior flagship Opus — identical cost to 5
3147
3645
  "anthropic/claude-opus-4.7",
3148
3646
  // Flagship Opus w/ adaptive thinking
3149
- "anthropic/claude-opus-4.6",
3150
- // 2,139ms
3151
- "xai/grok-4-1-fast-reasoning",
3152
- // 1,454ms, cheap fast reasoning
3647
+ "xai/grok-4.5",
3648
+ // reasoning band 0.94
3649
+ "deepseek/deepseek-v4-pro",
3650
+ // reasoning band 0.95
3651
+ "xai/grok-4.3",
3652
+ // $1.50/$4 — xAI reasoning model
3153
3653
  "openai/o4-mini",
3154
- // 2,328ms ($1.10/$4.40)
3654
+ // $1.10/$4.40
3155
3655
  "openai/o3"
3156
- // 2,862ms
3656
+ // $2/$8
3157
3657
  ]
3158
3658
  }
3159
3659
  },
@@ -3163,101 +3663,102 @@ var DEFAULT_ROUTING_CONFIG = {
3163
3663
  primary: "openai/gpt-4o-mini",
3164
3664
  // $0.15/$0.60 - best tool compliance at lowest cost
3165
3665
  fallback: [
3166
- "moonshot/kimi-k2.5",
3167
- // 1,646ms, strong tool use quality
3666
+ "openai/gpt-5.6-luna",
3667
+ // $0.20/$1.20 — lightweight agentic tier of GPT-5.6
3668
+ "zai/glm-5.3-flash",
3669
+ // $0.15/$0.50 — tool calls verified live 2026-08-27
3168
3670
  "anthropic/claude-haiku-4.5",
3169
- // 2,305ms
3170
- "xai/grok-4-1-fast-non-reasoning"
3171
- // 1,244ms, fast fallback
3671
+ // $1/$5
3672
+ "google/gemini-2.5-flash"
3673
+ // $0.30/$2.50
3172
3674
  ]
3173
3675
  },
3174
3676
  MEDIUM: {
3175
- primary: "moonshot/kimi-k2.7",
3176
- // $0.95/$4.00 — Moonshot flagship, strong tool use; promoted from K2.6 (2026-06-14) after BlockRun added K2.7 + hid K2.6. Same price.
3677
+ // Was moonshot/kimi-k2.7 (hidden 2026-08). GPT-5 Mini carries the
3678
+ // Terminal-Bench and tau2 trajectory evidence in portfolio.ts.
3679
+ primary: "openai/gpt-5-mini",
3680
+ // $0.25/$2 — 4/7 Terminal-Bench, 5/6 tau2 airline
3177
3681
  fallback: [
3178
- "moonshot/kimi-k2.6",
3179
- // identical-cost in-family hot swap (K2.6 still routable)
3180
- "moonshot/kimi-k2.5",
3181
- // $0.60/$3.00 — graceful-degradation backstop
3182
- "xai/grok-4-1-fast-non-reasoning",
3183
- // 1,244ms, fast fallback
3682
+ "google/gemini-3.5-flash",
3683
+ // $1.50/$9 — tool_agent band 0.88
3684
+ "zai/glm-5.3-flash",
3685
+ // $0.15/$0.50 — tools verified
3686
+ "openai/gpt-5.6-terra",
3687
+ // $2/$12
3184
3688
  "openai/gpt-4o-mini",
3185
- // 2,764ms, reliable tool calling
3689
+ // $0.15/$0.60 — reliable tool calling
3186
3690
  "anthropic/claude-haiku-4.5",
3187
- // 2,305ms
3188
- "deepseek/deepseek-chat"
3189
- // 1,431ms
3691
+ // $1/$5
3692
+ "deepseek/deepseek-chat",
3693
+ // $0.14/$0.28
3694
+ "moonshot/kimi-k3"
3695
+ // $3/$15 — tool_agent band 0.85
3190
3696
  ]
3191
3697
  },
3192
3698
  COMPLEX: {
3193
- primary: "anthropic/claude-sonnet-4.6",
3194
- // 2,110ms — best agentic quality
3699
+ // Sonnet 5 promoted over Sonnet 4.6: tau2 airline + retail reward 1.0,
3700
+ // Terminal-Bench safety band lead (portfolio.ts).
3701
+ primary: "anthropic/claude-sonnet-5",
3702
+ // $3/$15 — best agentic quality per trajectory evidence
3195
3703
  // Fallback chain de-Gemini'd 2026-04-22: Gemini's "high demand" 503s
3196
3704
  // correlate with Anthropic outages (everyone falls back together).
3197
3705
  // Prefer 503-resistant providers first.
3198
3706
  fallback: [
3199
- "anthropic/claude-sonnet-5",
3200
- // in-family hot swap — same cost, near-Opus agentic quality
3707
+ "anthropic/claude-sonnet-4.6",
3708
+ // in-family hot swap — same cost
3201
3709
  "anthropic/claude-opus-5",
3202
3710
  // Newest flagship Opus — in-family hot swap
3203
3711
  "anthropic/claude-opus-4.8",
3204
3712
  // Prior flagship Opus — identical cost to 5
3205
3713
  "anthropic/claude-opus-4.7",
3206
3714
  // Flagship Opus — in-family hot swap
3207
- "anthropic/claude-opus-4.6",
3208
- // 2,139ms
3209
- "xai/grok-4-0709",
3210
- // 1,348ms — strong tool use, independent infra
3211
- "moonshot/kimi-k2.7",
3212
- // Moonshot flagship — strong tool use, independent infra
3213
- "moonshot/kimi-k2.5",
3214
- // cost-stability backstop
3715
+ "xai/grok-4.5",
3716
+ // xAI flagship — strong tool use, independent infra
3717
+ "moonshot/kimi-k3",
3718
+ // Moonshot flagship — independent infra
3215
3719
  "openai/gpt-5.6-terra",
3216
- // GPT-5.6 balanced tier — newest generation, stable (Sol excluded: #202)
3720
+ // GPT-5.6 balanced tier — stable (Sol excluded: #202)
3217
3721
  "openai/gpt-5.5",
3218
3722
  // Prior flagship — native agent + computer use (exactly the agentic-tier use case)
3219
3723
  "openai/gpt-5.4",
3220
- // Previous flagship — 6,213ms, reliable
3724
+ // Previous flagship — reliable
3725
+ "openai/gpt-5.3-codex",
3726
+ // code_agent lead
3727
+ "zai/glm-5.3",
3728
+ // long-horizon coding
3729
+ "deepseek/deepseek-v4-pro",
3730
+ // retail high-risk 3/3
3221
3731
  "deepseek/deepseek-chat",
3222
- // 1,431ms — cheap, reliable
3223
- "free/gpt-oss-120b"
3224
- // NVIDIA free ultimate backstop (was seed-oss-36b; EOL'd 2026-08-03)
3732
+ // cheap, reliable
3733
+ "nvidia/step-3.7-flash"
3734
+ // NVIDIA free ultimate backstop
3225
3735
  ]
3226
3736
  },
3227
3737
  REASONING: {
3228
- primary: "anthropic/claude-sonnet-4.6",
3229
- // 2,110ms — strong tool use + reasoning
3738
+ primary: "anthropic/claude-sonnet-5",
3739
+ // $3/$15 — strong tool use + adaptive thinking
3230
3740
  fallback: [
3231
- "anthropic/claude-sonnet-5",
3232
- // in-family hot swap — same cost, adaptive thinking
3741
+ "anthropic/claude-sonnet-4.6",
3742
+ // in-family hot swap — same cost
3233
3743
  "anthropic/claude-opus-5",
3234
3744
  // Newest flagship Opus w/ adaptive thinking
3235
3745
  "anthropic/claude-opus-4.8",
3236
3746
  // Prior flagship Opus — identical cost to 5
3237
3747
  "anthropic/claude-opus-4.7",
3238
3748
  // Flagship Opus w/ adaptive thinking
3239
- "anthropic/claude-opus-4.6",
3240
- // 2,139ms
3241
- "xai/grok-4-1-fast-reasoning",
3242
- // 1,454ms
3749
+ "xai/grok-4.5",
3750
+ // reasoning band 0.94
3751
+ "deepseek/deepseek-v4-pro",
3752
+ // reasoning band 0.95
3243
3753
  "deepseek/deepseek-reasoner"
3244
- // 1,454ms
3754
+ // $0.14/$0.28
3245
3755
  ]
3246
3756
  }
3247
3757
  },
3248
- // Time-windowed promotions — auto-applied when active, ignored when expired
3249
- promotions: [
3250
- {
3251
- name: "GLM-5.1 Launch Promo ($0.001 flat)",
3252
- startDate: "2026-04-01",
3253
- endDate: "2026-05-01",
3254
- tierOverrides: {
3255
- SIMPLE: { primary: "zai/glm-5.1" }
3256
- },
3257
- profiles: ["auto"]
3258
- // only auto profile — eco stays free, premium stays premium
3259
- }
3260
- ],
3758
+ // Time-windowed promotions — auto-applied when active, ignored when expired.
3759
+ // The GLM-5.1 launch promo (2026-04-01 → 2026-05-01) was the last entry and
3760
+ // has expired; the list is kept empty so the mechanism stays wired.
3761
+ promotions: [],
3261
3762
  overrides: {
3262
3763
  maxTokensForceComplex: 1e5,
3263
3764
  structuredOutputMinTier: "MEDIUM",
@@ -3599,7 +4100,7 @@ async function createSolanaPaymentPayload(secretKey, fromAddress, recipient, amo
3599
4100
  }
3600
4101
  return null;
3601
4102
  };
3602
- let entry = await getBlockhashEntry(connection, rpcUrl, false);
4103
+ let entry = await getBlockhashEntry(connection, rpcUrl, options.forceFreshBlockhash ?? false);
3603
4104
  let serializedTx = findDistinctTx(entry);
3604
4105
  if (serializedTx === null) {
3605
4106
  entry = await getBlockhashEntry(connection, rpcUrl, true);
@@ -3850,7 +4351,7 @@ function getCostSummary() {
3850
4351
  }
3851
4352
 
3852
4353
  // src/version.ts
3853
- var SDK_VERSION = "3.13.1";
4354
+ var SDK_VERSION = "3.13.4";
3854
4355
  var USER_AGENT = `blockrun-ts/${SDK_VERSION}`;
3855
4356
 
3856
4357
  // src/client.ts
@@ -8504,6 +9005,68 @@ async function getOrCreateSolanaWallet() {
8504
9005
  var SOLANA_API_URL = "https://sol.blockrun.ai/api";
8505
9006
  var DEFAULT_MAX_TOKENS2 = 1024;
8506
9007
  var DEFAULT_TIMEOUT14 = 6e4;
9008
+ var STALE_BLOCKHASH_RETRY_BACKOFFS_MS = [500, 2e3];
9009
+ var MAX_PAYMENT_FAILURE_BYTES = 64 * 1024;
9010
+ var SafeStaleBlockhashError = class extends PaymentError {
9011
+ constructor() {
9012
+ super("Payment verification used an expired Solana blockhash; retrying with a fresh quote.");
9013
+ this.name = "SafeStaleBlockhashError";
9014
+ }
9015
+ };
9016
+ function normalizePaymentSignal(value) {
9017
+ return typeof value === "string" ? value.toLowerCase().replace(/[_\-\s:]/g, "") : "";
9018
+ }
9019
+ async function readPaymentFailureBody(response) {
9020
+ const reader = response.body?.getReader();
9021
+ if (!reader) return "";
9022
+ const decoder = new TextDecoder();
9023
+ let total = 0;
9024
+ let text = "";
9025
+ try {
9026
+ for (; ; ) {
9027
+ const { done, value } = await reader.read();
9028
+ if (done) return text + decoder.decode();
9029
+ total += value.byteLength;
9030
+ if (total > MAX_PAYMENT_FAILURE_BYTES) {
9031
+ void reader.cancel();
9032
+ return null;
9033
+ }
9034
+ text += decoder.decode(value, { stream: true });
9035
+ }
9036
+ } catch {
9037
+ return null;
9038
+ }
9039
+ }
9040
+ async function isSafeStaleBlockhashResponse(response) {
9041
+ const length = Number(response.headers.get("content-length") || "0");
9042
+ if (Number.isFinite(length) && length > MAX_PAYMENT_FAILURE_BYTES) return false;
9043
+ const text = await readPaymentFailureBody(response);
9044
+ if (text === null) return false;
9045
+ let body;
9046
+ try {
9047
+ const parsed = JSON.parse(text);
9048
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return false;
9049
+ body = parsed;
9050
+ } catch {
9051
+ return false;
9052
+ }
9053
+ const nested = body.error && typeof body.error === "object" && !Array.isArray(body.error) ? body.error : void 0;
9054
+ const errorLabel = typeof body.error === "string" ? body.error : "";
9055
+ const code = normalizePaymentSignal(body.code ?? nested?.code);
9056
+ const reason = normalizePaymentSignal(body.reason);
9057
+ const detail = normalizePaymentSignal(body.invalidMessage);
9058
+ const message = normalizePaymentSignal(nested?.message ?? body.message);
9059
+ const label = normalizePaymentSignal(errorLabel);
9060
+ if (code.includes("settlementfailed") || label.includes("settlementfailed") || message.includes("settlementfailed")) return false;
9061
+ const verifyPhase = code === "paymentinvalid" || label.includes("verificationfailed") || message.includes("verificationfailed");
9062
+ if (!verifyPhase) return false;
9063
+ return code === "paymentblockhashstale" || detail.includes("blockhashnotfound") || detail.includes("blockheightexceeded") || reason === "expiredsignature" || message.includes("expiredsignature");
9064
+ }
9065
+ async function waitForStaleRetry(attempt) {
9066
+ await new Promise(
9067
+ (resolve) => setTimeout(resolve, STALE_BLOCKHASH_RETRY_BACKOFFS_MS[attempt])
9068
+ );
9069
+ }
8507
9070
  var DEFAULT_SOLANA_RPC_URL = "https://sol.blockrun.ai/api/v1/solana/rpc";
8508
9071
  function resolveRpcConfig(rpcUrl, rpcHeaders) {
8509
9072
  const env = typeof process !== "undefined" && process.env ? process.env : {};
@@ -8950,26 +9513,34 @@ var SolanaLLMClient = class {
8950
9513
  }
8951
9514
  async requestWithPayment(endpoint, body) {
8952
9515
  const url = `${this.apiUrl}${endpoint}`;
8953
- const response = await this.fetchWithTimeout(url, {
8954
- method: "POST",
8955
- headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
8956
- body: JSON.stringify(body)
8957
- });
8958
- if (response.status === 402) {
8959
- return this.handlePaymentAndRetry(url, body, response);
8960
- }
8961
- if (!response.ok) {
8962
- let errorBody;
8963
- try {
8964
- errorBody = await response.json();
8965
- } catch {
8966
- errorBody = { error: "Request failed" };
9516
+ for (let staleRetries = 0; ; ) {
9517
+ const response = await this.fetchWithTimeout(url, {
9518
+ method: "POST",
9519
+ headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9520
+ body: JSON.stringify(body)
9521
+ });
9522
+ if (response.status === 402) {
9523
+ try {
9524
+ return await this.handlePaymentAndRetry(url, body, response, staleRetries > 0);
9525
+ } catch (error) {
9526
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9527
+ await waitForStaleRetry(staleRetries++);
9528
+ continue;
9529
+ }
8967
9530
  }
8968
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9531
+ if (!response.ok) {
9532
+ let errorBody;
9533
+ try {
9534
+ errorBody = await response.json();
9535
+ } catch {
9536
+ errorBody = { error: "Request failed" };
9537
+ }
9538
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9539
+ }
9540
+ return response.json();
8969
9541
  }
8970
- return response.json();
8971
9542
  }
8972
- async handlePaymentAndRetry(url, body, response) {
9543
+ async handlePaymentAndRetry(url, body, response, forceFreshBlockhash = false) {
8973
9544
  let paymentHeader = response.headers.get("payment-required");
8974
9545
  if (!paymentHeader) {
8975
9546
  try {
@@ -9011,7 +9582,8 @@ var SolanaLLMClient = class {
9011
9582
  extra: details.extra,
9012
9583
  extensions,
9013
9584
  rpcUrl: this.rpcUrl,
9014
- rpcHeaders: this.rpcHeaders
9585
+ rpcHeaders: this.rpcHeaders,
9586
+ forceFreshBlockhash
9015
9587
  }
9016
9588
  );
9017
9589
  const retryResponse = await this.fetchWithTimeout(url, {
@@ -9024,6 +9596,9 @@ var SolanaLLMClient = class {
9024
9596
  body: JSON.stringify(body)
9025
9597
  });
9026
9598
  if (retryResponse.status === 402) {
9599
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9600
+ throw new SafeStaleBlockhashError();
9601
+ }
9027
9602
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9028
9603
  }
9029
9604
  if (!retryResponse.ok) {
@@ -9042,26 +9617,34 @@ var SolanaLLMClient = class {
9042
9617
  }
9043
9618
  async requestWithPaymentRaw(endpoint, body) {
9044
9619
  const url = `${this.apiUrl}${endpoint}`;
9045
- const response = await this.fetchWithTimeout(url, {
9046
- method: "POST",
9047
- headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9048
- body: JSON.stringify(body)
9049
- });
9050
- if (response.status === 402) {
9051
- return this.handlePaymentAndRetryRaw(url, body, response);
9052
- }
9053
- if (!response.ok) {
9054
- let errorBody;
9055
- try {
9056
- errorBody = await response.json();
9057
- } catch {
9058
- errorBody = { error: "Request failed" };
9620
+ for (let staleRetries = 0; ; ) {
9621
+ const response = await this.fetchWithTimeout(url, {
9622
+ method: "POST",
9623
+ headers: { "Content-Type": "application/json", "User-Agent": USER_AGENT },
9624
+ body: JSON.stringify(body)
9625
+ });
9626
+ if (response.status === 402) {
9627
+ try {
9628
+ return await this.handlePaymentAndRetryRaw(url, body, response, staleRetries > 0);
9629
+ } catch (error) {
9630
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9631
+ await waitForStaleRetry(staleRetries++);
9632
+ continue;
9633
+ }
9059
9634
  }
9060
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9635
+ if (!response.ok) {
9636
+ let errorBody;
9637
+ try {
9638
+ errorBody = await response.json();
9639
+ } catch {
9640
+ errorBody = { error: "Request failed" };
9641
+ }
9642
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9643
+ }
9644
+ return response.json();
9061
9645
  }
9062
- return response.json();
9063
9646
  }
9064
- async handlePaymentAndRetryRaw(url, body, response) {
9647
+ async handlePaymentAndRetryRaw(url, body, response, forceFreshBlockhash = false) {
9065
9648
  let paymentHeader = response.headers.get("payment-required");
9066
9649
  if (!paymentHeader) {
9067
9650
  try {
@@ -9103,7 +9686,8 @@ var SolanaLLMClient = class {
9103
9686
  extra: details.extra,
9104
9687
  extensions,
9105
9688
  rpcUrl: this.rpcUrl,
9106
- rpcHeaders: this.rpcHeaders
9689
+ rpcHeaders: this.rpcHeaders,
9690
+ forceFreshBlockhash
9107
9691
  }
9108
9692
  );
9109
9693
  const retryResponse = await this.fetchWithTimeout(url, {
@@ -9116,6 +9700,9 @@ var SolanaLLMClient = class {
9116
9700
  body: JSON.stringify(body)
9117
9701
  });
9118
9702
  if (retryResponse.status === 402) {
9703
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9704
+ throw new SafeStaleBlockhashError();
9705
+ }
9119
9706
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9120
9707
  }
9121
9708
  if (!retryResponse.ok) {
@@ -9135,25 +9722,33 @@ var SolanaLLMClient = class {
9135
9722
  async getWithPaymentRaw(endpoint, params) {
9136
9723
  const query = params ? "?" + new URLSearchParams(params).toString() : "";
9137
9724
  const url = `${this.apiUrl}${endpoint}${query}`;
9138
- const response = await this.fetchWithTimeout(url, {
9139
- method: "GET",
9140
- headers: { "User-Agent": USER_AGENT }
9141
- });
9142
- if (response.status === 402) {
9143
- return this.handleGetPaymentAndRetryRaw(url, endpoint, params, response);
9144
- }
9145
- if (!response.ok) {
9146
- let errorBody;
9147
- try {
9148
- errorBody = await response.json();
9149
- } catch {
9150
- errorBody = { error: "Request failed" };
9725
+ for (let staleRetries = 0; ; ) {
9726
+ const response = await this.fetchWithTimeout(url, {
9727
+ method: "GET",
9728
+ headers: { "User-Agent": USER_AGENT }
9729
+ });
9730
+ if (response.status === 402) {
9731
+ try {
9732
+ return await this.handleGetPaymentAndRetryRaw(url, endpoint, params, response, staleRetries > 0);
9733
+ } catch (error) {
9734
+ if (!(error instanceof SafeStaleBlockhashError) || staleRetries >= STALE_BLOCKHASH_RETRY_BACKOFFS_MS.length) throw error;
9735
+ await waitForStaleRetry(staleRetries++);
9736
+ continue;
9737
+ }
9151
9738
  }
9152
- throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9739
+ if (!response.ok) {
9740
+ let errorBody;
9741
+ try {
9742
+ errorBody = await response.json();
9743
+ } catch {
9744
+ errorBody = { error: "Request failed" };
9745
+ }
9746
+ throw new APIError(`API error: ${response.status}`, response.status, sanitizeErrorResponse(errorBody));
9747
+ }
9748
+ return response.json();
9153
9749
  }
9154
- return response.json();
9155
9750
  }
9156
- async handleGetPaymentAndRetryRaw(url, endpoint, params, response) {
9751
+ async handleGetPaymentAndRetryRaw(url, endpoint, params, response, forceFreshBlockhash = false) {
9157
9752
  let paymentHeader = response.headers.get("payment-required");
9158
9753
  if (!paymentHeader) {
9159
9754
  try {
@@ -9195,7 +9790,8 @@ var SolanaLLMClient = class {
9195
9790
  extra: details.extra,
9196
9791
  extensions,
9197
9792
  rpcUrl: this.rpcUrl,
9198
- rpcHeaders: this.rpcHeaders
9793
+ rpcHeaders: this.rpcHeaders,
9794
+ forceFreshBlockhash
9199
9795
  }
9200
9796
  );
9201
9797
  const query = params ? "?" + new URLSearchParams(params).toString() : "";
@@ -9208,6 +9804,9 @@ var SolanaLLMClient = class {
9208
9804
  }
9209
9805
  });
9210
9806
  if (retryResponse.status === 402) {
9807
+ if (await isSafeStaleBlockhashResponse(retryResponse)) {
9808
+ throw new SafeStaleBlockhashError();
9809
+ }
9211
9810
  throw new PaymentError("Payment was rejected. Check your Solana USDC balance.");
9212
9811
  }
9213
9812
  if (!retryResponse.ok) {