forecasting-tools 0.2.87__tar.gz → 0.2.89__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (181) hide show
  1. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/PKG-INFO +6 -6
  2. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/other/data_analyzer.py +4 -6
  3. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/key_factors_researcher.py +1 -1
  4. forecasting_tools-0.2.89/forecasting_tools/ai_models/agent_wrappers.py +75 -0
  5. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/general_llm.py +1 -11
  6. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/prompt_optimizer.py +41 -52
  7. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/cp_benchmarking/benchmarker.py +36 -43
  8. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/numeric_report.py +40 -9
  9. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/questions.py +8 -4
  10. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/forecast_bot.py +11 -15
  11. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/chat_page.py +21 -34
  12. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/pyproject.toml +7 -7
  13. forecasting_tools-0.2.87/forecasting_tools/ai_models/agent_wrappers.py +0 -212
  14. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/LICENSE +0 -0
  15. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/README.md +0 -0
  16. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/__init__.py +0 -0
  17. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/__init__.py +0 -0
  18. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/__init__.py +0 -0
  19. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_member_agent.py +0 -0
  20. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_orchestrator.py +0 -0
  21. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/data_models.py +0 -0
  22. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/member_profiles.py +0 -0
  23. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/tools.py +0 -0
  24. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/base_rate_researcher.py +0 -0
  25. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/deduplicator.py +0 -0
  26. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/estimator.py +0 -0
  27. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/niche_list_researcher.py +0 -0
  28. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/configured_llms.py +0 -0
  29. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/general_researcher.py +0 -0
  30. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/question_generator.py +0 -0
  31. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/question_responder.py +0 -0
  32. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/question_router.py +0 -0
  33. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/research_coordinator.py +0 -0
  34. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/minor_tools.py +0 -0
  35. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/other/hosted_file.py +0 -0
  36. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/generated_question.py +0 -0
  37. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/harmful_question_identifier.py +0 -0
  38. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/q3_q4_quarterly_questions.json +0 -0
  39. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/question_decomposer.py +0 -0
  40. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/question_operationalizer.py +0 -0
  41. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/simple_question.py +0 -0
  42. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/topic_generator.py +0 -0
  43. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/computer_use.py +0 -0
  44. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/find_a_dataset.py +0 -0
  45. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/smart_searcher.py +0 -0
  46. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/__init__.py +0 -0
  47. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/agent_runner.py +0 -0
  48. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/data_models.py +0 -0
  49. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/effect_engine.py +0 -0
  50. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/industrial_basin.json +0 -0
  51. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/milbrook.json +0 -0
  52. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_business.json +0 -0
  53. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_dnd.json +0 -0
  54. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_election.json +0 -0
  55. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/willowbrook.json +0 -0
  56. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/z_broken_mafia.json +0 -0
  57. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/__init__.py +0 -0
  58. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/data_models.py +0 -0
  59. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/forecast_resolver.py +0 -0
  60. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_policy_agent.py +0 -0
  61. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_runner.py +0 -0
  62. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_storage.py +0 -0
  63. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/simulator.py +0 -0
  64. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/situation_generator.py +0 -0
  65. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/__init__.py +0 -0
  66. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/ai_utils/__init__.py +0 -0
  67. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/ai_utils/openai_utils.py +0 -0
  68. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/ai_utils/response_types.py +0 -0
  69. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/README.md +0 -0
  70. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/claude35sonnet.py +0 -0
  71. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/deepseek_r1.py +0 -0
  72. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpt4o.py +0 -0
  73. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpt4ovision.py +0 -0
  74. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpto1.py +0 -0
  75. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpto1preview.py +0 -0
  76. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/metaculus4o.py +0 -0
  77. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/perplexity.py +0 -0
  78. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/exa_searcher.py +0 -0
  79. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/__init__.py +0 -0
  80. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/ai_model.py +0 -0
  81. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/combined_llm_archetype.py +0 -0
  82. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/incurs_cost.py +0 -0
  83. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/named_model.py +0 -0
  84. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/outputs_text.py +0 -0
  85. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/priced_per_request.py +0 -0
  86. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/request_limited_model.py +0 -0
  87. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/retryable_model.py +0 -0
  88. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/time_limited_model.py +0 -0
  89. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/token_limited_model.py +0 -0
  90. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/tokens_are_calculatable.py +0 -0
  91. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/tokens_incur_cost.py +0 -0
  92. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_tracker.py +0 -0
  93. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/__init__.py +0 -0
  94. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/hard_limit_manager.py +0 -0
  95. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/monetary_cost_manager.py +0 -0
  96. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/refreshing_bucket_rate_limiter.py +0 -0
  97. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/bot_evaluator.py +0 -0
  98. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/bot_optimizer.py +0 -0
  99. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/control_prompt.py +0 -0
  100. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/customizable_bot.py +0 -0
  101. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/prompt_data_models.py +0 -0
  102. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/question_plus_research.py +0 -0
  103. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/cp_benchmarking/benchmark_displayer.py +0 -0
  104. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/cp_benchmarking/benchmark_for_bot.py +0 -0
  105. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/__init__.py +0 -0
  106. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/binary_report.py +0 -0
  107. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/coherence_link.py +0 -0
  108. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/conditional_models.py +0 -0
  109. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/conditional_report.py +0 -0
  110. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/data_organizer.py +0 -0
  111. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/forecast_report.py +0 -0
  112. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/markdown_tree.py +0 -0
  113. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/multiple_choice_report.py +0 -0
  114. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/timestamped_predictions.py +0 -0
  115. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/user_response.py +0 -0
  116. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/__init__.py +0 -0
  117. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/bot_lists.py +0 -0
  118. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/__init__.py +0 -0
  119. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q1_veritas_bot.py +0 -0
  120. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q1t_w_convert_to_binary.py +0 -0
  121. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q1t_w_personas_and_exa.py +0 -0
  122. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q2t_w_decomposition.py +0 -0
  123. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q3t_w_asknews.py +0 -0
  124. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q3t_w_exa.py +0 -0
  125. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q3t_w_q4vbinary.py +0 -0
  126. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4_veritas_bot.py +0 -0
  127. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4v_w_exa.py +0 -0
  128. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_dseekr1.py +0 -0
  129. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_o1_preview.py +0 -0
  130. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/main_bot.py +0 -0
  131. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/__init__.py +0 -0
  132. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/gpt_4_1_optimized_bot.py +0 -0
  133. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q1_template_bot.py +0 -0
  134. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q2_template_bot.py +0 -0
  135. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q3_template_bot.py +0 -0
  136. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q4_template_bot.py +0 -0
  137. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/research_only_bot_2025_fall.py +0 -0
  138. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/template_bot_2025_fall.py +0 -0
  139. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/template_bot_2026_spring.py +0 -0
  140. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/uniform_probability_bot.py +0 -0
  141. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/template_bot.py +0 -0
  142. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/Home.py +0 -0
  143. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/base_rate_page.py +0 -0
  144. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/benchmark_page.py +0 -0
  145. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/congress_v2_page.py +0 -0
  146. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/csv_agent.py +0 -0
  147. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/estimator_page.py +0 -0
  148. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/forecaster_page.py +0 -0
  149. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/intervention_leaderboard_page.py +0 -0
  150. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/key_factors_page.py +0 -0
  151. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/niche_list_researcher_page.py +0 -0
  152. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/simulator_page.py +0 -0
  153. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/base_rate_page_examples.json +0 -0
  154. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/congress_page_example.json +0 -0
  155. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/congress_v2_page_example.json +0 -0
  156. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/estimator_page_examples.json +0 -0
  157. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/forecast_page_examples.json +0 -0
  158. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/key_factors_page_examples.json +0 -0
  159. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/niche_list_page_examples.json +0 -0
  160. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/question_generator_page_examples.json +0 -0
  161. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/app_page.py +0 -0
  162. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/custom_auth.py +0 -0
  163. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/report_displayer.py +0 -0
  164. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/tool_page.py +0 -0
  165. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/__init__.py +0 -0
  166. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/asknews_cache.py +0 -0
  167. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/asknews_searcher.py +0 -0
  168. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/forecast_database_manager.py +0 -0
  169. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/metaculus_api.py +0 -0
  170. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/metaculus_client.py +0 -0
  171. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/prediction_extractor.py +0 -0
  172. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/structure_output.py +0 -0
  173. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/works_cited_creator.py +0 -0
  174. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/__init__.py +0 -0
  175. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/async_batching.py +0 -0
  176. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/coda_utils.py +0 -0
  177. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/custom_logger.py +0 -0
  178. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/file_manipulation.py +0 -0
  179. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/jsonable.py +0 -0
  180. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/misc.py +0 -0
  181. {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/stats.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: forecasting-tools
3
- Version: 0.2.87
3
+ Version: 0.2.89
4
4
  Summary: AI forecasting and research tools to help humans reason about and forecast the future
5
5
  License: MIT
6
6
  License-File: LICENSE
@@ -26,19 +26,19 @@ Requires-Dist: asknews (>=0.9.1,<0.14.0)
26
26
  Requires-Dist: asyncio (>=3.0.0,<5.0.0)
27
27
  Requires-Dist: exceptiongroup (>=1.2.2,<2.0.0)
28
28
  Requires-Dist: faker (>=37.0.0,<41.0.0)
29
- Requires-Dist: hyperbrowser (>=0.53.0,<0.84.0)
30
- Requires-Dist: litellm (>=1.59.1,<1.82.3,!=1.76.*,!=1.77.*)
29
+ Requires-Dist: hyperbrowser (>=0.53.0,<1.0.0)
30
+ Requires-Dist: litellm (>=1.59.1,<2.0.0,!=1.76.*,!=1.77.*,!=1.82.7,!=1.82.8)
31
31
  Requires-Dist: nest-asyncio (>=1.5.8,<2.0.0)
32
32
  Requires-Dist: numpy (>=1.26.0,<3.0.0)
33
33
  Requires-Dist: openai (>=1.51.0,<3.0.0)
34
- Requires-Dist: openai-agents[litellm] (>=0.2.0,<0.10.0)
34
+ Requires-Dist: openai-agents[litellm] (>=0.2.0,<0.14.0)
35
35
  Requires-Dist: pandas (>=2.2.3,<4.0.0)
36
36
  Requires-Dist: pendulum (>=3.1.0,<4.0.0)
37
- Requires-Dist: pillow (>=9.0.0,<12.0.0)
37
+ Requires-Dist: pillow (>=9.0.0,<13.0.0)
38
38
  Requires-Dist: plotly (>=5.24.1,<7.0.0)
39
39
  Requires-Dist: pydantic (>=2.9.2,<3.0.0)
40
40
  Requires-Dist: python-dotenv (>=1.0.0,<2.0.0)
41
- Requires-Dist: regex (>=2024.11.6,<2026.0.0)
41
+ Requires-Dist: regex (>=2024.11.6,<2027.0.0)
42
42
  Requires-Dist: requests (>=2.32.3,<3.0.0)
43
43
  Requires-Dist: scikit-learn (>=1.5.2,<2.0.0)
44
44
  Requires-Dist: streamlit (>=1.20.0,<2.0.0)
@@ -12,7 +12,6 @@ from forecasting_tools.ai_models.agent_wrappers import (
12
12
  CodingTool,
13
13
  agent_tool,
14
14
  event_to_tool_message,
15
- general_trace_or_span,
16
15
  )
17
16
  from forecasting_tools.util.misc import clean_indents
18
17
 
@@ -119,10 +118,9 @@ class DataAnalyzer:
119
118
 
120
119
 
121
120
  if __name__ == "__main__":
122
- with general_trace_or_span("Test Span 1"):
123
- answer = asyncio.run(
124
- DataAnalyzer().run_data_analysis(
125
- instructions="Please multiple 52.675 x 6.547 x 9867.5476 x 4356.5",
126
- )
121
+ answer = asyncio.run(
122
+ DataAnalyzer().run_data_analysis(
123
+ instructions="Please multiple 52.675 x 6.547 x 9867.5476 x 4356.5",
127
124
  )
125
+ )
128
126
  print(answer)
@@ -222,7 +222,7 @@ class KeyFactorsResearcher:
222
222
  # Grading Scale for {ScoreCardGrade.__class__.__name__}
223
223
  - {ScoreCardGrade.VERY_BAD.value}: Generally poor quality
224
224
  - {ScoreCardGrade.BAD.value}: Below average quality
225
- - {ScoreCardGrade.OK.value}: Below average quality
225
+ - {ScoreCardGrade.OK.value}: Average quality
226
226
  - {ScoreCardGrade.GOOD.value}: Above average quality
227
227
  - {ScoreCardGrade.VERY_GOOD.value}: Exceptional quality
228
228
 
@@ -0,0 +1,75 @@
1
+ import asyncio
2
+ import logging
3
+
4
+ import nest_asyncio
5
+ from agents import Agent, CodeInterpreterTool, FunctionTool, Runner
6
+ from agents import function_tool as ft
7
+ from agents.extensions.models.litellm_model import LitellmModel
8
+ from agents.stream_events import StreamEvent
9
+
10
+ from forecasting_tools.ai_models.model_tracker import ModelTracker
11
+
12
+ logger = logging.getLogger(__name__)
13
+
14
+
15
+ nest_asyncio.apply()
16
+
17
+
18
+ class AgentSdkLlm(LitellmModel):
19
+ """
20
+ Wrapper around openai-agent-sdk's LiteLlm Model for later extension
21
+ """
22
+
23
+ async def get_response(self, *args, **kwargs): # NOSONAR
24
+ ModelTracker.give_cost_tracking_warning_if_needed(self.model)
25
+ response = await super().get_response(*args, **kwargs)
26
+ await asyncio.sleep(
27
+ 0.0001
28
+ ) # For whatever reason, it seems you need to await a coroutine to get the litellm cost callback to work
29
+ return response
30
+
31
+
32
+ AgentRunner = Runner # Alias for Runner for later extension
33
+ AgentTool = FunctionTool # Alias for FunctionTool for later extension
34
+ AiAgent = Agent # Alias for Agent for later extension
35
+ CodingTool = CodeInterpreterTool # Alias for CodeInterpreterTool for later extension
36
+ agent_tool = ft # Alias for function_tool for later extension
37
+
38
+
39
+ def event_to_tool_message(event: StreamEvent) -> str | None:
40
+ text = ""
41
+ if event.type == "run_item_stream_event":
42
+ item = event.item
43
+ if item.type == "message_output_item":
44
+ content = item.raw_item.content[0]
45
+ if content.type == "output_text":
46
+ # text = content.text
47
+ text = "" # the text is already streamed separate from this function
48
+ elif content.type == "output_refusal":
49
+ text = content.refusal
50
+ else:
51
+ text = "Error: unknown content type"
52
+ elif item.type == "tool_call_item":
53
+ if item.raw_item.type == "code_interpreter_call":
54
+ text = (
55
+ f"\nCode interpreter code:\n```python\n{item.raw_item.code}\n```\n"
56
+ )
57
+ else:
58
+ tool_name = getattr(item.raw_item, "name", "unknown_tool")
59
+ tool_args = getattr(item.raw_item, "arguments", {})
60
+ text = f"Tool call: {tool_name}({tool_args})"
61
+ elif item.type == "tool_call_output_item":
62
+ output = getattr(item, "output", str(item.raw_item))
63
+ text = f"Tool output:\n\n{output}"
64
+ elif item.type == "handoff_call_item":
65
+ handoff_info = getattr(item.raw_item, "name", "handoff")
66
+ text = f"Handoff call: {handoff_info}"
67
+ elif item.type == "handoff_output_item":
68
+ text = f"Handoff output: {str(item.raw_item)}"
69
+ elif item.type == "reasoning_item":
70
+ text = f"Reasoning: {str(item.raw_item)}"
71
+ # elif event.type == "agent_updated_stream_event":
72
+ # text += f"Agent updated: {event.new_agent.name}\n\n"
73
+ if text == "":
74
+ return None
75
+ return text
@@ -21,7 +21,6 @@ from openai.types.responses import (
21
21
  ResponseReasoningItem,
22
22
  )
23
23
 
24
- from forecasting_tools.ai_models.agent_wrappers import track_generation
25
24
  from forecasting_tools.ai_models.ai_utils.openai_utils import (
26
25
  OpenAiUtils,
27
26
  VisionMessageData,
@@ -251,16 +250,7 @@ class GeneralLlm(
251
250
  ) -> Any:
252
251
  logger.debug(f"Invoking model with prompt: {prompt}")
253
252
 
254
- with track_generation(
255
- input=self.model_input_to_message(prompt),
256
- model=self.model,
257
- ) as span:
258
- direct_call_response = await self._mockable_direct_call_to_model(prompt)
259
- answer = direct_call_response.data
260
- span.span_data.output = [{"role": "assistant", "content": answer}]
261
- # span.span_data.usage = usage.model_dump()
262
- span.span_data.model = self.model
263
- span.span_data.model_config = self.litellm_kwargs
253
+ direct_call_response = await self._mockable_direct_call_to_model(prompt)
264
254
 
265
255
  logger.debug(f"Model responded with: {direct_call_response}")
266
256
  return direct_call_response
@@ -9,12 +9,7 @@ from pydantic import BaseModel, Field
9
9
  from forecasting_tools.agents_and_tools.minor_tools import (
10
10
  perplexity_reasoning_pro_search,
11
11
  )
12
- from forecasting_tools.ai_models.agent_wrappers import (
13
- AgentRunner,
14
- AgentSdkLlm,
15
- AiAgent,
16
- general_trace_or_span,
17
- )
12
+ from forecasting_tools.ai_models.agent_wrappers import AgentRunner, AgentSdkLlm, AiAgent
18
13
  from forecasting_tools.auto_optimizers.prompt_data_models import PromptIdea
19
14
  from forecasting_tools.helpers.structure_output import structure_output
20
15
  from forecasting_tools.util.misc import clean_indents, retry_async_function
@@ -112,63 +107,57 @@ class PromptOptimizer:
112
107
  )
113
108
 
114
109
  async def create_optimized_prompt(self) -> OptimizationRun:
115
- with general_trace_or_span("Prompt Optimizer"):
116
- return await self._create_optimized_prompt()
110
+ return await self._create_optimized_prompt()
117
111
 
118
112
  async def _create_optimized_prompt(self) -> OptimizationRun:
119
113
  iteration_num = 0
120
- with general_trace_or_span("Initial Population Generation (Iteration 1)"):
121
- iteration_num += 1
122
- logger.info(
123
- f"Generating initial prompt population of size {self.initial_prompt_population_size}"
124
- )
125
- seed_prompt = ImplementedPrompt(
126
- text=self.initial_prompt,
127
- idea=PromptIdea(
128
- short_name="Initial Seed",
129
- full_text="The user-provided initial prompt",
130
- ),
131
- originating_ideas=[],
132
- )
133
- starting_prompts: list[ImplementedPrompt] = [seed_prompt]
134
- prompts_still_needed = self.initial_prompt_population_size - len(
135
- starting_prompts
136
- )
137
- if prompts_still_needed > 0:
138
- additional_initial_prompts = await self._mutate_prompt(
139
- starting_prompts[0],
140
- prompts_still_needed,
141
- )
142
- starting_prompts.extend(additional_initial_prompts)
143
-
144
- offspring_prompts: list[ImplementedPrompt] = starting_prompts
145
- assert (
146
- seed_prompt in offspring_prompts
147
- ), "Seed prompt not found in offspring prompts"
148
- all_evaluated_prompts: list[ScoredPrompt] = (
149
- await self._evaluate_new_members(offspring_prompts)
114
+ iteration_num += 1
115
+ logger.info(
116
+ f"Generating initial prompt population of size {self.initial_prompt_population_size}"
117
+ )
118
+ seed_prompt = ImplementedPrompt(
119
+ text=self.initial_prompt,
120
+ idea=PromptIdea(
121
+ short_name="Initial Seed",
122
+ full_text="The user-provided initial prompt",
123
+ ),
124
+ originating_ideas=[],
125
+ )
126
+ starting_prompts: list[ImplementedPrompt] = [seed_prompt]
127
+ prompts_still_needed = self.initial_prompt_population_size - len(
128
+ starting_prompts
129
+ )
130
+ if prompts_still_needed > 0:
131
+ additional_initial_prompts = await self._mutate_prompt(
132
+ starting_prompts[0],
133
+ prompts_still_needed,
150
134
  )
151
- survivors = await self._kill_the_weak(all_evaluated_prompts)
135
+ starting_prompts.extend(additional_initial_prompts)
136
+
137
+ offspring_prompts: list[ImplementedPrompt] = starting_prompts
138
+ assert (
139
+ seed_prompt in offspring_prompts
140
+ ), "Seed prompt not found in offspring prompts"
141
+ all_evaluated_prompts: list[ScoredPrompt] = await self._evaluate_new_members(
142
+ offspring_prompts
143
+ )
144
+ survivors = await self._kill_the_weak(all_evaluated_prompts)
152
145
 
153
146
  while iteration_num < self.iterations:
154
147
  iteration_num += 1
155
- with general_trace_or_span(
156
- f"Prompt Optimizer Iteration {iteration_num + 1}",
157
- data={"survivors": [s.model_dump() for s in survivors]},
158
- ):
159
- logger.info(
160
- f"Starting iteration {iteration_num + 1}/{self.iterations} - Current population size: {len(offspring_prompts)}"
161
- )
148
+ logger.info(
149
+ f"Starting iteration {iteration_num + 1}/{self.iterations} - Current population size: {len(offspring_prompts)}"
150
+ )
162
151
 
163
- offspring_prompts = await self._generate_new_prompts(survivors)
152
+ offspring_prompts = await self._generate_new_prompts(survivors)
164
153
 
165
- evaluated_prompts = await self._evaluate_new_members(offspring_prompts)
166
- all_evaluated_prompts.extend(evaluated_prompts)
167
- updated_population = survivors + evaluated_prompts
154
+ evaluated_prompts = await self._evaluate_new_members(offspring_prompts)
155
+ all_evaluated_prompts.extend(evaluated_prompts)
156
+ updated_population = survivors + evaluated_prompts
168
157
 
169
- survivors = await self._kill_the_weak(updated_population)
158
+ survivors = await self._kill_the_weak(updated_population)
170
159
 
171
- self._log_duplicate_prompts(all_evaluated_prompts)
160
+ self._log_duplicate_prompts(all_evaluated_prompts)
172
161
 
173
162
  return OptimizationRun(scored_prompts=all_evaluated_prompts)
174
163
 
@@ -5,7 +5,6 @@ from typing import Sequence
5
5
 
6
6
  import typeguard
7
7
 
8
- from forecasting_tools.ai_models.agent_wrappers import general_trace_or_span
9
8
  from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
10
9
  MonetaryCostManager,
11
10
  )
@@ -80,50 +79,44 @@ class Benchmarker:
80
79
  self.code_to_snapshot = additional_code_to_snapshot
81
80
 
82
81
  async def run_benchmark(self) -> list[BenchmarkForBot]:
83
- with general_trace_or_span("Benchmarker"):
84
- if self.questions_to_use is None:
85
- assert (
86
- self.number_of_questions_to_use is not None
87
- ), "number_of_questions_to_use must be provided if questions_to_use is not provided"
88
- chosen_questions = MetaculusApi.get_benchmark_questions(
89
- self.number_of_questions_to_use,
90
- )
91
- else:
92
- chosen_questions = self.questions_to_use
93
-
94
- chosen_questions = typeguard.check_type(
95
- chosen_questions, list[MetaculusQuestion]
96
- )
97
-
98
- if self.number_of_questions_to_use is not None:
99
- assert len(chosen_questions) == self.number_of_questions_to_use
100
-
101
- benchmarks: list[BenchmarkForBot] = self._initialize_benchmarks(
102
- self.forecast_bots, chosen_questions
82
+ if self.questions_to_use is None:
83
+ assert (
84
+ self.number_of_questions_to_use is not None
85
+ ), "number_of_questions_to_use must be provided if questions_to_use is not provided"
86
+ chosen_questions = MetaculusApi.get_benchmark_questions(
87
+ self.number_of_questions_to_use,
103
88
  )
104
-
105
- batches = self._batch_questions(
106
- self.forecast_bots,
107
- benchmarks,
108
- chosen_questions,
109
- self.concurrent_question_batch_size,
89
+ else:
90
+ chosen_questions = self.questions_to_use
91
+
92
+ chosen_questions = typeguard.check_type(
93
+ chosen_questions, list[MetaculusQuestion]
94
+ )
95
+
96
+ if self.number_of_questions_to_use is not None:
97
+ assert len(chosen_questions) == self.number_of_questions_to_use
98
+
99
+ benchmarks: list[BenchmarkForBot] = self._initialize_benchmarks(
100
+ self.forecast_bots, chosen_questions
101
+ )
102
+
103
+ batches = self._batch_questions(
104
+ self.forecast_bots,
105
+ benchmarks,
106
+ chosen_questions,
107
+ self.concurrent_question_batch_size,
108
+ )
109
+ try:
110
+ for i, batch in enumerate(batches):
111
+ await self._run_a_batch(batch)
112
+ if batch.is_last_batch_for_benchmark:
113
+ self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
114
+ except KeyboardInterrupt:
115
+ logger.warning(
116
+ "KeyboardInterrupt detected, saving current benchmark progress."
110
117
  )
111
- try:
112
- for i, batch in enumerate(batches):
113
- with general_trace_or_span(
114
- f"{batch.benchmark.name} - Batch {i+1} of {len(batches)}"
115
- ):
116
- await self._run_a_batch(batch)
117
- if batch.is_last_batch_for_benchmark:
118
- self._append_benchmarks_to_jsonl_if_configured(
119
- [batch.benchmark]
120
- )
121
- except KeyboardInterrupt:
122
- logger.warning(
123
- "KeyboardInterrupt detected, saving current benchmark progress."
124
- )
125
- self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
126
- raise
118
+ self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
119
+ raise
127
120
  return benchmarks
128
121
 
129
122
  async def _run_a_batch(self, batch: QuestionBatch) -> None:
@@ -84,8 +84,11 @@ class DatePercentile(BaseModel):
84
84
 
85
85
  class NumericDistribution(BaseModel):
86
86
  declared_percentiles: list[Percentile]
87
- open_upper_bound: bool
88
- open_lower_bound: bool
87
+ open_upper_bound: bool # If False, cdf[-1] must be 1.0 (no mass above upper bound)
88
+ open_lower_bound: bool # If False, cdf[0] must be 0.0 (no mass strictly below lower
89
+ # bound). Note: cdf[0] = P(outcome < lower_bound), so a closed lower bound does not
90
+ # prevent assigning probability to the outcome equalling lower_bound exactly —
91
+ # that mass goes in the first inbound bucket: cdf[1] - cdf[0].
89
92
  upper_bound: float
90
93
  lower_bound: float
91
94
  zero_point: float | None
@@ -314,6 +317,22 @@ class NumericDistribution(BaseModel):
314
317
  ]
315
318
  return representative_percentiles
316
319
 
320
+ def get_percentiles_at_target_heights(
321
+ self, target_heights: list[float] | None = None
322
+ ) -> list[Percentile]:
323
+ if target_heights is None:
324
+ target_heights = [0.1, 0.2, 0.4, 0.6, 0.8, 0.9]
325
+
326
+ heights = [p.percentile for p in self.declared_percentiles]
327
+ values = [p.value for p in self.declared_percentiles]
328
+
329
+ result = []
330
+ for target in target_heights:
331
+ interpolated_value = float(np.interp(target, heights, values))
332
+ result.append(Percentile(percentile=target, value=interpolated_value))
333
+
334
+ return result
335
+
317
336
  @property
318
337
  @typing_extensions.deprecated(
319
338
  "NumericDistribution.cdf (property) will be replaced with NumericDistribution.get_cdf (method). Please switch.",
@@ -328,12 +347,22 @@ class NumericDistribution(BaseModel):
328
347
  between upper and lower bound (taking into account probability assigned above and below the bounds)
329
348
  that is compatible with Metaculus questions.
330
349
 
331
- cdf stands for 'continuous distribution function'
350
+ cdf stands for 'cumulative distribution function'
332
351
 
333
352
  At Metaculus CDFs are often represented with 201 points. Each point has:
334
- - percentile ("X% of values are below this point". This is the y axis of the cdf graph)
353
+ - percentile (the y axis of the cdf graph — see boundary notes below)
335
354
  - 'value' or 'nominal location' (The real world number that answers the question)
336
355
  - cdf location (a number between 0 and 1 representing where the point is on the cdf x axis, where 0 is range min, and 1 is range max)
356
+
357
+ Important boundary semantics (note the asymmetry):
358
+ - cdf[0].percentile = P(outcome < lower_bound) — strictly less than (not equal to)
359
+ - cdf[-1].percentile = P(outcome <= upper_bound) — less than or equal to
360
+ For questions with a closed lower bound, cdf[0].percentile must be 0.0 because
361
+ outcomes strictly below the lower bound are impossible. This does NOT mean zero
362
+ probability at the lower bound itself — probability of the outcome equalling
363
+ lower_bound belongs in the first inbound bucket: cdf[1].percentile - cdf[0].percentile.
364
+ For example, to express an 80% chance the outcome is exactly lower_bound, set
365
+ cdf[1].percentile = 0.8 (and cdf[0].percentile = 0.0).
337
366
  """
338
367
 
339
368
  cdf_size = self.cdf_size or NumericDefaults.DEFAULT_CDF_SIZE
@@ -511,6 +540,11 @@ class NumericDistribution(BaseModel):
511
540
  - caps the maximum growth to 0.2
512
541
 
513
542
  Note, thresholds change with different `inbound_outcome_count`s
543
+
544
+ Boundary convention: cdf[0] is P(outcome < lower_bound) — strictly less than.
545
+ For closed lower bounds this is forced to 0.0 after standardization; probability
546
+ of landing exactly on lower_bound belongs in the first inbound bucket (cdf[1]).
547
+ cdf[-1] is P(outcome <= upper_bound) — less than or equal to (standard convention).
514
548
  """
515
549
 
516
550
  lower_open = self.open_lower_bound
@@ -633,12 +667,9 @@ class NumericReport(ForecastReport):
633
667
  def make_readable_prediction(cls, prediction: NumericDistribution) -> str:
634
668
  num_percentiles = len(prediction.declared_percentiles)
635
669
  if num_percentiles > 10:
636
- num_display_percentiles = 5
670
+ representative_percentiles = prediction.get_percentiles_at_target_heights()
637
671
  else:
638
- num_display_percentiles = num_percentiles
639
- representative_percentiles = prediction.get_representative_percentiles(
640
- num_display_percentiles
641
- )
672
+ representative_percentiles = prediction.declared_percentiles
642
673
  readable = "Probability distribution:\n"
643
674
  for percentile in representative_percentiles:
644
675
  if prediction.is_date:
@@ -474,8 +474,10 @@ class DateQuestion(MetaculusQuestion, BoundedQuestionMixin):
474
474
  question_type: Literal["date"] = "date"
475
475
  upper_bound: datetime
476
476
  lower_bound: datetime
477
- open_upper_bound: bool
478
- open_lower_bound: bool
477
+ open_upper_bound: bool # If False, cdf[-1] must be 1.0 (no mass above upper bound)
478
+ open_lower_bound: bool # If False, cdf[0] must be 0.0. Note: cdf[0] = P(outcome <
479
+ # lower_bound) strictly, so probability of landing exactly on lower_bound belongs
480
+ # in the first inbound bucket (cdf[1]), not in cdf[0].
479
481
  zero_point: float | None = None
480
482
  cdf_size: int = 201
481
483
 
@@ -530,8 +532,10 @@ class NumericQuestion(MetaculusQuestion, BoundedQuestionMixin):
530
532
  question_type: Literal["numeric"] = "numeric"
531
533
  upper_bound: float
532
534
  lower_bound: float
533
- open_upper_bound: bool
534
- open_lower_bound: bool
535
+ open_upper_bound: bool # If False, cdf[-1] must be 1.0 (no mass above upper bound)
536
+ open_lower_bound: bool # If False, cdf[0] must be 0.0. Note: cdf[0] = P(outcome <
537
+ # lower_bound) strictly, so probability of landing exactly on lower_bound belongs
538
+ # in the first inbound bucket (cdf[1]), not in cdf[0].
535
539
  zero_point: float | None = None
536
540
  cdf_size: int = (
537
541
  201 # Normal numeric questions have 201 points, but discrete questions have fewer
@@ -12,7 +12,6 @@ from typing import Any, Coroutine, Literal, Sequence, TypeVar, cast, overload
12
12
  from exceptiongroup import ExceptionGroup
13
13
  from pydantic import BaseModel
14
14
 
15
- from forecasting_tools.ai_models.agent_wrappers import general_trace_or_span
16
15
  from forecasting_tools.ai_models.general_llm import GeneralLlm
17
16
  from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
18
17
  MonetaryCostManager,
@@ -338,20 +337,17 @@ class ForecastBot(ABC):
338
337
  async def _run_individual_question_with_error_propagation(
339
338
  self, question: MetaculusQuestion
340
339
  ) -> ForecastReport:
341
- with general_trace_or_span(
342
- f"{self.__class__.__name__} - Question: {question.page_url}"
343
- ):
344
- try:
345
- return await self._run_individual_question(question)
346
- except Exception as e:
347
- error_message = (
348
- f"Error while processing question url: '{question.page_url}'"
349
- )
350
- logger.error(f"{error_message}: {e}")
351
- self._reraise_exception_with_prepended_message(e, error_message)
352
- assert (
353
- False
354
- ), "This is to satisfy type checker. The previous function should raise an exception"
340
+ try:
341
+ return await self._run_individual_question(question)
342
+ except Exception as e:
343
+ error_message = (
344
+ f"Error while processing question url: '{question.page_url}'"
345
+ )
346
+ logger.error(f"{error_message}: {e}")
347
+ self._reraise_exception_with_prepended_message(e, error_message)
348
+ assert (
349
+ False
350
+ ), "This is to satisfy type checker. The previous function should raise an exception"
355
351
 
356
352
  async def _run_individual_question(
357
353
  self, question: MetaculusQuestion
@@ -43,7 +43,6 @@ from forecasting_tools.ai_models.agent_wrappers import (
43
43
  AgentTool,
44
44
  AiAgent,
45
45
  event_to_tool_message,
46
- general_trace_or_span,
47
46
  )
48
47
  from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
49
48
  MonetaryCostManager,
@@ -84,7 +83,6 @@ class ChatSession(BaseModel, Jsonable):
84
83
  name: str
85
84
  messages: list[dict]
86
85
  model_choice: str = DEFAULT_MODEL
87
- trace_id: str | None = None
88
86
  last_chat_cost: float | None = None
89
87
  last_chat_duration: float | None = None
90
88
  time_stamp: datetime = Field(default_factory=datetime.now)
@@ -235,11 +233,6 @@ class ChatPage(AppPage):
235
233
  st.markdown(
236
234
  f"**Last Chat Duration:** {st.session_state.last_chat_duration:.2f} seconds"
237
235
  )
238
- if "trace_id" in st.session_state.keys():
239
- trace_id = st.session_state.trace_id
240
- st.markdown(
241
- f"**Conversation in Foresight Project:** [link](https://platform.openai.com/traces/trace?trace_id={trace_id})"
242
- )
243
236
 
244
237
  @classmethod
245
238
  def display_premade_examples(cls) -> None:
@@ -265,8 +258,6 @@ class ChatPage(AppPage):
265
258
  if st.button(session.name, key=session.name):
266
259
  st.session_state.messages = session.messages
267
260
  st.session_state.model_choice = session.model_choice
268
- if session.trace_id:
269
- st.session_state.trace_id = session.trace_id
270
261
  if session.last_chat_cost:
271
262
  st.session_state.last_chat_cost = session.last_chat_cost
272
263
  if session.last_chat_duration:
@@ -299,7 +290,6 @@ class ChatPage(AppPage):
299
290
  name=st.session_state["chat_save_name"],
300
291
  model_choice=st.session_state["model_choice"],
301
292
  messages=st.session_state.messages,
302
- trace_id=st.session_state.trace_id,
303
293
  last_chat_cost=st.session_state.last_chat_cost,
304
294
  last_chat_duration=st.session_state.last_chat_duration,
305
295
  )
@@ -464,29 +454,27 @@ class ChatPage(AppPage):
464
454
  handoffs=[],
465
455
  )
466
456
 
467
- with general_trace_or_span("Chat App") as chat_trace:
468
- result = AgentRunner.run_streamed(
469
- agent, st.session_state.messages, max_turns=20
470
- )
471
- streamed_text = ""
472
- with st.chat_message("assistant"):
473
- placeholder = st.empty()
474
- with st.spinner("Thinking..."):
475
- async for event in result.stream_events():
476
- if event.type == "raw_response_event" and isinstance(
477
- event.data, ResponseTextDeltaEvent
478
- ):
479
- streamed_text += event.data.delta
480
- placeholder.write(streamed_text)
481
-
482
- new_reasoning = event_to_tool_message(event)
483
- if new_reasoning:
484
- st.sidebar.write(new_reasoning)
485
-
486
- # logger.info(f"Chat finished with output: {streamed_text}")
487
- st.session_state.messages = result.to_input_list()
488
- st.session_state.trace_id = chat_trace.trace_id
489
- cls._update_last_message_if_gemini_bug(model_choice)
457
+ result = AgentRunner.run_streamed(
458
+ agent, st.session_state.messages, max_turns=20
459
+ )
460
+ streamed_text = ""
461
+ with st.chat_message("assistant"):
462
+ placeholder = st.empty()
463
+ with st.spinner("Thinking..."):
464
+ async for event in result.stream_events():
465
+ if event.type == "raw_response_event" and isinstance(
466
+ event.data, ResponseTextDeltaEvent
467
+ ):
468
+ streamed_text += event.data.delta
469
+ placeholder.write(streamed_text)
470
+
471
+ new_reasoning = event_to_tool_message(event)
472
+ if new_reasoning:
473
+ st.sidebar.write(new_reasoning)
474
+
475
+ # logger.info(f"Chat finished with output: {streamed_text}")
476
+ st.session_state.messages = result.to_input_list()
477
+ cls._update_last_message_if_gemini_bug(model_choice)
490
478
 
491
479
  ForecastDatabaseManager.add_general_report_to_database(
492
480
  question_text=prompt_input,
@@ -519,7 +507,6 @@ class ChatPage(AppPage):
519
507
  @classmethod
520
508
  def clear_chat_history(cls) -> None:
521
509
  st.session_state.messages = [cls.DEFAULT_MESSAGE]
522
- st.session_state.trace_id = None
523
510
  st.session_state.chat_files = []
524
511
 
525
512