forecasting-tools 0.2.89__tar.gz → 0.2.91__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/PKG-INFO +56 -369
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/README.md +53 -366
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/__init__.py +3 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/exa_searcher.py +2 -1
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/general_llm.py +13 -22
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/.gitignore +2 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/__init__.py +18 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/_training_data.py +123 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/benchmark.ipynb +1352 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/calibration_adjuster.py +209 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/constant_shift_adjuster.py +63 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/decision_tree_adjuster.py +227 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/k_means_adjuster.py +266 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/logistic_recalibration_adjuster.py +132 -0
- forecasting_tools-0.2.91/forecasting_tools/calibration_adjustment/step_adjuster.py +183 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/markdown_tree.py +3 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/bot_lists.py +4 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/main_bot.py +6 -1
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/gpt_4_1_optimized_bot.py +0 -1
- forecasting_tools-0.2.91/forecasting_tools/forecast_bots/official_bots/template_bot_2026_summer.py +736 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/template_bot.py +3 -3
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/congress_v2_page.py +0 -1
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/metaculus_api.py +24 -28
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/metaculus_client.py +13 -8
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/misc.py +16 -2
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/pyproject.toml +3 -3
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/LICENSE +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/ai_congress_v2/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_member_agent.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_orchestrator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/ai_congress_v2/data_models.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/ai_congress_v2/member_profiles.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/ai_congress_v2/tools.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/base_rates/base_rate_researcher.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/base_rates/deduplicator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/base_rates/estimator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/base_rates/niche_list_researcher.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/deprecated/configured_llms.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/deprecated/general_researcher.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/deprecated/question_generator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/deprecated/question_responder.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/deprecated/question_router.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/deprecated/research_coordinator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/minor_tools.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/other/data_analyzer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/other/hosted_file.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/generated_question.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/harmful_question_identifier.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/q3_q4_quarterly_questions.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/question_decomposer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/question_operationalizer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/simple_question.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/question_generators/topic_generator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/research/computer_use.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/research/find_a_dataset.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/research/key_factors_researcher.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/research/smart_searcher.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/agent_runner.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/data_models.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/effect_engine.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/industrial_basin.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/milbrook.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_business.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_dnd.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_election.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/willowbrook.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/z_broken_mafia.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/data_models.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/forecast_resolver.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_policy_agent.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_runner.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_storage.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/simulator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/agents_and_tools/situation_simulator/situation_generator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/agent_wrappers.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/ai_utils/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/ai_utils/openai_utils.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/ai_utils/response_types.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/README.md +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/claude35sonnet.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/deepseek_r1.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/gpt4o.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/gpt4ovision.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/gpto1.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/gpto1preview.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/metaculus4o.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/deprecated_model_classes/perplexity.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/ai_model.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/combined_llm_archetype.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/incurs_cost.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/named_model.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/outputs_text.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/priced_per_request.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/request_limited_model.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/retryable_model.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/time_limited_model.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/token_limited_model.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/tokens_are_calculatable.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_interfaces/tokens_incur_cost.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/model_tracker.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/resource_managers/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/resource_managers/hard_limit_manager.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/resource_managers/monetary_cost_manager.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/ai_models/resource_managers/refreshing_bucket_rate_limiter.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/bot_evaluator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/bot_optimizer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/control_prompt.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/customizable_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/prompt_data_models.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/prompt_optimizer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/auto_optimizers/question_plus_research.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/cp_benchmarking/benchmark_displayer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/cp_benchmarking/benchmark_for_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/cp_benchmarking/benchmarker.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/binary_report.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/coherence_link.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/conditional_models.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/conditional_report.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/data_organizer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/forecast_report.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/multiple_choice_report.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/numeric_report.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/questions.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/timestamped_predictions.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/data_models/user_response.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q1_veritas_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q1t_w_convert_to_binary.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q1t_w_personas_and_exa.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q2t_w_decomposition.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q3t_w_asknews.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q3t_w_exa.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q3t_w_q4vbinary.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q4_veritas_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q4v_w_exa.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_dseekr1.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_o1_preview.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/forecast_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/q1_template_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/q2_template_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/q3_template_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/q4_template_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/research_only_bot_2025_fall.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/template_bot_2025_fall.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/template_bot_2026_spring.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/forecast_bots/official_bots/uniform_probability_bot.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/Home.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/base_rate_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/benchmark_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/chat_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/csv_agent.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/estimator_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/forecaster_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/intervention_leaderboard_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/key_factors_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/niche_list_researcher_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/app_pages/simulator_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/base_rate_page_examples.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/congress_page_example.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/congress_v2_page_example.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/estimator_page_examples.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/forecast_page_examples.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/key_factors_page_examples.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/niche_list_page_examples.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/example_outputs/question_generator_page_examples.json +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/helpers/app_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/helpers/custom_auth.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/helpers/report_displayer.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/front_end/helpers/tool_page.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/asknews_cache.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/asknews_searcher.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/forecast_database_manager.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/prediction_extractor.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/structure_output.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/helpers/works_cited_creator.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/__init__.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/async_batching.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/coda_utils.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/custom_logger.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/file_manipulation.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/jsonable.py +0 -0
- {forecasting_tools-0.2.89 → forecasting_tools-0.2.91}/forecasting_tools/util/stats.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: forecasting-tools
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.91
|
|
4
4
|
Summary: AI forecasting and research tools to help humans reason about and forecast the future
|
|
5
5
|
License: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -31,7 +31,7 @@ Requires-Dist: litellm (>=1.59.1,<2.0.0,!=1.76.*,!=1.77.*,!=1.82.7,!=1.82.8)
|
|
|
31
31
|
Requires-Dist: nest-asyncio (>=1.5.8,<2.0.0)
|
|
32
32
|
Requires-Dist: numpy (>=1.26.0,<3.0.0)
|
|
33
33
|
Requires-Dist: openai (>=1.51.0,<3.0.0)
|
|
34
|
-
Requires-Dist: openai-agents[litellm] (>=0.2.0,<0.
|
|
34
|
+
Requires-Dist: openai-agents[litellm] (>=0.2.0,<0.20.0)
|
|
35
35
|
Requires-Dist: pandas (>=2.2.3,<4.0.0)
|
|
36
36
|
Requires-Dist: pendulum (>=3.1.0,<4.0.0)
|
|
37
37
|
Requires-Dist: pillow (>=9.0.0,<13.0.0)
|
|
@@ -43,7 +43,7 @@ Requires-Dist: requests (>=2.32.3,<3.0.0)
|
|
|
43
43
|
Requires-Dist: scikit-learn (>=1.5.2,<2.0.0)
|
|
44
44
|
Requires-Dist: streamlit (>=1.20.0,<2.0.0)
|
|
45
45
|
Requires-Dist: tenacity (>=8.0.0,<10.0.0)
|
|
46
|
-
Requires-Dist: tiktoken (>=0.8.0,<0.
|
|
46
|
+
Requires-Dist: tiktoken (>=0.8.0,<0.20.0)
|
|
47
47
|
Requires-Dist: typeguard (>=4.3.0,<5.0.0)
|
|
48
48
|
Requires-Dist: unidecode (>=1.4.0,<2.0.0)
|
|
49
49
|
Project-URL: Repository, https://github.com/Metaculus/forecasting-tools
|
|
@@ -71,18 +71,11 @@ This repository contains forecasting and research tools built with Python and St
|
|
|
71
71
|
Here are the tools most likely to be useful to you:
|
|
72
72
|
- 🎯 **Forecasting Bot:** General forecaster that integrates with the Metaculus AI benchmarking competition and provides a number of utilities. You can forecast with a pre-existing bot or override the class to customize your own (without redoing all the aggregation/API code, etc)
|
|
73
73
|
- 🔌 **Metaculus API Wrapper:** for interacting with questions and tournaments
|
|
74
|
-
- 📊 **Benchmarking:** Randomly sample quality questions from Metaculus and run your bot against them so you can get an early sense of how your bot is doing by comparing to the community prediction and expected baseline scores.
|
|
75
74
|
- 🤖 **In-House Metaculus Bots**: You can see all the bots that Metaculus is running on their site in `run_bots.py`
|
|
76
75
|
|
|
77
76
|
Here are some other features of the project (not all are documented yet):
|
|
78
|
-
- **
|
|
79
|
-
- **Key Factor Analysis:** Key Factors Analysis for scoring, ranking, and prioritizing important variables in forecasting questions
|
|
80
|
-
- **Base Rate Researcher:** for calculating event probabilities (still experimental)
|
|
81
|
-
- **Niche List Researcher:** for analyzing very specific lists of past events or items (still experimental)
|
|
82
|
-
- **Fermi Estimator:** for breaking down numerical estimates (still experimental)
|
|
77
|
+
- **General LLM Wrapper:** A unified interface around litellm with retry logic, the Metaculus proxy, structured outputs, and cost tracking
|
|
83
78
|
- **Monetary Cost Manager:** for tracking AI and API expenses
|
|
84
|
-
- **Prompt Optimizer:** for letting AI iterate through 100+ forecasting bot prompts
|
|
85
|
-
- **Question Decomposer/Operationalizer:** To turn a question or topic into relevant forecastable sub-questions
|
|
86
79
|
- **Other experimental tools:** See the demo site for other AI forecasting tools that this project supports (not all are documented). Also see the `scripts` folder for other common workflows and entry points into the code.
|
|
87
80
|
|
|
88
81
|
All the examples below are in a Jupyter Notebook called `README.ipynb` which you can run locally to test the package (make sure to run the first cell though).
|
|
@@ -108,9 +101,8 @@ They both have roughly the same parameters. See below on how to use the Template
|
|
|
108
101
|
|
|
109
102
|
|
|
110
103
|
```python
|
|
111
|
-
from forecasting_tools import TemplateBot,
|
|
104
|
+
from forecasting_tools import TemplateBot, MetaculusClient, GeneralLlm
|
|
112
105
|
|
|
113
|
-
# Initialize the bot
|
|
114
106
|
bot = TemplateBot(
|
|
115
107
|
research_reports_per_question=3, # Number of separate research attempts per question
|
|
116
108
|
predictions_per_research_report=5, # Number of predictions to make per research report
|
|
@@ -123,10 +115,10 @@ bot = TemplateBot(
|
|
|
123
115
|
}
|
|
124
116
|
)
|
|
125
117
|
|
|
126
|
-
TOURNAMENT_ID =
|
|
118
|
+
TOURNAMENT_ID = MetaculusClient.CURRENT_METACULUS_CUP_ID
|
|
127
119
|
reports = await bot.forecast_on_tournament(TOURNAMENT_ID)
|
|
128
120
|
|
|
129
|
-
#
|
|
121
|
+
# If the tournament is not active, no reports will be returned
|
|
130
122
|
for report in reports:
|
|
131
123
|
print(f"\nQuestion: {report.question.question_text}")
|
|
132
124
|
print(f"Prediction: {report.prediction}")
|
|
@@ -151,19 +143,18 @@ for report in reports:
|
|
|
151
143
|
from forecasting_tools import (
|
|
152
144
|
TemplateBot,
|
|
153
145
|
BinaryQuestion,
|
|
154
|
-
|
|
155
|
-
DataOrganizer
|
|
146
|
+
MetaculusClient,
|
|
147
|
+
DataOrganizer,
|
|
156
148
|
)
|
|
157
149
|
|
|
158
|
-
# Initialize the bot
|
|
159
150
|
bot = TemplateBot(
|
|
160
151
|
research_reports_per_question=3,
|
|
161
152
|
predictions_per_research_report=5,
|
|
162
153
|
publish_reports_to_metaculus=False,
|
|
163
154
|
)
|
|
164
155
|
|
|
165
|
-
|
|
166
|
-
question1 =
|
|
156
|
+
metaculus_client = MetaculusClient()
|
|
157
|
+
question1 = metaculus_client.get_question_by_url(
|
|
167
158
|
"https://www.metaculus.com/questions/578/human-extinction-by-2100/"
|
|
168
159
|
)
|
|
169
160
|
question2 = BinaryQuestion(
|
|
@@ -175,17 +166,14 @@ question2 = BinaryQuestion(
|
|
|
175
166
|
|
|
176
167
|
reports = await bot.forecast_questions([question1, question2])
|
|
177
168
|
|
|
178
|
-
|
|
179
|
-
# Print results
|
|
180
169
|
for report in reports:
|
|
181
170
|
print(f"Question: {report.question.question_text}")
|
|
182
171
|
print(f"Prediction: {report.prediction}")
|
|
183
172
|
shortened_explanation = report.explanation.replace('\n', ' ')[:100]
|
|
184
173
|
print(f"Reasoning: {shortened_explanation}...")
|
|
185
174
|
|
|
186
|
-
# You can also save and load questions and reports
|
|
187
175
|
file_path = "temp/reports.json"
|
|
188
|
-
DataOrganizer.save_reports_to_file_path(reports, file_path) #
|
|
176
|
+
DataOrganizer.save_reports_to_file_path(reports, file_path) # Overwrites the file if it already exists
|
|
189
177
|
loaded_reports = DataOrganizer.load_reports_from_file_path(file_path)
|
|
190
178
|
```
|
|
191
179
|
|
|
@@ -208,7 +196,7 @@ Note: You'll need to have your environment variables set up (see the section bel
|
|
|
208
196
|
|
|
209
197
|
## Customizing the Bot
|
|
210
198
|
### General Customization
|
|
211
|
-
Generally all you have to do to make your own bot is inherit from the TemplateBot and override any combination of the 3 forecasting methods and the 1 research method. This saves you the headache of interacting with the Metaculus API, implementing aggregation of predictions, creating benchmarking interfaces, etc. Below is an example. It may also be helpful to look at the TemplateBot code (forecasting_tools/
|
|
199
|
+
Generally all you have to do to make your own bot is inherit from the TemplateBot and override any combination of the 3 forecasting methods and the 1 research method. This saves you the headache of interacting with the Metaculus API, implementing aggregation of predictions, creating benchmarking interfaces, etc. Below is an example. It may also be helpful to look at the `TemplateBot` code (`forecasting_tools/forecast_bots/template_bot.py`) for a more complete example.
|
|
212
200
|
|
|
213
201
|
|
|
214
202
|
```python
|
|
@@ -222,9 +210,9 @@ from forecasting_tools import (
|
|
|
222
210
|
PredictedOptionList,
|
|
223
211
|
NumericDistribution,
|
|
224
212
|
SmartSearcher,
|
|
225
|
-
|
|
213
|
+
MetaculusClient,
|
|
226
214
|
GeneralLlm,
|
|
227
|
-
PredictionExtractor
|
|
215
|
+
PredictionExtractor,
|
|
228
216
|
)
|
|
229
217
|
from forecasting_tools.util.misc import clean_indents
|
|
230
218
|
|
|
@@ -277,7 +265,7 @@ class MyCustomBot(TemplateBot):
|
|
|
277
265
|
...
|
|
278
266
|
|
|
279
267
|
custom_bot = MyCustomBot()
|
|
280
|
-
question =
|
|
268
|
+
question = MetaculusClient().get_question_by_url(
|
|
281
269
|
"https://www.metaculus.com/questions/578/human-extinction-by-2100/"
|
|
282
270
|
)
|
|
283
271
|
report = await custom_bot.forecast_question(question)
|
|
@@ -322,9 +310,9 @@ class NotepadBot(TemplateBot):
|
|
|
322
310
|
notepad = await self._get_notepad(question)
|
|
323
311
|
|
|
324
312
|
if notepad.total_predictions_attempted % 2 == 0:
|
|
325
|
-
model = "
|
|
313
|
+
model = "openrouter/openai/gpt-4o"
|
|
326
314
|
else:
|
|
327
|
-
model = "
|
|
315
|
+
model = "anthropic/claude-3-5-sonnet-20240620"
|
|
328
316
|
|
|
329
317
|
personality = notepad.note_entries["personality"]
|
|
330
318
|
prompt = f"You are a {personality}. Forecast this question: {question.question_text}. The last thing you write is your final answer as: 'Probability: ZZ%', 0-100"
|
|
@@ -345,127 +333,29 @@ Whether running locally or through Github actions, you will need to set environm
|
|
|
345
333
|
|
|
346
334
|
# Important Utilities
|
|
347
335
|
|
|
348
|
-
##
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
```python
|
|
353
|
-
from forecasting_tools import Benchmarker, TemplateBot, BenchmarkForBot
|
|
354
|
-
|
|
355
|
-
class CustomBot(TemplateBot):
|
|
356
|
-
...
|
|
357
|
-
|
|
358
|
-
# Run benchmark on multiple bots
|
|
359
|
-
bots = [TemplateBot(), CustomBot()] # Add your custom bots here
|
|
360
|
-
benchmarker = Benchmarker(
|
|
361
|
-
forecast_bots=bots,
|
|
362
|
-
number_of_questions_to_use=2, # Recommended 100+ for meaningful results
|
|
363
|
-
file_path_to_save_reports="benchmarks/",
|
|
364
|
-
# It will create a file name for you if given a folder.
|
|
365
|
-
# If a file name is given, and the file already exists, it will overwrite it.
|
|
366
|
-
concurrent_question_batch_size=5,
|
|
367
|
-
)
|
|
368
|
-
benchmarks: list[BenchmarkForBot] = await benchmarker.run_benchmark()
|
|
369
|
-
|
|
370
|
-
# View results
|
|
371
|
-
for benchmark in benchmarks[:2]:
|
|
372
|
-
print("--------------------------------")
|
|
373
|
-
print(f"Bot: {benchmark.name}")
|
|
374
|
-
print(f"Score: {benchmark.average_expected_baseline_score}") # Higher is better
|
|
375
|
-
print(f"Num reports in benchmark: {len(benchmark.forecast_reports)}")
|
|
376
|
-
print(f"Time: {benchmark.time_taken_in_minutes}min")
|
|
377
|
-
print(f"Cost: ${benchmark.total_cost}")
|
|
378
|
-
```
|
|
379
|
-
|
|
380
|
-
--------------------------------
|
|
381
|
-
Bot: TemplateBot
|
|
382
|
-
Score: 53.24105782939477
|
|
383
|
-
Num reports in benchmark: 2
|
|
384
|
-
Time: 0.23375582297643024min
|
|
385
|
-
Cost: $0.03020605
|
|
386
|
-
--------------------------------
|
|
387
|
-
Bot: CustomBot
|
|
388
|
-
Score: 53.24105782939476
|
|
389
|
-
Num reports in benchmark: 2
|
|
390
|
-
Time: 0.20734789768854778min
|
|
391
|
-
Cost: $0.019155650000000003
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
The ideal number of questions to get a good sense of whether one bot is better than another can vary. 100+ should tell your something decent. See [this analysis](https://forum.effectivealtruism.org/posts/DzqSh7akX28JEHf9H/comparing-two-forecasters-in-an-ideal-world) for exploration of the numbers. With too few questions, the results could just be statistical noise, though how many questions you need depends highly on the difference in skill of your bot versions.
|
|
395
|
-
|
|
396
|
-
If you use the average expected baseline score, higher score is better. The scoring measures the expected value of your score without needing an actual resolution by assuming that the community prediction is the 'true probability'. Under this assumption, expected baseline scores are a proper score (see analysis in `scripts/simulate_a_tournament.ipynb`)
|
|
397
|
-
|
|
398
|
-
As of May 29, 2025 the benchmarker automatically selects a random set of questions from Metaculus that:
|
|
399
|
-
- Are binary questions (yes/no)
|
|
400
|
-
- Are currently open
|
|
401
|
-
- Opened within the last year
|
|
402
|
-
- Have at least 30 forecasters
|
|
403
|
-
- Have a community prediction
|
|
404
|
-
- Are not part of a group question
|
|
405
|
-
|
|
406
|
-
Note that sometimes there are not many questions matching these filters (e.g. at the beginning of a new year when a majority of open questions were just resolved). As of last edit there are plans to expand this to numeric and multiple choice, but right now it just benchmarks binary questions.
|
|
407
|
-
|
|
408
|
-
You can grab these questions without using the Benchmarker by running the below
|
|
409
|
-
|
|
336
|
+
## Metaculus Client
|
|
337
|
+
The `MetaculusClient` wraps the Metaculus API for interacting with questions and tournaments. Grabbing questions returns a pydantic object that supports the important fields for Binary, Multiple Choice, Numeric, and Date questions. Instantiate the client once and reuse it (it picks up `METACULUS_TOKEN` from the environment by default).
|
|
410
338
|
|
|
411
339
|
|
|
412
340
|
```python
|
|
413
|
-
from forecasting_tools import
|
|
414
|
-
|
|
415
|
-
questions = MetaculusApi.get_benchmark_questions(
|
|
416
|
-
num_of_questions_to_return=100,
|
|
417
|
-
)
|
|
418
|
-
```
|
|
419
|
-
|
|
420
|
-
You can also save/load benchmarks to/from json
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
```python
|
|
424
|
-
from forecasting_tools import BenchmarkForBot
|
|
425
|
-
|
|
426
|
-
# Load
|
|
427
|
-
file_path = "benchmarks/benchmark.json"
|
|
428
|
-
benchmarks: list[BenchmarkForBot] = BenchmarkForBot.load_json_from_file_path(file_path)
|
|
429
|
-
|
|
430
|
-
# Save
|
|
431
|
-
new_benchmarks: list[BenchmarkForBot] = benchmarks
|
|
432
|
-
BenchmarkForBot.save_object_list_to_file_path(new_benchmarks, file_path) # Will overwrite the file if it already exists
|
|
433
|
-
|
|
434
|
-
# To/From Json String
|
|
435
|
-
single_benchmark = benchmarks[0]
|
|
436
|
-
json_object: dict = single_benchmark.to_json()
|
|
437
|
-
new_benchmark: BenchmarkForBot = BenchmarkForBot.from_json(json_object)
|
|
438
|
-
```
|
|
439
|
-
|
|
440
|
-
Once you have benchmark files in your project directory you can run `streamlit run forecasting_tools/benchmarking/benchmark_displayer.py` to get a UI with the benchmarks. You can also put `forecasting-tools.run_benchmark_streamlit_page()` into a new file, and run this file with streamlit to achieve the same results. This will allow you to see metrics side by side, explore code of past bots, see the actual bot responses, etc. It will pull in any files in your directory that contain "bench" in the name and are json. Results may take a while to load for large benchmark files.
|
|
441
|
-
|
|
442
|
-

|
|
443
|
-

|
|
444
|
-
|
|
445
|
-
## Metaculus API
|
|
446
|
-
The Metaculus API wrapper helps interact with Metaculus questions and tournaments. Grabbing questions returns a pydantic object, and supports important information for Binary, Multiple Choice, Numeric,and Date questions.
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
```python
|
|
450
|
-
from forecasting_tools import MetaculusApi, ApiFilter, DataOrganizer
|
|
341
|
+
from forecasting_tools import MetaculusClient, ApiFilter, DataOrganizer
|
|
451
342
|
from datetime import datetime
|
|
452
343
|
|
|
344
|
+
metaculus_client = MetaculusClient()
|
|
453
345
|
|
|
454
|
-
|
|
455
|
-
question = MetaculusApi.get_question_by_post_id(578)
|
|
346
|
+
question = metaculus_client.get_question_by_post_id(578)
|
|
456
347
|
print(f"Question found with url: {question.page_url}")
|
|
457
348
|
|
|
458
|
-
|
|
459
|
-
|
|
349
|
+
question = metaculus_client.get_question_by_url(
|
|
350
|
+
"https://www.metaculus.com/questions/578/human-extinction-by-2100/"
|
|
351
|
+
)
|
|
460
352
|
print(f"Question found with url: {question.page_url}")
|
|
461
353
|
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
tournament_id=MetaculusApi.CURRENT_QUARTERLY_CUP_ID
|
|
354
|
+
questions = metaculus_client.get_all_open_questions_from_tournament(
|
|
355
|
+
tournament_id=MetaculusClient.CURRENT_METACULUS_CUP_ID
|
|
465
356
|
)
|
|
466
357
|
print(f"Num tournament questions: {len(questions)}")
|
|
467
358
|
|
|
468
|
-
# Get questions matching a filter
|
|
469
359
|
api_filter = ApiFilter(
|
|
470
360
|
num_forecasters_gte=40,
|
|
471
361
|
close_time_gt=datetime(2023, 12, 31),
|
|
@@ -474,35 +364,31 @@ api_filter = ApiFilter(
|
|
|
474
364
|
allowed_types=["binary", "multiple_choice", "numeric", "date"],
|
|
475
365
|
allowed_statuses=["resolved"],
|
|
476
366
|
)
|
|
477
|
-
questions = await
|
|
367
|
+
questions = await metaculus_client.get_questions_matching_filter(
|
|
478
368
|
api_filter=api_filter,
|
|
479
369
|
num_questions=50, # Remove this field to make it not error if you don't get 50 questions. However it will only go through one page of questions which may miss questions matching the ApiFilter since some filters are handled locally.
|
|
480
|
-
randomly_sample=False
|
|
370
|
+
randomly_sample=False,
|
|
481
371
|
)
|
|
482
372
|
print(f"Num filtered questions: {len(questions)}")
|
|
483
373
|
|
|
484
|
-
# Load and save questions/reports
|
|
485
374
|
file_path = "temp/questions.json"
|
|
486
|
-
DataOrganizer.save_questions_to_file_path(questions, file_path) #
|
|
375
|
+
DataOrganizer.save_questions_to_file_path(questions, file_path) # Overwrites the file if it already exists
|
|
487
376
|
questions = DataOrganizer.load_questions_from_file_path(file_path)
|
|
488
377
|
|
|
489
|
-
|
|
490
|
-
benchmark_questions = MetaculusApi.get_benchmark_questions(
|
|
378
|
+
benchmark_questions = metaculus_client.get_benchmark_questions(
|
|
491
379
|
num_of_questions_to_return=20
|
|
492
380
|
)
|
|
493
381
|
print(f"Num benchmark questions: {len(benchmark_questions)}")
|
|
494
382
|
|
|
495
|
-
|
|
496
|
-
MetaculusApi.post_binary_question_prediction(
|
|
383
|
+
metaculus_client.post_binary_question_prediction(
|
|
497
384
|
question_id=578, # Note that the question ID is not always the same as the post ID
|
|
498
|
-
prediction_in_decimal=0.012 # Must be between 0.
|
|
385
|
+
prediction_in_decimal=0.012, # Must be between 0.001 and 0.999
|
|
499
386
|
)
|
|
500
387
|
print("Posted prediction")
|
|
501
388
|
|
|
502
|
-
|
|
503
|
-
MetaculusApi.post_question_comment(
|
|
389
|
+
metaculus_client.post_question_comment(
|
|
504
390
|
post_id=578,
|
|
505
|
-
comment_text="Here's example reasoning for testing... This will be a private comment..."
|
|
391
|
+
comment_text="Here's example reasoning for testing... This will be a private comment...",
|
|
506
392
|
)
|
|
507
393
|
print("Posted comment")
|
|
508
394
|
```
|
|
@@ -516,237 +402,39 @@ print("Posted comment")
|
|
|
516
402
|
Posted comment
|
|
517
403
|
|
|
518
404
|
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
## Smart Searcher
|
|
522
|
-
The Smart Searcher acts like an LLM with internet access. It works a lot like Perplexity.ai API, except:
|
|
523
|
-
- It has clickable citations that highlights and links directly to the paragraph cited using text fragments
|
|
524
|
-
- You can ask the AI to use filters for domain, date, and keywords
|
|
525
|
-
- There are options for structured output (Pydantic objects, lists, dict, list\[dict\], etc.)
|
|
526
|
-
- Concurrent search execution for faster results
|
|
527
|
-
- Optional detailed works cited list
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
```python
|
|
531
|
-
|
|
532
|
-
searcher = SmartSearcher(
|
|
533
|
-
temperature=0,
|
|
534
|
-
num_searches_to_run=2,
|
|
535
|
-
num_sites_per_search=10, # Results returned per search
|
|
536
|
-
include_works_cited_list=False # Add detailed citations at the end
|
|
537
|
-
)
|
|
538
|
-
|
|
539
|
-
response = await searcher.invoke(
|
|
540
|
-
"What is the recent news for Apple?"
|
|
541
|
-
)
|
|
542
|
-
|
|
543
|
-
print(response)
|
|
544
|
-
```
|
|
545
|
-
|
|
546
|
-
Example output:
|
|
547
|
-
> Recent news about Apple includes several significant developments:
|
|
548
|
-
>
|
|
549
|
-
> 1. **Expansion in India**: Apple is planning to open four more stores in India, with two in Delhi and Mumbai, and two in Bengaluru and Pune. This decision follows record revenues in India for the September 2024 quarter, driven by strong iPhone sales. Tim Cook, Apple's CEO, highlighted the enthusiasm and growth in the Indian market during the company's earnings call \[[1](https://telecomtalk.info/tim-cook-makes-major-announcement-for-apple-in-india/984260/#:~:text=This%20is%20not%20a%20new,first%20time%20Apple%20confirmed%20it.)\]\[[4](https://telecomtalk.info/tim-cook-makes-major-announcement-for-apple-in-india/984260/#:~:text=This%20is%20not%20a%20new,set%20an%20all%2Dtime%20revenue%20record.)\]\[[5](https://telecomtalk.info/tim-cook-makes-major-announcement-for-apple-in-india/984260/#:~:text=Previously%2C%20Diedre%20O%27Brien%2C%20Apple%27s%20senior,East%2C%20India%20and%20South%20Asia.)\]\[[8](https://telecomtalk.info/tim-cook-makes-major-announcement-for-apple-in-india/984260/#:~:text=At%20the%20company%27s%20earnings%20call,four%20new%20stores%20in%20India.)\].
|
|
550
|
-
>
|
|
551
|
-
> 2. **Product Launches**: Apple is set to launch new iMac, Mac mini, and MacBook Pro models with M4 series chips on November 8, 2024. Additionally, the Vision Pro headset will be available in South Korea and the United Arab Emirates starting November 15, 2024. The second season of the Apple TV+ sci-fi series "Silo" will also premiere on November 15, 2024 \[[2](https://www.macrumors.com/2024/11/01/what-to-expect-from-apple-this-november/#:~:text=And%20the%20Vision%20Pro%20launches,the%20App%20Store%2C%20and%20more.)\]\[[12](https://www.macrumors.com/2024/11/01/what-to-expect-from-apple-this-november/#:~:text=As%20for%20hardware%2C%20the%20new,announcements%20in%20store%20this%20November.)\].
|
|
552
|
-
>
|
|
553
|
-
> ... etc ...
|
|
405
|
+
### Group Questions
|
|
554
406
|
|
|
555
|
-
|
|
407
|
+
Several of the methods above accept a `group_question_mode` parameter that controls how Metaculus group questions (e.g. "How many people will die of coronavirus in [period]?") are handled:
|
|
408
|
+
- `"exclude"` — drop group questions from the result.
|
|
409
|
+
- `"unpack_subquestions"` — turn each subquestion into a separate normal question.
|
|
556
410
|
|
|
411
|
+
For backwards compatibility, the default is `"exclude"` for `get_question_by_post_id`, `get_question_by_url`, `ApiFilter` (used by `get_questions_matching_filter`), and `get_benchmark_questions` — so group questions don't get overweighted in benchmarks. The exception is `get_all_open_questions_from_tournament`, which defaults to `"unpack_subquestions"` so all subquestions are forecasted as normal questions.
|
|
557
412
|
|
|
558
413
|
```python
|
|
559
|
-
from
|
|
560
|
-
from forecasting_tools import SmartSearcher
|
|
561
|
-
|
|
562
|
-
class Company(BaseModel):
|
|
563
|
-
name: str = Field(description="Full company name")
|
|
564
|
-
market_cap: float = Field(description="Market capitalization in billions USD")
|
|
565
|
-
key_products: list[str] = Field(description="Main products or services")
|
|
566
|
-
relevance: str = Field(description="Why this company is relevant to the search")
|
|
567
|
-
|
|
568
|
-
searcher = SmartSearcher(temperature=0, num_searches_to_run=4, num_sites_per_search=10)
|
|
569
|
-
|
|
570
|
-
schema_instructions = searcher.get_schema_format_instructions_for_pydantic_type(Company)
|
|
571
|
-
prompt = f"""Find companies that are leading the development of autonomous vehicles.
|
|
572
|
-
Return as a list of companies with their details. Remember to give me a list of the schema provided.
|
|
573
|
-
|
|
574
|
-
{schema_instructions}"""
|
|
575
|
-
|
|
576
|
-
companies = await searcher.invoke_and_return_verified_type(prompt, list[Company])
|
|
577
|
-
|
|
578
|
-
for company in companies:
|
|
579
|
-
print(f"\n{company.name} (${company.market_cap}B)")
|
|
580
|
-
print(f"Relevance: {company.relevance}")
|
|
581
|
-
print("Key Products:")
|
|
582
|
-
for product in company.key_products:
|
|
583
|
-
print(f"- {product}")
|
|
584
|
-
```
|
|
585
|
-
|
|
586
|
-
The schema instructions will format the Pydantic model into clear instructions for the AI about the expected output format and field descriptions.
|
|
587
|
-
|
|
414
|
+
from forecasting_tools import MetaculusClient, ApiFilter
|
|
588
415
|
|
|
589
|
-
|
|
590
|
-
The Key Factors Researcher helps identify and analyze key factors that should be considered for a forecasting question. As of last update, this is the most reliable of the tools, and gives something useful and accurate almost every time. It asks a lot of questions, turns search results into a long list of bullet points, rates each bullet point on ~8 criteria, and returns the top results.
|
|
416
|
+
metaculus_client = MetaculusClient()
|
|
591
417
|
|
|
418
|
+
# Unpack a group question into its subquestions
|
|
419
|
+
result = metaculus_client.get_question_by_post_id(
|
|
420
|
+
post_id=..., # a group-question post
|
|
421
|
+
group_question_mode="unpack_subquestions",
|
|
422
|
+
) # returns list[MetaculusQuestion] for group posts
|
|
592
423
|
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
ScoredKeyFactor
|
|
598
|
-
)
|
|
599
|
-
|
|
600
|
-
# Consider using MetaculusApi.get_question_by_id or MetaculusApi.get_question_by_url instead
|
|
601
|
-
question = BinaryQuestion(
|
|
602
|
-
question_text="Will YouTube be blocked in Russia?",
|
|
603
|
-
background_info="...", # Or 'None'
|
|
604
|
-
resolution_criteria="...", # Or 'None'
|
|
605
|
-
fine_print="...", # Or 'None'
|
|
606
|
-
)
|
|
607
|
-
|
|
608
|
-
# Find key factors
|
|
609
|
-
key_factors = await KeyFactorsResearcher.find_and_sort_key_factors(
|
|
610
|
-
metaculus_question=question,
|
|
611
|
-
num_key_factors_to_return=5, # Number of final factors to return
|
|
612
|
-
num_questions_to_research_with=26 # Number of research questions to generate
|
|
613
|
-
)
|
|
614
|
-
|
|
615
|
-
print(ScoredKeyFactor.turn_key_factors_into_markdown_list(key_factors))
|
|
616
|
-
```
|
|
617
|
-
|
|
618
|
-
Example output:
|
|
619
|
-
> - The Russian authorities have slowed YouTube speeds to near unusable levels, indicating a potential groundwork for a future ban. [Source Published on 2024-09-12](https://meduza.io/en/feature/2024/09/12/the-russian-authorities-slowed-youtube-speeds-to-near-unusable-levels-so-why-are-kremlin-critics-getting-more-views#:~:text=Kolezev%20attributed%20this%20to%20the,suddenly%20stopped%20working%20in%20Russia.)
|
|
620
|
-
> - Russian lawmaker Alexander Khinshtein stated that YouTube speeds would be deliberately slowed by up to 70% due to Google's non-compliance with Russian demands, indicating escalating measures against YouTube. [Source Published on 2024-07-25](https://www.yahoo.com/news/russia-slow-youtube-speeds-google-180512830.html#:~:text=Russia%20will%20deliberately%20slow%20YouTube,forces%20and%20promoting%20extremist%20content.)
|
|
621
|
-
> - The press secretary of President Vladimir Putin, Dmitry Peskov, denied that the authorities intended to block YouTube, attributing access issues to outdated equipment due to sanctions. [Source Published on 2024-08-17](https://www.wsws.org/en/articles/2024/08/17/pbyj-a17.html#:~:text=%5BAP%20Photo%2FAP%20Photo%5D%20On%20July,two%20years%20due%20to%20sanctions.)
|
|
622
|
-
> - YouTube is currently the last Western social media platform still operational in Russia, with over 93 million users in the country. [Source Published on 2024-07-26](https://www.techradar.com/pro/vpn/youtube-is-getting-throttled-in-russia-heres-how-to-unblock-it#:~:text=If%20you%27re%20in%20Russia%20and,platform%20to%20work%20in%20Russia.)
|
|
623
|
-
> - Russian users reported mass YouTube outages amid growing official criticism, with reports of thousands of glitches in August 2024. [Source Published on 2024-08-09](https://www.aljazeera.com/news/2024/8/9/russian-users-report-mass-youtube-outage-amid-growing-official-criticism?traffic_source=rss#:~:text=Responding%20to%20this%2C%20a%20YouTube,reported%20about%20YouTube%20in%20Russia.)
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
The simplified pydantic structure of the scored key factors is:
|
|
627
|
-
```python
|
|
628
|
-
class ScoredKeyFactor():
|
|
629
|
-
text: str
|
|
630
|
-
factor_type: KeyFactorType (Pro, Con, or Base_Rate)
|
|
631
|
-
citation: str
|
|
632
|
-
source_publish_date: datetime | None
|
|
633
|
-
url: str
|
|
634
|
-
score_card: ScoreCard
|
|
635
|
-
score: int
|
|
636
|
-
display_text: str
|
|
637
|
-
```
|
|
638
|
-
|
|
639
|
-
## Base Rate Researcher
|
|
640
|
-
The Base Rate Researcher helps calculate historical base rates for events. As of last update, it gives decent results around 50% of the time. It orchestrates the Niche List Researcher and the Fermi Estimator to find base rate.
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
```python
|
|
644
|
-
from forecasting_tools import BaseRateResearcher
|
|
645
|
-
|
|
646
|
-
# Initialize researcher
|
|
647
|
-
researcher = BaseRateResearcher(
|
|
648
|
-
"How often has Apple been successfully sued for patent violations?"
|
|
649
|
-
)
|
|
650
|
-
|
|
651
|
-
# Get base rate analysis
|
|
652
|
-
report = await researcher.make_base_rate_report()
|
|
653
|
-
|
|
654
|
-
print(f"Historical rate: {report.historical_rate:.2%}")
|
|
655
|
-
print(report.markdown_report)
|
|
656
|
-
```
|
|
657
|
-
|
|
658
|
-
## Niche List Researcher
|
|
659
|
-
The Niche List Researcher helps analyze specific lists of events or items. The researcher will:
|
|
660
|
-
1. Generate a comprehensive list of potential matches
|
|
661
|
-
2. Remove duplicates
|
|
662
|
-
3. Fact check each item against multiple criteria
|
|
663
|
-
4. Return only validated items (unless include_incorrect_items=True)
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
```python
|
|
667
|
-
from forecasting_tools import NicheListResearcher
|
|
668
|
-
|
|
669
|
-
researcher = NicheListResearcher(
|
|
670
|
-
type_of_thing_to_generate="Times Apple was successfully sued for patent violations between 2000-2024"
|
|
671
|
-
)
|
|
672
|
-
|
|
673
|
-
fact_checked_items = await researcher.research_niche_reference_class(
|
|
674
|
-
return_invalid_items=False
|
|
675
|
-
)
|
|
676
|
-
|
|
677
|
-
for item in fact_checked_items:
|
|
678
|
-
print(item)
|
|
679
|
-
```
|
|
680
|
-
|
|
681
|
-
The simplified pydantic structure of the fact checked items is:
|
|
682
|
-
```python
|
|
683
|
-
class FactCheckedItem():
|
|
684
|
-
item_name: str
|
|
685
|
-
description: str
|
|
686
|
-
is_uncertain: bool | None = None
|
|
687
|
-
initial_citations: list[str] | None = None
|
|
688
|
-
fact_check: FactCheck
|
|
689
|
-
type_description: str
|
|
690
|
-
is_valid: bool
|
|
691
|
-
supporting_urls: list[str]
|
|
692
|
-
one_line_fact_check_summary: str
|
|
693
|
-
|
|
694
|
-
class FactCheck(BaseModel):
|
|
695
|
-
criteria_assessments: list[CriteriaAssessment]
|
|
696
|
-
is_valid: bool
|
|
697
|
-
|
|
698
|
-
class CriteriaAssessment():
|
|
699
|
-
short_name: str
|
|
700
|
-
description: str
|
|
701
|
-
validity_assessment: str
|
|
702
|
-
is_valid_or_unknown: bool | None
|
|
703
|
-
citation_proving_assessment: str | None
|
|
704
|
-
url_proving_assessment: str | None:
|
|
705
|
-
```
|
|
706
|
-
|
|
707
|
-
## Fermi Estimator
|
|
708
|
-
The Fermi Estimator helps break down numerical estimates using Fermi estimation techniques.
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
```python
|
|
713
|
-
from forecasting_tools import Estimator
|
|
714
|
-
|
|
715
|
-
estimator = Estimator(
|
|
716
|
-
type_of_thing_to_estimate="books published worldwide each year",
|
|
717
|
-
previous_research=None # Optional: Pass in existing research
|
|
424
|
+
# Same option on a filtered query
|
|
425
|
+
api_filter = ApiFilter(
|
|
426
|
+
allowed_statuses=["open"],
|
|
427
|
+
group_question_mode="unpack_subquestions",
|
|
718
428
|
)
|
|
719
|
-
|
|
720
|
-
size, explanation = await estimator.estimate_size()
|
|
721
|
-
|
|
722
|
-
print(f"Estimate: {size:,}")
|
|
723
|
-
print(explanation)
|
|
429
|
+
questions = await metaculus_client.get_questions_matching_filter(api_filter=api_filter)
|
|
724
430
|
```
|
|
725
431
|
|
|
726
|
-
Example output (Fake data with links not added):
|
|
727
|
-
> I estimate that there are 2,750,000 'books published worldwide each year'.
|
|
728
|
-
>
|
|
729
|
-
> **Facts**:
|
|
730
|
-
> - Traditional publishers release approximately 500,000 new titles annually in English-speaking countries [1]
|
|
731
|
-
> - China publishes around 450,000 new books annually [2]
|
|
732
|
-
> - The global book market was valued at $92.68 billion in 2023 [3]
|
|
733
|
-
> - Self-published titles have grown by 264% in the last 5 years [4]
|
|
734
|
-
> - Non-English language markets account for about 50% of global publishing [5]
|
|
735
|
-
>
|
|
736
|
-
> **Estimation Steps and Assumptions**:
|
|
737
|
-
> 1. Start with traditional English publishing: 500,000 titles
|
|
738
|
-
> 2. Add Chinese market: 500,000 + 450,000 = 950,000
|
|
739
|
-
> 3. Account for other major languages (50% of market): 950,000 * 2 = 1,900,000
|
|
740
|
-
> 4. Add self-published titles (estimated 45% of total): 1,900,000 * 1.45 = 2,755,000
|
|
741
|
-
>
|
|
742
|
-
> **Background Research**: [Additional research details...]
|
|
743
|
-
|
|
744
432
|
## General LLM
|
|
745
|
-
The `GeneralLlm` class is a wrapper around
|
|
433
|
+
The `GeneralLlm` class is a wrapper around litellm's acompletion function that adds some functionality like retry logic, calling the metaculus proxy, and cost callback handling. Litellm supports every model, most every parameter, and acts as one interface for every provider. See the litellm's acompletion function for a full list of parameters. Not all models will support all parameters. Additionally the Metaculus proxy doesn't support all models.
|
|
746
434
|
|
|
747
435
|
|
|
748
436
|
```python
|
|
749
|
-
|
|
437
|
+
prompt = "What is the weather in Tokyo?"
|
|
750
438
|
result = await GeneralLlm(model="gpt-4o").invoke(prompt)
|
|
751
439
|
result = await GeneralLlm(model="claude-3-5-sonnet-20241022").invoke(prompt)
|
|
752
440
|
result = await GeneralLlm(model="metaculus/claude-3-5-sonnet-20241022").invoke(prompt) # Adding 'metaculus' Calls the Metaculus proxy
|
|
@@ -859,7 +547,7 @@ The `MonetaryCostManager` helps to track AI and API costs. It tracks expenses an
|
|
|
859
547
|
```python
|
|
860
548
|
from forecasting_tools import MonetaryCostManager
|
|
861
549
|
from forecasting_tools import (
|
|
862
|
-
ExaSearcher,
|
|
550
|
+
ExaSearcher, GeneralLlm
|
|
863
551
|
)
|
|
864
552
|
|
|
865
553
|
max_cost = 5.00
|
|
@@ -867,7 +555,6 @@ max_cost = 5.00
|
|
|
867
555
|
with MonetaryCostManager(max_cost) as cost_manager:
|
|
868
556
|
prompt = "What is the weather in Tokyo?"
|
|
869
557
|
result = await GeneralLlm(model="gpt-4o").invoke(prompt)
|
|
870
|
-
result = await SmartSearcher(model="claude-3-5-sonnet-20241022").invoke(prompt)
|
|
871
558
|
result = await ExaSearcher().invoke(prompt)
|
|
872
559
|
# ... etc ...
|
|
873
560
|
|