forecasting-tools 0.2.87__tar.gz → 0.2.89__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/PKG-INFO +6 -6
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/other/data_analyzer.py +4 -6
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/key_factors_researcher.py +1 -1
- forecasting_tools-0.2.89/forecasting_tools/ai_models/agent_wrappers.py +75 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/general_llm.py +1 -11
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/prompt_optimizer.py +41 -52
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/cp_benchmarking/benchmarker.py +36 -43
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/numeric_report.py +40 -9
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/questions.py +8 -4
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/forecast_bot.py +11 -15
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/chat_page.py +21 -34
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/pyproject.toml +7 -7
- forecasting_tools-0.2.87/forecasting_tools/ai_models/agent_wrappers.py +0 -212
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/LICENSE +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/README.md +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_member_agent.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_orchestrator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/member_profiles.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/ai_congress_v2/tools.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/base_rate_researcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/deduplicator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/estimator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/base_rates/niche_list_researcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/configured_llms.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/general_researcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/question_generator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/question_responder.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/question_router.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/deprecated/research_coordinator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/minor_tools.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/other/hosted_file.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/generated_question.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/harmful_question_identifier.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/q3_q4_quarterly_questions.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/question_decomposer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/question_operationalizer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/simple_question.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/question_generators/topic_generator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/computer_use.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/find_a_dataset.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/research/smart_searcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/agent_runner.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/effect_engine.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/industrial_basin.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/milbrook.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_business.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_dnd.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_election.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/willowbrook.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/z_broken_mafia.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/forecast_resolver.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_policy_agent.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_runner.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_storage.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/simulator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/agents_and_tools/situation_simulator/situation_generator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/ai_utils/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/ai_utils/openai_utils.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/ai_utils/response_types.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/README.md +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/claude35sonnet.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/deepseek_r1.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpt4o.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpt4ovision.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpto1.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/gpto1preview.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/metaculus4o.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/deprecated_model_classes/perplexity.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/exa_searcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/ai_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/combined_llm_archetype.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/incurs_cost.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/named_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/outputs_text.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/priced_per_request.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/request_limited_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/retryable_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/time_limited_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/token_limited_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/tokens_are_calculatable.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_interfaces/tokens_incur_cost.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/model_tracker.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/hard_limit_manager.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/monetary_cost_manager.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/resource_managers/refreshing_bucket_rate_limiter.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/bot_evaluator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/bot_optimizer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/control_prompt.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/customizable_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/prompt_data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/auto_optimizers/question_plus_research.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/cp_benchmarking/benchmark_displayer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/cp_benchmarking/benchmark_for_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/binary_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/coherence_link.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/conditional_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/conditional_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/data_organizer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/forecast_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/markdown_tree.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/multiple_choice_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/timestamped_predictions.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/user_response.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/bot_lists.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q1_veritas_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q1t_w_convert_to_binary.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q1t_w_personas_and_exa.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q2t_w_decomposition.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q3t_w_asknews.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q3t_w_exa.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q3t_w_q4vbinary.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4_veritas_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4v_w_exa.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_dseekr1.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_o1_preview.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/main_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/gpt_4_1_optimized_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q1_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q2_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q3_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/q4_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/research_only_bot_2025_fall.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/template_bot_2025_fall.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/template_bot_2026_spring.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/official_bots/uniform_probability_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/forecast_bots/template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/Home.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/base_rate_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/benchmark_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/congress_v2_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/csv_agent.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/estimator_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/forecaster_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/intervention_leaderboard_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/key_factors_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/niche_list_researcher_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/app_pages/simulator_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/base_rate_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/congress_page_example.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/congress_v2_page_example.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/estimator_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/forecast_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/key_factors_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/niche_list_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/example_outputs/question_generator_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/app_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/custom_auth.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/report_displayer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/front_end/helpers/tool_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/asknews_cache.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/asknews_searcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/forecast_database_manager.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/metaculus_api.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/metaculus_client.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/prediction_extractor.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/structure_output.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/helpers/works_cited_creator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/async_batching.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/coda_utils.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/custom_logger.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/file_manipulation.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/jsonable.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/misc.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/util/stats.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: forecasting-tools
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.89
|
|
4
4
|
Summary: AI forecasting and research tools to help humans reason about and forecast the future
|
|
5
5
|
License: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -26,19 +26,19 @@ Requires-Dist: asknews (>=0.9.1,<0.14.0)
|
|
|
26
26
|
Requires-Dist: asyncio (>=3.0.0,<5.0.0)
|
|
27
27
|
Requires-Dist: exceptiongroup (>=1.2.2,<2.0.0)
|
|
28
28
|
Requires-Dist: faker (>=37.0.0,<41.0.0)
|
|
29
|
-
Requires-Dist: hyperbrowser (>=0.53.0,<0.
|
|
30
|
-
Requires-Dist: litellm (>=1.59.1,<
|
|
29
|
+
Requires-Dist: hyperbrowser (>=0.53.0,<1.0.0)
|
|
30
|
+
Requires-Dist: litellm (>=1.59.1,<2.0.0,!=1.76.*,!=1.77.*,!=1.82.7,!=1.82.8)
|
|
31
31
|
Requires-Dist: nest-asyncio (>=1.5.8,<2.0.0)
|
|
32
32
|
Requires-Dist: numpy (>=1.26.0,<3.0.0)
|
|
33
33
|
Requires-Dist: openai (>=1.51.0,<3.0.0)
|
|
34
|
-
Requires-Dist: openai-agents[litellm] (>=0.2.0,<0.
|
|
34
|
+
Requires-Dist: openai-agents[litellm] (>=0.2.0,<0.14.0)
|
|
35
35
|
Requires-Dist: pandas (>=2.2.3,<4.0.0)
|
|
36
36
|
Requires-Dist: pendulum (>=3.1.0,<4.0.0)
|
|
37
|
-
Requires-Dist: pillow (>=9.0.0,<
|
|
37
|
+
Requires-Dist: pillow (>=9.0.0,<13.0.0)
|
|
38
38
|
Requires-Dist: plotly (>=5.24.1,<7.0.0)
|
|
39
39
|
Requires-Dist: pydantic (>=2.9.2,<3.0.0)
|
|
40
40
|
Requires-Dist: python-dotenv (>=1.0.0,<2.0.0)
|
|
41
|
-
Requires-Dist: regex (>=2024.11.6,<
|
|
41
|
+
Requires-Dist: regex (>=2024.11.6,<2027.0.0)
|
|
42
42
|
Requires-Dist: requests (>=2.32.3,<3.0.0)
|
|
43
43
|
Requires-Dist: scikit-learn (>=1.5.2,<2.0.0)
|
|
44
44
|
Requires-Dist: streamlit (>=1.20.0,<2.0.0)
|
|
@@ -12,7 +12,6 @@ from forecasting_tools.ai_models.agent_wrappers import (
|
|
|
12
12
|
CodingTool,
|
|
13
13
|
agent_tool,
|
|
14
14
|
event_to_tool_message,
|
|
15
|
-
general_trace_or_span,
|
|
16
15
|
)
|
|
17
16
|
from forecasting_tools.util.misc import clean_indents
|
|
18
17
|
|
|
@@ -119,10 +118,9 @@ class DataAnalyzer:
|
|
|
119
118
|
|
|
120
119
|
|
|
121
120
|
if __name__ == "__main__":
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
instructions="Please multiple 52.675 x 6.547 x 9867.5476 x 4356.5",
|
|
126
|
-
)
|
|
121
|
+
answer = asyncio.run(
|
|
122
|
+
DataAnalyzer().run_data_analysis(
|
|
123
|
+
instructions="Please multiple 52.675 x 6.547 x 9867.5476 x 4356.5",
|
|
127
124
|
)
|
|
125
|
+
)
|
|
128
126
|
print(answer)
|
|
@@ -222,7 +222,7 @@ class KeyFactorsResearcher:
|
|
|
222
222
|
# Grading Scale for {ScoreCardGrade.__class__.__name__}
|
|
223
223
|
- {ScoreCardGrade.VERY_BAD.value}: Generally poor quality
|
|
224
224
|
- {ScoreCardGrade.BAD.value}: Below average quality
|
|
225
|
-
- {ScoreCardGrade.OK.value}:
|
|
225
|
+
- {ScoreCardGrade.OK.value}: Average quality
|
|
226
226
|
- {ScoreCardGrade.GOOD.value}: Above average quality
|
|
227
227
|
- {ScoreCardGrade.VERY_GOOD.value}: Exceptional quality
|
|
228
228
|
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import logging
|
|
3
|
+
|
|
4
|
+
import nest_asyncio
|
|
5
|
+
from agents import Agent, CodeInterpreterTool, FunctionTool, Runner
|
|
6
|
+
from agents import function_tool as ft
|
|
7
|
+
from agents.extensions.models.litellm_model import LitellmModel
|
|
8
|
+
from agents.stream_events import StreamEvent
|
|
9
|
+
|
|
10
|
+
from forecasting_tools.ai_models.model_tracker import ModelTracker
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
nest_asyncio.apply()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class AgentSdkLlm(LitellmModel):
|
|
19
|
+
"""
|
|
20
|
+
Wrapper around openai-agent-sdk's LiteLlm Model for later extension
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
async def get_response(self, *args, **kwargs): # NOSONAR
|
|
24
|
+
ModelTracker.give_cost_tracking_warning_if_needed(self.model)
|
|
25
|
+
response = await super().get_response(*args, **kwargs)
|
|
26
|
+
await asyncio.sleep(
|
|
27
|
+
0.0001
|
|
28
|
+
) # For whatever reason, it seems you need to await a coroutine to get the litellm cost callback to work
|
|
29
|
+
return response
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
AgentRunner = Runner # Alias for Runner for later extension
|
|
33
|
+
AgentTool = FunctionTool # Alias for FunctionTool for later extension
|
|
34
|
+
AiAgent = Agent # Alias for Agent for later extension
|
|
35
|
+
CodingTool = CodeInterpreterTool # Alias for CodeInterpreterTool for later extension
|
|
36
|
+
agent_tool = ft # Alias for function_tool for later extension
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def event_to_tool_message(event: StreamEvent) -> str | None:
|
|
40
|
+
text = ""
|
|
41
|
+
if event.type == "run_item_stream_event":
|
|
42
|
+
item = event.item
|
|
43
|
+
if item.type == "message_output_item":
|
|
44
|
+
content = item.raw_item.content[0]
|
|
45
|
+
if content.type == "output_text":
|
|
46
|
+
# text = content.text
|
|
47
|
+
text = "" # the text is already streamed separate from this function
|
|
48
|
+
elif content.type == "output_refusal":
|
|
49
|
+
text = content.refusal
|
|
50
|
+
else:
|
|
51
|
+
text = "Error: unknown content type"
|
|
52
|
+
elif item.type == "tool_call_item":
|
|
53
|
+
if item.raw_item.type == "code_interpreter_call":
|
|
54
|
+
text = (
|
|
55
|
+
f"\nCode interpreter code:\n```python\n{item.raw_item.code}\n```\n"
|
|
56
|
+
)
|
|
57
|
+
else:
|
|
58
|
+
tool_name = getattr(item.raw_item, "name", "unknown_tool")
|
|
59
|
+
tool_args = getattr(item.raw_item, "arguments", {})
|
|
60
|
+
text = f"Tool call: {tool_name}({tool_args})"
|
|
61
|
+
elif item.type == "tool_call_output_item":
|
|
62
|
+
output = getattr(item, "output", str(item.raw_item))
|
|
63
|
+
text = f"Tool output:\n\n{output}"
|
|
64
|
+
elif item.type == "handoff_call_item":
|
|
65
|
+
handoff_info = getattr(item.raw_item, "name", "handoff")
|
|
66
|
+
text = f"Handoff call: {handoff_info}"
|
|
67
|
+
elif item.type == "handoff_output_item":
|
|
68
|
+
text = f"Handoff output: {str(item.raw_item)}"
|
|
69
|
+
elif item.type == "reasoning_item":
|
|
70
|
+
text = f"Reasoning: {str(item.raw_item)}"
|
|
71
|
+
# elif event.type == "agent_updated_stream_event":
|
|
72
|
+
# text += f"Agent updated: {event.new_agent.name}\n\n"
|
|
73
|
+
if text == "":
|
|
74
|
+
return None
|
|
75
|
+
return text
|
{forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/ai_models/general_llm.py
RENAMED
|
@@ -21,7 +21,6 @@ from openai.types.responses import (
|
|
|
21
21
|
ResponseReasoningItem,
|
|
22
22
|
)
|
|
23
23
|
|
|
24
|
-
from forecasting_tools.ai_models.agent_wrappers import track_generation
|
|
25
24
|
from forecasting_tools.ai_models.ai_utils.openai_utils import (
|
|
26
25
|
OpenAiUtils,
|
|
27
26
|
VisionMessageData,
|
|
@@ -251,16 +250,7 @@ class GeneralLlm(
|
|
|
251
250
|
) -> Any:
|
|
252
251
|
logger.debug(f"Invoking model with prompt: {prompt}")
|
|
253
252
|
|
|
254
|
-
|
|
255
|
-
input=self.model_input_to_message(prompt),
|
|
256
|
-
model=self.model,
|
|
257
|
-
) as span:
|
|
258
|
-
direct_call_response = await self._mockable_direct_call_to_model(prompt)
|
|
259
|
-
answer = direct_call_response.data
|
|
260
|
-
span.span_data.output = [{"role": "assistant", "content": answer}]
|
|
261
|
-
# span.span_data.usage = usage.model_dump()
|
|
262
|
-
span.span_data.model = self.model
|
|
263
|
-
span.span_data.model_config = self.litellm_kwargs
|
|
253
|
+
direct_call_response = await self._mockable_direct_call_to_model(prompt)
|
|
264
254
|
|
|
265
255
|
logger.debug(f"Model responded with: {direct_call_response}")
|
|
266
256
|
return direct_call_response
|
|
@@ -9,12 +9,7 @@ from pydantic import BaseModel, Field
|
|
|
9
9
|
from forecasting_tools.agents_and_tools.minor_tools import (
|
|
10
10
|
perplexity_reasoning_pro_search,
|
|
11
11
|
)
|
|
12
|
-
from forecasting_tools.ai_models.agent_wrappers import
|
|
13
|
-
AgentRunner,
|
|
14
|
-
AgentSdkLlm,
|
|
15
|
-
AiAgent,
|
|
16
|
-
general_trace_or_span,
|
|
17
|
-
)
|
|
12
|
+
from forecasting_tools.ai_models.agent_wrappers import AgentRunner, AgentSdkLlm, AiAgent
|
|
18
13
|
from forecasting_tools.auto_optimizers.prompt_data_models import PromptIdea
|
|
19
14
|
from forecasting_tools.helpers.structure_output import structure_output
|
|
20
15
|
from forecasting_tools.util.misc import clean_indents, retry_async_function
|
|
@@ -112,63 +107,57 @@ class PromptOptimizer:
|
|
|
112
107
|
)
|
|
113
108
|
|
|
114
109
|
async def create_optimized_prompt(self) -> OptimizationRun:
|
|
115
|
-
|
|
116
|
-
return await self._create_optimized_prompt()
|
|
110
|
+
return await self._create_optimized_prompt()
|
|
117
111
|
|
|
118
112
|
async def _create_optimized_prompt(self) -> OptimizationRun:
|
|
119
113
|
iteration_num = 0
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
prompts_still_needed,
|
|
141
|
-
)
|
|
142
|
-
starting_prompts.extend(additional_initial_prompts)
|
|
143
|
-
|
|
144
|
-
offspring_prompts: list[ImplementedPrompt] = starting_prompts
|
|
145
|
-
assert (
|
|
146
|
-
seed_prompt in offspring_prompts
|
|
147
|
-
), "Seed prompt not found in offspring prompts"
|
|
148
|
-
all_evaluated_prompts: list[ScoredPrompt] = (
|
|
149
|
-
await self._evaluate_new_members(offspring_prompts)
|
|
114
|
+
iteration_num += 1
|
|
115
|
+
logger.info(
|
|
116
|
+
f"Generating initial prompt population of size {self.initial_prompt_population_size}"
|
|
117
|
+
)
|
|
118
|
+
seed_prompt = ImplementedPrompt(
|
|
119
|
+
text=self.initial_prompt,
|
|
120
|
+
idea=PromptIdea(
|
|
121
|
+
short_name="Initial Seed",
|
|
122
|
+
full_text="The user-provided initial prompt",
|
|
123
|
+
),
|
|
124
|
+
originating_ideas=[],
|
|
125
|
+
)
|
|
126
|
+
starting_prompts: list[ImplementedPrompt] = [seed_prompt]
|
|
127
|
+
prompts_still_needed = self.initial_prompt_population_size - len(
|
|
128
|
+
starting_prompts
|
|
129
|
+
)
|
|
130
|
+
if prompts_still_needed > 0:
|
|
131
|
+
additional_initial_prompts = await self._mutate_prompt(
|
|
132
|
+
starting_prompts[0],
|
|
133
|
+
prompts_still_needed,
|
|
150
134
|
)
|
|
151
|
-
|
|
135
|
+
starting_prompts.extend(additional_initial_prompts)
|
|
136
|
+
|
|
137
|
+
offspring_prompts: list[ImplementedPrompt] = starting_prompts
|
|
138
|
+
assert (
|
|
139
|
+
seed_prompt in offspring_prompts
|
|
140
|
+
), "Seed prompt not found in offspring prompts"
|
|
141
|
+
all_evaluated_prompts: list[ScoredPrompt] = await self._evaluate_new_members(
|
|
142
|
+
offspring_prompts
|
|
143
|
+
)
|
|
144
|
+
survivors = await self._kill_the_weak(all_evaluated_prompts)
|
|
152
145
|
|
|
153
146
|
while iteration_num < self.iterations:
|
|
154
147
|
iteration_num += 1
|
|
155
|
-
|
|
156
|
-
f"
|
|
157
|
-
|
|
158
|
-
):
|
|
159
|
-
logger.info(
|
|
160
|
-
f"Starting iteration {iteration_num + 1}/{self.iterations} - Current population size: {len(offspring_prompts)}"
|
|
161
|
-
)
|
|
148
|
+
logger.info(
|
|
149
|
+
f"Starting iteration {iteration_num + 1}/{self.iterations} - Current population size: {len(offspring_prompts)}"
|
|
150
|
+
)
|
|
162
151
|
|
|
163
|
-
|
|
152
|
+
offspring_prompts = await self._generate_new_prompts(survivors)
|
|
164
153
|
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
154
|
+
evaluated_prompts = await self._evaluate_new_members(offspring_prompts)
|
|
155
|
+
all_evaluated_prompts.extend(evaluated_prompts)
|
|
156
|
+
updated_population = survivors + evaluated_prompts
|
|
168
157
|
|
|
169
|
-
|
|
158
|
+
survivors = await self._kill_the_weak(updated_population)
|
|
170
159
|
|
|
171
|
-
|
|
160
|
+
self._log_duplicate_prompts(all_evaluated_prompts)
|
|
172
161
|
|
|
173
162
|
return OptimizationRun(scored_prompts=all_evaluated_prompts)
|
|
174
163
|
|
|
@@ -5,7 +5,6 @@ from typing import Sequence
|
|
|
5
5
|
|
|
6
6
|
import typeguard
|
|
7
7
|
|
|
8
|
-
from forecasting_tools.ai_models.agent_wrappers import general_trace_or_span
|
|
9
8
|
from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
|
|
10
9
|
MonetaryCostManager,
|
|
11
10
|
)
|
|
@@ -80,50 +79,44 @@ class Benchmarker:
|
|
|
80
79
|
self.code_to_snapshot = additional_code_to_snapshot
|
|
81
80
|
|
|
82
81
|
async def run_benchmark(self) -> list[BenchmarkForBot]:
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
self.number_of_questions_to_use,
|
|
90
|
-
)
|
|
91
|
-
else:
|
|
92
|
-
chosen_questions = self.questions_to_use
|
|
93
|
-
|
|
94
|
-
chosen_questions = typeguard.check_type(
|
|
95
|
-
chosen_questions, list[MetaculusQuestion]
|
|
96
|
-
)
|
|
97
|
-
|
|
98
|
-
if self.number_of_questions_to_use is not None:
|
|
99
|
-
assert len(chosen_questions) == self.number_of_questions_to_use
|
|
100
|
-
|
|
101
|
-
benchmarks: list[BenchmarkForBot] = self._initialize_benchmarks(
|
|
102
|
-
self.forecast_bots, chosen_questions
|
|
82
|
+
if self.questions_to_use is None:
|
|
83
|
+
assert (
|
|
84
|
+
self.number_of_questions_to_use is not None
|
|
85
|
+
), "number_of_questions_to_use must be provided if questions_to_use is not provided"
|
|
86
|
+
chosen_questions = MetaculusApi.get_benchmark_questions(
|
|
87
|
+
self.number_of_questions_to_use,
|
|
103
88
|
)
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
89
|
+
else:
|
|
90
|
+
chosen_questions = self.questions_to_use
|
|
91
|
+
|
|
92
|
+
chosen_questions = typeguard.check_type(
|
|
93
|
+
chosen_questions, list[MetaculusQuestion]
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
if self.number_of_questions_to_use is not None:
|
|
97
|
+
assert len(chosen_questions) == self.number_of_questions_to_use
|
|
98
|
+
|
|
99
|
+
benchmarks: list[BenchmarkForBot] = self._initialize_benchmarks(
|
|
100
|
+
self.forecast_bots, chosen_questions
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
batches = self._batch_questions(
|
|
104
|
+
self.forecast_bots,
|
|
105
|
+
benchmarks,
|
|
106
|
+
chosen_questions,
|
|
107
|
+
self.concurrent_question_batch_size,
|
|
108
|
+
)
|
|
109
|
+
try:
|
|
110
|
+
for i, batch in enumerate(batches):
|
|
111
|
+
await self._run_a_batch(batch)
|
|
112
|
+
if batch.is_last_batch_for_benchmark:
|
|
113
|
+
self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
|
|
114
|
+
except KeyboardInterrupt:
|
|
115
|
+
logger.warning(
|
|
116
|
+
"KeyboardInterrupt detected, saving current benchmark progress."
|
|
110
117
|
)
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
with general_trace_or_span(
|
|
114
|
-
f"{batch.benchmark.name} - Batch {i+1} of {len(batches)}"
|
|
115
|
-
):
|
|
116
|
-
await self._run_a_batch(batch)
|
|
117
|
-
if batch.is_last_batch_for_benchmark:
|
|
118
|
-
self._append_benchmarks_to_jsonl_if_configured(
|
|
119
|
-
[batch.benchmark]
|
|
120
|
-
)
|
|
121
|
-
except KeyboardInterrupt:
|
|
122
|
-
logger.warning(
|
|
123
|
-
"KeyboardInterrupt detected, saving current benchmark progress."
|
|
124
|
-
)
|
|
125
|
-
self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
|
|
126
|
-
raise
|
|
118
|
+
self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
|
|
119
|
+
raise
|
|
127
120
|
return benchmarks
|
|
128
121
|
|
|
129
122
|
async def _run_a_batch(self, batch: QuestionBatch) -> None:
|
|
@@ -84,8 +84,11 @@ class DatePercentile(BaseModel):
|
|
|
84
84
|
|
|
85
85
|
class NumericDistribution(BaseModel):
|
|
86
86
|
declared_percentiles: list[Percentile]
|
|
87
|
-
open_upper_bound: bool
|
|
88
|
-
open_lower_bound: bool
|
|
87
|
+
open_upper_bound: bool # If False, cdf[-1] must be 1.0 (no mass above upper bound)
|
|
88
|
+
open_lower_bound: bool # If False, cdf[0] must be 0.0 (no mass strictly below lower
|
|
89
|
+
# bound). Note: cdf[0] = P(outcome < lower_bound), so a closed lower bound does not
|
|
90
|
+
# prevent assigning probability to the outcome equalling lower_bound exactly —
|
|
91
|
+
# that mass goes in the first inbound bucket: cdf[1] - cdf[0].
|
|
89
92
|
upper_bound: float
|
|
90
93
|
lower_bound: float
|
|
91
94
|
zero_point: float | None
|
|
@@ -314,6 +317,22 @@ class NumericDistribution(BaseModel):
|
|
|
314
317
|
]
|
|
315
318
|
return representative_percentiles
|
|
316
319
|
|
|
320
|
+
def get_percentiles_at_target_heights(
|
|
321
|
+
self, target_heights: list[float] | None = None
|
|
322
|
+
) -> list[Percentile]:
|
|
323
|
+
if target_heights is None:
|
|
324
|
+
target_heights = [0.1, 0.2, 0.4, 0.6, 0.8, 0.9]
|
|
325
|
+
|
|
326
|
+
heights = [p.percentile for p in self.declared_percentiles]
|
|
327
|
+
values = [p.value for p in self.declared_percentiles]
|
|
328
|
+
|
|
329
|
+
result = []
|
|
330
|
+
for target in target_heights:
|
|
331
|
+
interpolated_value = float(np.interp(target, heights, values))
|
|
332
|
+
result.append(Percentile(percentile=target, value=interpolated_value))
|
|
333
|
+
|
|
334
|
+
return result
|
|
335
|
+
|
|
317
336
|
@property
|
|
318
337
|
@typing_extensions.deprecated(
|
|
319
338
|
"NumericDistribution.cdf (property) will be replaced with NumericDistribution.get_cdf (method). Please switch.",
|
|
@@ -328,12 +347,22 @@ class NumericDistribution(BaseModel):
|
|
|
328
347
|
between upper and lower bound (taking into account probability assigned above and below the bounds)
|
|
329
348
|
that is compatible with Metaculus questions.
|
|
330
349
|
|
|
331
|
-
cdf stands for '
|
|
350
|
+
cdf stands for 'cumulative distribution function'
|
|
332
351
|
|
|
333
352
|
At Metaculus CDFs are often represented with 201 points. Each point has:
|
|
334
|
-
- percentile (
|
|
353
|
+
- percentile (the y axis of the cdf graph — see boundary notes below)
|
|
335
354
|
- 'value' or 'nominal location' (The real world number that answers the question)
|
|
336
355
|
- cdf location (a number between 0 and 1 representing where the point is on the cdf x axis, where 0 is range min, and 1 is range max)
|
|
356
|
+
|
|
357
|
+
Important boundary semantics (note the asymmetry):
|
|
358
|
+
- cdf[0].percentile = P(outcome < lower_bound) — strictly less than (not equal to)
|
|
359
|
+
- cdf[-1].percentile = P(outcome <= upper_bound) — less than or equal to
|
|
360
|
+
For questions with a closed lower bound, cdf[0].percentile must be 0.0 because
|
|
361
|
+
outcomes strictly below the lower bound are impossible. This does NOT mean zero
|
|
362
|
+
probability at the lower bound itself — probability of the outcome equalling
|
|
363
|
+
lower_bound belongs in the first inbound bucket: cdf[1].percentile - cdf[0].percentile.
|
|
364
|
+
For example, to express an 80% chance the outcome is exactly lower_bound, set
|
|
365
|
+
cdf[1].percentile = 0.8 (and cdf[0].percentile = 0.0).
|
|
337
366
|
"""
|
|
338
367
|
|
|
339
368
|
cdf_size = self.cdf_size or NumericDefaults.DEFAULT_CDF_SIZE
|
|
@@ -511,6 +540,11 @@ class NumericDistribution(BaseModel):
|
|
|
511
540
|
- caps the maximum growth to 0.2
|
|
512
541
|
|
|
513
542
|
Note, thresholds change with different `inbound_outcome_count`s
|
|
543
|
+
|
|
544
|
+
Boundary convention: cdf[0] is P(outcome < lower_bound) — strictly less than.
|
|
545
|
+
For closed lower bounds this is forced to 0.0 after standardization; probability
|
|
546
|
+
of landing exactly on lower_bound belongs in the first inbound bucket (cdf[1]).
|
|
547
|
+
cdf[-1] is P(outcome <= upper_bound) — less than or equal to (standard convention).
|
|
514
548
|
"""
|
|
515
549
|
|
|
516
550
|
lower_open = self.open_lower_bound
|
|
@@ -633,12 +667,9 @@ class NumericReport(ForecastReport):
|
|
|
633
667
|
def make_readable_prediction(cls, prediction: NumericDistribution) -> str:
|
|
634
668
|
num_percentiles = len(prediction.declared_percentiles)
|
|
635
669
|
if num_percentiles > 10:
|
|
636
|
-
|
|
670
|
+
representative_percentiles = prediction.get_percentiles_at_target_heights()
|
|
637
671
|
else:
|
|
638
|
-
|
|
639
|
-
representative_percentiles = prediction.get_representative_percentiles(
|
|
640
|
-
num_display_percentiles
|
|
641
|
-
)
|
|
672
|
+
representative_percentiles = prediction.declared_percentiles
|
|
642
673
|
readable = "Probability distribution:\n"
|
|
643
674
|
for percentile in representative_percentiles:
|
|
644
675
|
if prediction.is_date:
|
{forecasting_tools-0.2.87 → forecasting_tools-0.2.89}/forecasting_tools/data_models/questions.py
RENAMED
|
@@ -474,8 +474,10 @@ class DateQuestion(MetaculusQuestion, BoundedQuestionMixin):
|
|
|
474
474
|
question_type: Literal["date"] = "date"
|
|
475
475
|
upper_bound: datetime
|
|
476
476
|
lower_bound: datetime
|
|
477
|
-
open_upper_bound: bool
|
|
478
|
-
open_lower_bound: bool
|
|
477
|
+
open_upper_bound: bool # If False, cdf[-1] must be 1.0 (no mass above upper bound)
|
|
478
|
+
open_lower_bound: bool # If False, cdf[0] must be 0.0. Note: cdf[0] = P(outcome <
|
|
479
|
+
# lower_bound) strictly, so probability of landing exactly on lower_bound belongs
|
|
480
|
+
# in the first inbound bucket (cdf[1]), not in cdf[0].
|
|
479
481
|
zero_point: float | None = None
|
|
480
482
|
cdf_size: int = 201
|
|
481
483
|
|
|
@@ -530,8 +532,10 @@ class NumericQuestion(MetaculusQuestion, BoundedQuestionMixin):
|
|
|
530
532
|
question_type: Literal["numeric"] = "numeric"
|
|
531
533
|
upper_bound: float
|
|
532
534
|
lower_bound: float
|
|
533
|
-
open_upper_bound: bool
|
|
534
|
-
open_lower_bound: bool
|
|
535
|
+
open_upper_bound: bool # If False, cdf[-1] must be 1.0 (no mass above upper bound)
|
|
536
|
+
open_lower_bound: bool # If False, cdf[0] must be 0.0. Note: cdf[0] = P(outcome <
|
|
537
|
+
# lower_bound) strictly, so probability of landing exactly on lower_bound belongs
|
|
538
|
+
# in the first inbound bucket (cdf[1]), not in cdf[0].
|
|
535
539
|
zero_point: float | None = None
|
|
536
540
|
cdf_size: int = (
|
|
537
541
|
201 # Normal numeric questions have 201 points, but discrete questions have fewer
|
|
@@ -12,7 +12,6 @@ from typing import Any, Coroutine, Literal, Sequence, TypeVar, cast, overload
|
|
|
12
12
|
from exceptiongroup import ExceptionGroup
|
|
13
13
|
from pydantic import BaseModel
|
|
14
14
|
|
|
15
|
-
from forecasting_tools.ai_models.agent_wrappers import general_trace_or_span
|
|
16
15
|
from forecasting_tools.ai_models.general_llm import GeneralLlm
|
|
17
16
|
from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
|
|
18
17
|
MonetaryCostManager,
|
|
@@ -338,20 +337,17 @@ class ForecastBot(ABC):
|
|
|
338
337
|
async def _run_individual_question_with_error_propagation(
|
|
339
338
|
self, question: MetaculusQuestion
|
|
340
339
|
) -> ForecastReport:
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
assert (
|
|
353
|
-
False
|
|
354
|
-
), "This is to satisfy type checker. The previous function should raise an exception"
|
|
340
|
+
try:
|
|
341
|
+
return await self._run_individual_question(question)
|
|
342
|
+
except Exception as e:
|
|
343
|
+
error_message = (
|
|
344
|
+
f"Error while processing question url: '{question.page_url}'"
|
|
345
|
+
)
|
|
346
|
+
logger.error(f"{error_message}: {e}")
|
|
347
|
+
self._reraise_exception_with_prepended_message(e, error_message)
|
|
348
|
+
assert (
|
|
349
|
+
False
|
|
350
|
+
), "This is to satisfy type checker. The previous function should raise an exception"
|
|
355
351
|
|
|
356
352
|
async def _run_individual_question(
|
|
357
353
|
self, question: MetaculusQuestion
|
|
@@ -43,7 +43,6 @@ from forecasting_tools.ai_models.agent_wrappers import (
|
|
|
43
43
|
AgentTool,
|
|
44
44
|
AiAgent,
|
|
45
45
|
event_to_tool_message,
|
|
46
|
-
general_trace_or_span,
|
|
47
46
|
)
|
|
48
47
|
from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
|
|
49
48
|
MonetaryCostManager,
|
|
@@ -84,7 +83,6 @@ class ChatSession(BaseModel, Jsonable):
|
|
|
84
83
|
name: str
|
|
85
84
|
messages: list[dict]
|
|
86
85
|
model_choice: str = DEFAULT_MODEL
|
|
87
|
-
trace_id: str | None = None
|
|
88
86
|
last_chat_cost: float | None = None
|
|
89
87
|
last_chat_duration: float | None = None
|
|
90
88
|
time_stamp: datetime = Field(default_factory=datetime.now)
|
|
@@ -235,11 +233,6 @@ class ChatPage(AppPage):
|
|
|
235
233
|
st.markdown(
|
|
236
234
|
f"**Last Chat Duration:** {st.session_state.last_chat_duration:.2f} seconds"
|
|
237
235
|
)
|
|
238
|
-
if "trace_id" in st.session_state.keys():
|
|
239
|
-
trace_id = st.session_state.trace_id
|
|
240
|
-
st.markdown(
|
|
241
|
-
f"**Conversation in Foresight Project:** [link](https://platform.openai.com/traces/trace?trace_id={trace_id})"
|
|
242
|
-
)
|
|
243
236
|
|
|
244
237
|
@classmethod
|
|
245
238
|
def display_premade_examples(cls) -> None:
|
|
@@ -265,8 +258,6 @@ class ChatPage(AppPage):
|
|
|
265
258
|
if st.button(session.name, key=session.name):
|
|
266
259
|
st.session_state.messages = session.messages
|
|
267
260
|
st.session_state.model_choice = session.model_choice
|
|
268
|
-
if session.trace_id:
|
|
269
|
-
st.session_state.trace_id = session.trace_id
|
|
270
261
|
if session.last_chat_cost:
|
|
271
262
|
st.session_state.last_chat_cost = session.last_chat_cost
|
|
272
263
|
if session.last_chat_duration:
|
|
@@ -299,7 +290,6 @@ class ChatPage(AppPage):
|
|
|
299
290
|
name=st.session_state["chat_save_name"],
|
|
300
291
|
model_choice=st.session_state["model_choice"],
|
|
301
292
|
messages=st.session_state.messages,
|
|
302
|
-
trace_id=st.session_state.trace_id,
|
|
303
293
|
last_chat_cost=st.session_state.last_chat_cost,
|
|
304
294
|
last_chat_duration=st.session_state.last_chat_duration,
|
|
305
295
|
)
|
|
@@ -464,29 +454,27 @@ class ChatPage(AppPage):
|
|
|
464
454
|
handoffs=[],
|
|
465
455
|
)
|
|
466
456
|
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
st.session_state.trace_id = chat_trace.trace_id
|
|
489
|
-
cls._update_last_message_if_gemini_bug(model_choice)
|
|
457
|
+
result = AgentRunner.run_streamed(
|
|
458
|
+
agent, st.session_state.messages, max_turns=20
|
|
459
|
+
)
|
|
460
|
+
streamed_text = ""
|
|
461
|
+
with st.chat_message("assistant"):
|
|
462
|
+
placeholder = st.empty()
|
|
463
|
+
with st.spinner("Thinking..."):
|
|
464
|
+
async for event in result.stream_events():
|
|
465
|
+
if event.type == "raw_response_event" and isinstance(
|
|
466
|
+
event.data, ResponseTextDeltaEvent
|
|
467
|
+
):
|
|
468
|
+
streamed_text += event.data.delta
|
|
469
|
+
placeholder.write(streamed_text)
|
|
470
|
+
|
|
471
|
+
new_reasoning = event_to_tool_message(event)
|
|
472
|
+
if new_reasoning:
|
|
473
|
+
st.sidebar.write(new_reasoning)
|
|
474
|
+
|
|
475
|
+
# logger.info(f"Chat finished with output: {streamed_text}")
|
|
476
|
+
st.session_state.messages = result.to_input_list()
|
|
477
|
+
cls._update_last_message_if_gemini_bug(model_choice)
|
|
490
478
|
|
|
491
479
|
ForecastDatabaseManager.add_general_report_to_database(
|
|
492
480
|
question_text=prompt_input,
|
|
@@ -519,7 +507,6 @@ class ChatPage(AppPage):
|
|
|
519
507
|
@classmethod
|
|
520
508
|
def clear_chat_history(cls) -> None:
|
|
521
509
|
st.session_state.messages = [cls.DEFAULT_MESSAGE]
|
|
522
|
-
st.session_state.trace_id = None
|
|
523
510
|
st.session_state.chat_files = []
|
|
524
511
|
|
|
525
512
|
|