forecasting-tools 0.2.87__tar.gz → 0.2.88__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/PKG-INFO +1 -1
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/other/data_analyzer.py +4 -6
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/research/key_factors_researcher.py +1 -1
- forecasting_tools-0.2.88/forecasting_tools/ai_models/agent_wrappers.py +75 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/general_llm.py +1 -11
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/prompt_optimizer.py +41 -52
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/cp_benchmarking/benchmarker.py +36 -43
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/numeric_report.py +18 -5
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/forecast_bot.py +11 -15
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/chat_page.py +21 -34
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/pyproject.toml +1 -1
- forecasting_tools-0.2.87/forecasting_tools/ai_models/agent_wrappers.py +0 -212
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/LICENSE +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/README.md +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/ai_congress_v2/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_member_agent.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/ai_congress_v2/congress_orchestrator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/ai_congress_v2/data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/ai_congress_v2/member_profiles.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/ai_congress_v2/tools.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/base_rates/base_rate_researcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/base_rates/deduplicator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/base_rates/estimator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/base_rates/niche_list_researcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/deprecated/configured_llms.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/deprecated/general_researcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/deprecated/question_generator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/deprecated/question_responder.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/deprecated/question_router.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/deprecated/research_coordinator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/minor_tools.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/other/hosted_file.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/generated_question.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/harmful_question_identifier.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/q3_q4_quarterly_questions.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/question_decomposer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/question_operationalizer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/simple_question.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/question_generators/topic_generator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/research/computer_use.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/research/find_a_dataset.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/research/smart_searcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/agent_runner.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/effect_engine.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/industrial_basin.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/milbrook.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_business.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_dnd.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/simple_election.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/willowbrook.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/example_situations/z_broken_mafia.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/forecast_resolver.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_policy_agent.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_runner.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/intervention_testing/intervention_storage.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/simulator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/agents_and_tools/situation_simulator/situation_generator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/ai_utils/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/ai_utils/openai_utils.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/ai_utils/response_types.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/README.md +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/claude35sonnet.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/deepseek_r1.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/gpt4o.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/gpt4ovision.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/gpto1.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/gpto1preview.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/metaculus4o.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/deprecated_model_classes/perplexity.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/exa_searcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/ai_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/combined_llm_archetype.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/incurs_cost.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/named_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/outputs_text.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/priced_per_request.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/request_limited_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/retryable_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/time_limited_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/token_limited_model.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/tokens_are_calculatable.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_interfaces/tokens_incur_cost.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/model_tracker.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/resource_managers/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/resource_managers/hard_limit_manager.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/resource_managers/monetary_cost_manager.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/resource_managers/refreshing_bucket_rate_limiter.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/bot_evaluator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/bot_optimizer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/control_prompt.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/customizable_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/prompt_data_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/auto_optimizers/question_plus_research.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/cp_benchmarking/benchmark_displayer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/cp_benchmarking/benchmark_for_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/binary_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/coherence_link.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/conditional_models.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/conditional_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/data_organizer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/forecast_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/markdown_tree.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/multiple_choice_report.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/questions.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/timestamped_predictions.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/data_models/user_response.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/bot_lists.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q1_veritas_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q1t_w_convert_to_binary.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q1t_w_personas_and_exa.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q2t_w_decomposition.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q3t_w_asknews.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q3t_w_exa.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q3t_w_q4vbinary.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q4_veritas_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q4v_w_exa.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_dseekr1.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/experiments/q4v_w_exa_and_o1_preview.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/main_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/gpt_4_1_optimized_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/q1_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/q2_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/q3_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/q4_template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/research_only_bot_2025_fall.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/template_bot_2025_fall.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/template_bot_2026_spring.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/official_bots/uniform_probability_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/forecast_bots/template_bot.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/Home.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/base_rate_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/benchmark_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/congress_v2_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/csv_agent.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/estimator_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/forecaster_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/intervention_leaderboard_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/key_factors_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/niche_list_researcher_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/app_pages/simulator_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/base_rate_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/congress_page_example.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/congress_v2_page_example.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/estimator_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/forecast_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/key_factors_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/niche_list_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/example_outputs/question_generator_page_examples.json +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/helpers/app_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/helpers/custom_auth.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/helpers/report_displayer.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/front_end/helpers/tool_page.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/asknews_cache.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/asknews_searcher.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/forecast_database_manager.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/metaculus_api.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/metaculus_client.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/prediction_extractor.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/structure_output.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/helpers/works_cited_creator.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/__init__.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/async_batching.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/coda_utils.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/custom_logger.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/file_manipulation.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/jsonable.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/misc.py +0 -0
- {forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/util/stats.py +0 -0
|
@@ -12,7 +12,6 @@ from forecasting_tools.ai_models.agent_wrappers import (
|
|
|
12
12
|
CodingTool,
|
|
13
13
|
agent_tool,
|
|
14
14
|
event_to_tool_message,
|
|
15
|
-
general_trace_or_span,
|
|
16
15
|
)
|
|
17
16
|
from forecasting_tools.util.misc import clean_indents
|
|
18
17
|
|
|
@@ -119,10 +118,9 @@ class DataAnalyzer:
|
|
|
119
118
|
|
|
120
119
|
|
|
121
120
|
if __name__ == "__main__":
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
instructions="Please multiple 52.675 x 6.547 x 9867.5476 x 4356.5",
|
|
126
|
-
)
|
|
121
|
+
answer = asyncio.run(
|
|
122
|
+
DataAnalyzer().run_data_analysis(
|
|
123
|
+
instructions="Please multiple 52.675 x 6.547 x 9867.5476 x 4356.5",
|
|
127
124
|
)
|
|
125
|
+
)
|
|
128
126
|
print(answer)
|
|
@@ -222,7 +222,7 @@ class KeyFactorsResearcher:
|
|
|
222
222
|
# Grading Scale for {ScoreCardGrade.__class__.__name__}
|
|
223
223
|
- {ScoreCardGrade.VERY_BAD.value}: Generally poor quality
|
|
224
224
|
- {ScoreCardGrade.BAD.value}: Below average quality
|
|
225
|
-
- {ScoreCardGrade.OK.value}:
|
|
225
|
+
- {ScoreCardGrade.OK.value}: Average quality
|
|
226
226
|
- {ScoreCardGrade.GOOD.value}: Above average quality
|
|
227
227
|
- {ScoreCardGrade.VERY_GOOD.value}: Exceptional quality
|
|
228
228
|
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import logging
|
|
3
|
+
|
|
4
|
+
import nest_asyncio
|
|
5
|
+
from agents import Agent, CodeInterpreterTool, FunctionTool, Runner
|
|
6
|
+
from agents import function_tool as ft
|
|
7
|
+
from agents.extensions.models.litellm_model import LitellmModel
|
|
8
|
+
from agents.stream_events import StreamEvent
|
|
9
|
+
|
|
10
|
+
from forecasting_tools.ai_models.model_tracker import ModelTracker
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
nest_asyncio.apply()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class AgentSdkLlm(LitellmModel):
|
|
19
|
+
"""
|
|
20
|
+
Wrapper around openai-agent-sdk's LiteLlm Model for later extension
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
async def get_response(self, *args, **kwargs): # NOSONAR
|
|
24
|
+
ModelTracker.give_cost_tracking_warning_if_needed(self.model)
|
|
25
|
+
response = await super().get_response(*args, **kwargs)
|
|
26
|
+
await asyncio.sleep(
|
|
27
|
+
0.0001
|
|
28
|
+
) # For whatever reason, it seems you need to await a coroutine to get the litellm cost callback to work
|
|
29
|
+
return response
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
AgentRunner = Runner # Alias for Runner for later extension
|
|
33
|
+
AgentTool = FunctionTool # Alias for FunctionTool for later extension
|
|
34
|
+
AiAgent = Agent # Alias for Agent for later extension
|
|
35
|
+
CodingTool = CodeInterpreterTool # Alias for CodeInterpreterTool for later extension
|
|
36
|
+
agent_tool = ft # Alias for function_tool for later extension
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def event_to_tool_message(event: StreamEvent) -> str | None:
|
|
40
|
+
text = ""
|
|
41
|
+
if event.type == "run_item_stream_event":
|
|
42
|
+
item = event.item
|
|
43
|
+
if item.type == "message_output_item":
|
|
44
|
+
content = item.raw_item.content[0]
|
|
45
|
+
if content.type == "output_text":
|
|
46
|
+
# text = content.text
|
|
47
|
+
text = "" # the text is already streamed separate from this function
|
|
48
|
+
elif content.type == "output_refusal":
|
|
49
|
+
text = content.refusal
|
|
50
|
+
else:
|
|
51
|
+
text = "Error: unknown content type"
|
|
52
|
+
elif item.type == "tool_call_item":
|
|
53
|
+
if item.raw_item.type == "code_interpreter_call":
|
|
54
|
+
text = (
|
|
55
|
+
f"\nCode interpreter code:\n```python\n{item.raw_item.code}\n```\n"
|
|
56
|
+
)
|
|
57
|
+
else:
|
|
58
|
+
tool_name = getattr(item.raw_item, "name", "unknown_tool")
|
|
59
|
+
tool_args = getattr(item.raw_item, "arguments", {})
|
|
60
|
+
text = f"Tool call: {tool_name}({tool_args})"
|
|
61
|
+
elif item.type == "tool_call_output_item":
|
|
62
|
+
output = getattr(item, "output", str(item.raw_item))
|
|
63
|
+
text = f"Tool output:\n\n{output}"
|
|
64
|
+
elif item.type == "handoff_call_item":
|
|
65
|
+
handoff_info = getattr(item.raw_item, "name", "handoff")
|
|
66
|
+
text = f"Handoff call: {handoff_info}"
|
|
67
|
+
elif item.type == "handoff_output_item":
|
|
68
|
+
text = f"Handoff output: {str(item.raw_item)}"
|
|
69
|
+
elif item.type == "reasoning_item":
|
|
70
|
+
text = f"Reasoning: {str(item.raw_item)}"
|
|
71
|
+
# elif event.type == "agent_updated_stream_event":
|
|
72
|
+
# text += f"Agent updated: {event.new_agent.name}\n\n"
|
|
73
|
+
if text == "":
|
|
74
|
+
return None
|
|
75
|
+
return text
|
{forecasting_tools-0.2.87 → forecasting_tools-0.2.88}/forecasting_tools/ai_models/general_llm.py
RENAMED
|
@@ -21,7 +21,6 @@ from openai.types.responses import (
|
|
|
21
21
|
ResponseReasoningItem,
|
|
22
22
|
)
|
|
23
23
|
|
|
24
|
-
from forecasting_tools.ai_models.agent_wrappers import track_generation
|
|
25
24
|
from forecasting_tools.ai_models.ai_utils.openai_utils import (
|
|
26
25
|
OpenAiUtils,
|
|
27
26
|
VisionMessageData,
|
|
@@ -251,16 +250,7 @@ class GeneralLlm(
|
|
|
251
250
|
) -> Any:
|
|
252
251
|
logger.debug(f"Invoking model with prompt: {prompt}")
|
|
253
252
|
|
|
254
|
-
|
|
255
|
-
input=self.model_input_to_message(prompt),
|
|
256
|
-
model=self.model,
|
|
257
|
-
) as span:
|
|
258
|
-
direct_call_response = await self._mockable_direct_call_to_model(prompt)
|
|
259
|
-
answer = direct_call_response.data
|
|
260
|
-
span.span_data.output = [{"role": "assistant", "content": answer}]
|
|
261
|
-
# span.span_data.usage = usage.model_dump()
|
|
262
|
-
span.span_data.model = self.model
|
|
263
|
-
span.span_data.model_config = self.litellm_kwargs
|
|
253
|
+
direct_call_response = await self._mockable_direct_call_to_model(prompt)
|
|
264
254
|
|
|
265
255
|
logger.debug(f"Model responded with: {direct_call_response}")
|
|
266
256
|
return direct_call_response
|
|
@@ -9,12 +9,7 @@ from pydantic import BaseModel, Field
|
|
|
9
9
|
from forecasting_tools.agents_and_tools.minor_tools import (
|
|
10
10
|
perplexity_reasoning_pro_search,
|
|
11
11
|
)
|
|
12
|
-
from forecasting_tools.ai_models.agent_wrappers import
|
|
13
|
-
AgentRunner,
|
|
14
|
-
AgentSdkLlm,
|
|
15
|
-
AiAgent,
|
|
16
|
-
general_trace_or_span,
|
|
17
|
-
)
|
|
12
|
+
from forecasting_tools.ai_models.agent_wrappers import AgentRunner, AgentSdkLlm, AiAgent
|
|
18
13
|
from forecasting_tools.auto_optimizers.prompt_data_models import PromptIdea
|
|
19
14
|
from forecasting_tools.helpers.structure_output import structure_output
|
|
20
15
|
from forecasting_tools.util.misc import clean_indents, retry_async_function
|
|
@@ -112,63 +107,57 @@ class PromptOptimizer:
|
|
|
112
107
|
)
|
|
113
108
|
|
|
114
109
|
async def create_optimized_prompt(self) -> OptimizationRun:
|
|
115
|
-
|
|
116
|
-
return await self._create_optimized_prompt()
|
|
110
|
+
return await self._create_optimized_prompt()
|
|
117
111
|
|
|
118
112
|
async def _create_optimized_prompt(self) -> OptimizationRun:
|
|
119
113
|
iteration_num = 0
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
prompts_still_needed,
|
|
141
|
-
)
|
|
142
|
-
starting_prompts.extend(additional_initial_prompts)
|
|
143
|
-
|
|
144
|
-
offspring_prompts: list[ImplementedPrompt] = starting_prompts
|
|
145
|
-
assert (
|
|
146
|
-
seed_prompt in offspring_prompts
|
|
147
|
-
), "Seed prompt not found in offspring prompts"
|
|
148
|
-
all_evaluated_prompts: list[ScoredPrompt] = (
|
|
149
|
-
await self._evaluate_new_members(offspring_prompts)
|
|
114
|
+
iteration_num += 1
|
|
115
|
+
logger.info(
|
|
116
|
+
f"Generating initial prompt population of size {self.initial_prompt_population_size}"
|
|
117
|
+
)
|
|
118
|
+
seed_prompt = ImplementedPrompt(
|
|
119
|
+
text=self.initial_prompt,
|
|
120
|
+
idea=PromptIdea(
|
|
121
|
+
short_name="Initial Seed",
|
|
122
|
+
full_text="The user-provided initial prompt",
|
|
123
|
+
),
|
|
124
|
+
originating_ideas=[],
|
|
125
|
+
)
|
|
126
|
+
starting_prompts: list[ImplementedPrompt] = [seed_prompt]
|
|
127
|
+
prompts_still_needed = self.initial_prompt_population_size - len(
|
|
128
|
+
starting_prompts
|
|
129
|
+
)
|
|
130
|
+
if prompts_still_needed > 0:
|
|
131
|
+
additional_initial_prompts = await self._mutate_prompt(
|
|
132
|
+
starting_prompts[0],
|
|
133
|
+
prompts_still_needed,
|
|
150
134
|
)
|
|
151
|
-
|
|
135
|
+
starting_prompts.extend(additional_initial_prompts)
|
|
136
|
+
|
|
137
|
+
offspring_prompts: list[ImplementedPrompt] = starting_prompts
|
|
138
|
+
assert (
|
|
139
|
+
seed_prompt in offspring_prompts
|
|
140
|
+
), "Seed prompt not found in offspring prompts"
|
|
141
|
+
all_evaluated_prompts: list[ScoredPrompt] = await self._evaluate_new_members(
|
|
142
|
+
offspring_prompts
|
|
143
|
+
)
|
|
144
|
+
survivors = await self._kill_the_weak(all_evaluated_prompts)
|
|
152
145
|
|
|
153
146
|
while iteration_num < self.iterations:
|
|
154
147
|
iteration_num += 1
|
|
155
|
-
|
|
156
|
-
f"
|
|
157
|
-
|
|
158
|
-
):
|
|
159
|
-
logger.info(
|
|
160
|
-
f"Starting iteration {iteration_num + 1}/{self.iterations} - Current population size: {len(offspring_prompts)}"
|
|
161
|
-
)
|
|
148
|
+
logger.info(
|
|
149
|
+
f"Starting iteration {iteration_num + 1}/{self.iterations} - Current population size: {len(offspring_prompts)}"
|
|
150
|
+
)
|
|
162
151
|
|
|
163
|
-
|
|
152
|
+
offspring_prompts = await self._generate_new_prompts(survivors)
|
|
164
153
|
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
154
|
+
evaluated_prompts = await self._evaluate_new_members(offspring_prompts)
|
|
155
|
+
all_evaluated_prompts.extend(evaluated_prompts)
|
|
156
|
+
updated_population = survivors + evaluated_prompts
|
|
168
157
|
|
|
169
|
-
|
|
158
|
+
survivors = await self._kill_the_weak(updated_population)
|
|
170
159
|
|
|
171
|
-
|
|
160
|
+
self._log_duplicate_prompts(all_evaluated_prompts)
|
|
172
161
|
|
|
173
162
|
return OptimizationRun(scored_prompts=all_evaluated_prompts)
|
|
174
163
|
|
|
@@ -5,7 +5,6 @@ from typing import Sequence
|
|
|
5
5
|
|
|
6
6
|
import typeguard
|
|
7
7
|
|
|
8
|
-
from forecasting_tools.ai_models.agent_wrappers import general_trace_or_span
|
|
9
8
|
from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
|
|
10
9
|
MonetaryCostManager,
|
|
11
10
|
)
|
|
@@ -80,50 +79,44 @@ class Benchmarker:
|
|
|
80
79
|
self.code_to_snapshot = additional_code_to_snapshot
|
|
81
80
|
|
|
82
81
|
async def run_benchmark(self) -> list[BenchmarkForBot]:
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
self.number_of_questions_to_use,
|
|
90
|
-
)
|
|
91
|
-
else:
|
|
92
|
-
chosen_questions = self.questions_to_use
|
|
93
|
-
|
|
94
|
-
chosen_questions = typeguard.check_type(
|
|
95
|
-
chosen_questions, list[MetaculusQuestion]
|
|
96
|
-
)
|
|
97
|
-
|
|
98
|
-
if self.number_of_questions_to_use is not None:
|
|
99
|
-
assert len(chosen_questions) == self.number_of_questions_to_use
|
|
100
|
-
|
|
101
|
-
benchmarks: list[BenchmarkForBot] = self._initialize_benchmarks(
|
|
102
|
-
self.forecast_bots, chosen_questions
|
|
82
|
+
if self.questions_to_use is None:
|
|
83
|
+
assert (
|
|
84
|
+
self.number_of_questions_to_use is not None
|
|
85
|
+
), "number_of_questions_to_use must be provided if questions_to_use is not provided"
|
|
86
|
+
chosen_questions = MetaculusApi.get_benchmark_questions(
|
|
87
|
+
self.number_of_questions_to_use,
|
|
103
88
|
)
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
89
|
+
else:
|
|
90
|
+
chosen_questions = self.questions_to_use
|
|
91
|
+
|
|
92
|
+
chosen_questions = typeguard.check_type(
|
|
93
|
+
chosen_questions, list[MetaculusQuestion]
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
if self.number_of_questions_to_use is not None:
|
|
97
|
+
assert len(chosen_questions) == self.number_of_questions_to_use
|
|
98
|
+
|
|
99
|
+
benchmarks: list[BenchmarkForBot] = self._initialize_benchmarks(
|
|
100
|
+
self.forecast_bots, chosen_questions
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
batches = self._batch_questions(
|
|
104
|
+
self.forecast_bots,
|
|
105
|
+
benchmarks,
|
|
106
|
+
chosen_questions,
|
|
107
|
+
self.concurrent_question_batch_size,
|
|
108
|
+
)
|
|
109
|
+
try:
|
|
110
|
+
for i, batch in enumerate(batches):
|
|
111
|
+
await self._run_a_batch(batch)
|
|
112
|
+
if batch.is_last_batch_for_benchmark:
|
|
113
|
+
self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
|
|
114
|
+
except KeyboardInterrupt:
|
|
115
|
+
logger.warning(
|
|
116
|
+
"KeyboardInterrupt detected, saving current benchmark progress."
|
|
110
117
|
)
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
with general_trace_or_span(
|
|
114
|
-
f"{batch.benchmark.name} - Batch {i+1} of {len(batches)}"
|
|
115
|
-
):
|
|
116
|
-
await self._run_a_batch(batch)
|
|
117
|
-
if batch.is_last_batch_for_benchmark:
|
|
118
|
-
self._append_benchmarks_to_jsonl_if_configured(
|
|
119
|
-
[batch.benchmark]
|
|
120
|
-
)
|
|
121
|
-
except KeyboardInterrupt:
|
|
122
|
-
logger.warning(
|
|
123
|
-
"KeyboardInterrupt detected, saving current benchmark progress."
|
|
124
|
-
)
|
|
125
|
-
self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
|
|
126
|
-
raise
|
|
118
|
+
self._append_benchmarks_to_jsonl_if_configured([batch.benchmark])
|
|
119
|
+
raise
|
|
127
120
|
return benchmarks
|
|
128
121
|
|
|
129
122
|
async def _run_a_batch(self, batch: QuestionBatch) -> None:
|
|
@@ -314,6 +314,22 @@ class NumericDistribution(BaseModel):
|
|
|
314
314
|
]
|
|
315
315
|
return representative_percentiles
|
|
316
316
|
|
|
317
|
+
def get_percentiles_at_target_heights(
|
|
318
|
+
self, target_heights: list[float] | None = None
|
|
319
|
+
) -> list[Percentile]:
|
|
320
|
+
if target_heights is None:
|
|
321
|
+
target_heights = [0.1, 0.2, 0.4, 0.6, 0.8, 0.9]
|
|
322
|
+
|
|
323
|
+
heights = [p.percentile for p in self.declared_percentiles]
|
|
324
|
+
values = [p.value for p in self.declared_percentiles]
|
|
325
|
+
|
|
326
|
+
result = []
|
|
327
|
+
for target in target_heights:
|
|
328
|
+
interpolated_value = float(np.interp(target, heights, values))
|
|
329
|
+
result.append(Percentile(percentile=target, value=interpolated_value))
|
|
330
|
+
|
|
331
|
+
return result
|
|
332
|
+
|
|
317
333
|
@property
|
|
318
334
|
@typing_extensions.deprecated(
|
|
319
335
|
"NumericDistribution.cdf (property) will be replaced with NumericDistribution.get_cdf (method). Please switch.",
|
|
@@ -633,12 +649,9 @@ class NumericReport(ForecastReport):
|
|
|
633
649
|
def make_readable_prediction(cls, prediction: NumericDistribution) -> str:
|
|
634
650
|
num_percentiles = len(prediction.declared_percentiles)
|
|
635
651
|
if num_percentiles > 10:
|
|
636
|
-
|
|
652
|
+
representative_percentiles = prediction.get_percentiles_at_target_heights()
|
|
637
653
|
else:
|
|
638
|
-
|
|
639
|
-
representative_percentiles = prediction.get_representative_percentiles(
|
|
640
|
-
num_display_percentiles
|
|
641
|
-
)
|
|
654
|
+
representative_percentiles = prediction.declared_percentiles
|
|
642
655
|
readable = "Probability distribution:\n"
|
|
643
656
|
for percentile in representative_percentiles:
|
|
644
657
|
if prediction.is_date:
|
|
@@ -12,7 +12,6 @@ from typing import Any, Coroutine, Literal, Sequence, TypeVar, cast, overload
|
|
|
12
12
|
from exceptiongroup import ExceptionGroup
|
|
13
13
|
from pydantic import BaseModel
|
|
14
14
|
|
|
15
|
-
from forecasting_tools.ai_models.agent_wrappers import general_trace_or_span
|
|
16
15
|
from forecasting_tools.ai_models.general_llm import GeneralLlm
|
|
17
16
|
from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
|
|
18
17
|
MonetaryCostManager,
|
|
@@ -338,20 +337,17 @@ class ForecastBot(ABC):
|
|
|
338
337
|
async def _run_individual_question_with_error_propagation(
|
|
339
338
|
self, question: MetaculusQuestion
|
|
340
339
|
) -> ForecastReport:
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
assert (
|
|
353
|
-
False
|
|
354
|
-
), "This is to satisfy type checker. The previous function should raise an exception"
|
|
340
|
+
try:
|
|
341
|
+
return await self._run_individual_question(question)
|
|
342
|
+
except Exception as e:
|
|
343
|
+
error_message = (
|
|
344
|
+
f"Error while processing question url: '{question.page_url}'"
|
|
345
|
+
)
|
|
346
|
+
logger.error(f"{error_message}: {e}")
|
|
347
|
+
self._reraise_exception_with_prepended_message(e, error_message)
|
|
348
|
+
assert (
|
|
349
|
+
False
|
|
350
|
+
), "This is to satisfy type checker. The previous function should raise an exception"
|
|
355
351
|
|
|
356
352
|
async def _run_individual_question(
|
|
357
353
|
self, question: MetaculusQuestion
|
|
@@ -43,7 +43,6 @@ from forecasting_tools.ai_models.agent_wrappers import (
|
|
|
43
43
|
AgentTool,
|
|
44
44
|
AiAgent,
|
|
45
45
|
event_to_tool_message,
|
|
46
|
-
general_trace_or_span,
|
|
47
46
|
)
|
|
48
47
|
from forecasting_tools.ai_models.resource_managers.monetary_cost_manager import (
|
|
49
48
|
MonetaryCostManager,
|
|
@@ -84,7 +83,6 @@ class ChatSession(BaseModel, Jsonable):
|
|
|
84
83
|
name: str
|
|
85
84
|
messages: list[dict]
|
|
86
85
|
model_choice: str = DEFAULT_MODEL
|
|
87
|
-
trace_id: str | None = None
|
|
88
86
|
last_chat_cost: float | None = None
|
|
89
87
|
last_chat_duration: float | None = None
|
|
90
88
|
time_stamp: datetime = Field(default_factory=datetime.now)
|
|
@@ -235,11 +233,6 @@ class ChatPage(AppPage):
|
|
|
235
233
|
st.markdown(
|
|
236
234
|
f"**Last Chat Duration:** {st.session_state.last_chat_duration:.2f} seconds"
|
|
237
235
|
)
|
|
238
|
-
if "trace_id" in st.session_state.keys():
|
|
239
|
-
trace_id = st.session_state.trace_id
|
|
240
|
-
st.markdown(
|
|
241
|
-
f"**Conversation in Foresight Project:** [link](https://platform.openai.com/traces/trace?trace_id={trace_id})"
|
|
242
|
-
)
|
|
243
236
|
|
|
244
237
|
@classmethod
|
|
245
238
|
def display_premade_examples(cls) -> None:
|
|
@@ -265,8 +258,6 @@ class ChatPage(AppPage):
|
|
|
265
258
|
if st.button(session.name, key=session.name):
|
|
266
259
|
st.session_state.messages = session.messages
|
|
267
260
|
st.session_state.model_choice = session.model_choice
|
|
268
|
-
if session.trace_id:
|
|
269
|
-
st.session_state.trace_id = session.trace_id
|
|
270
261
|
if session.last_chat_cost:
|
|
271
262
|
st.session_state.last_chat_cost = session.last_chat_cost
|
|
272
263
|
if session.last_chat_duration:
|
|
@@ -299,7 +290,6 @@ class ChatPage(AppPage):
|
|
|
299
290
|
name=st.session_state["chat_save_name"],
|
|
300
291
|
model_choice=st.session_state["model_choice"],
|
|
301
292
|
messages=st.session_state.messages,
|
|
302
|
-
trace_id=st.session_state.trace_id,
|
|
303
293
|
last_chat_cost=st.session_state.last_chat_cost,
|
|
304
294
|
last_chat_duration=st.session_state.last_chat_duration,
|
|
305
295
|
)
|
|
@@ -464,29 +454,27 @@ class ChatPage(AppPage):
|
|
|
464
454
|
handoffs=[],
|
|
465
455
|
)
|
|
466
456
|
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
st.session_state.trace_id = chat_trace.trace_id
|
|
489
|
-
cls._update_last_message_if_gemini_bug(model_choice)
|
|
457
|
+
result = AgentRunner.run_streamed(
|
|
458
|
+
agent, st.session_state.messages, max_turns=20
|
|
459
|
+
)
|
|
460
|
+
streamed_text = ""
|
|
461
|
+
with st.chat_message("assistant"):
|
|
462
|
+
placeholder = st.empty()
|
|
463
|
+
with st.spinner("Thinking..."):
|
|
464
|
+
async for event in result.stream_events():
|
|
465
|
+
if event.type == "raw_response_event" and isinstance(
|
|
466
|
+
event.data, ResponseTextDeltaEvent
|
|
467
|
+
):
|
|
468
|
+
streamed_text += event.data.delta
|
|
469
|
+
placeholder.write(streamed_text)
|
|
470
|
+
|
|
471
|
+
new_reasoning = event_to_tool_message(event)
|
|
472
|
+
if new_reasoning:
|
|
473
|
+
st.sidebar.write(new_reasoning)
|
|
474
|
+
|
|
475
|
+
# logger.info(f"Chat finished with output: {streamed_text}")
|
|
476
|
+
st.session_state.messages = result.to_input_list()
|
|
477
|
+
cls._update_last_message_if_gemini_bug(model_choice)
|
|
490
478
|
|
|
491
479
|
ForecastDatabaseManager.add_general_report_to_database(
|
|
492
480
|
question_text=prompt_input,
|
|
@@ -519,7 +507,6 @@ class ChatPage(AppPage):
|
|
|
519
507
|
@classmethod
|
|
520
508
|
def clear_chat_history(cls) -> None:
|
|
521
509
|
st.session_state.messages = [cls.DEFAULT_MESSAGE]
|
|
522
|
-
st.session_state.trace_id = None
|
|
523
510
|
st.session_state.chat_files = []
|
|
524
511
|
|
|
525
512
|
|