agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,1863 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from typing import Any, Callable, Iterable, List, Mapping, Optional, Sequence
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
from ..base.base_optimizer import BaseOptimizer
|
|
10
|
+
from ..components import (
|
|
11
|
+
ComponentDiagnosis,
|
|
12
|
+
diagnose_agent_report_evaluation,
|
|
13
|
+
diagnose_text,
|
|
14
|
+
relevant_search_paths,
|
|
15
|
+
)
|
|
16
|
+
from ..deployment import (
|
|
17
|
+
AgentDeploymentExport,
|
|
18
|
+
AgentPromotionCheck,
|
|
19
|
+
AgentRollbackDecision,
|
|
20
|
+
check_agent_deployment_rollback,
|
|
21
|
+
export_agent_deployment,
|
|
22
|
+
)
|
|
23
|
+
from ..targets import AgentCandidate, CandidateEvaluation, OptimizationTarget
|
|
24
|
+
from ..types import EvaluationResult, OptimizationResult
|
|
25
|
+
from .agent import AgentOptimizer, _dedupe_diagnoses, _normalize_diagnoses
|
|
26
|
+
from .agent_bandit import AgentBanditOptimizer
|
|
27
|
+
from .agent_curriculum import AgentCurriculumOptimizer
|
|
28
|
+
from .agent_evolution import AgentEvolutionOptimizer
|
|
29
|
+
from .agent_pareto import AgentParetoOptimizer
|
|
30
|
+
from .agent_social_memory import AgentSocialMemoryOptimizer
|
|
31
|
+
from .agent_tpe import AgentTPEOptimizer
|
|
32
|
+
from .council import CouncilAgentOptimizer, SocietyAgentOptimizer
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
FEEDBACK_SCHEMA_VERSION = "agent-opt.feedback.v1"
|
|
36
|
+
MULTI_INTERACTION_SCHEMA_VERSION = "agent-opt.multi-interaction.v1"
|
|
37
|
+
DEFAULT_MULTI_INTERACTION_BACKENDS = (
|
|
38
|
+
"curriculum",
|
|
39
|
+
"council",
|
|
40
|
+
"society",
|
|
41
|
+
"social_memory",
|
|
42
|
+
"evolution",
|
|
43
|
+
"pareto",
|
|
44
|
+
"tpe",
|
|
45
|
+
"bandit",
|
|
46
|
+
"agent",
|
|
47
|
+
)
|
|
48
|
+
MULTI_INTERACTION_BACKEND_PROFILES: dict[str, dict[str, Any]] = {
|
|
49
|
+
"society": {
|
|
50
|
+
"allocation_kind": "role_graph_society_search",
|
|
51
|
+
"roles": (
|
|
52
|
+
"sutradhara",
|
|
53
|
+
"smriti",
|
|
54
|
+
"arjuna",
|
|
55
|
+
"hanuman",
|
|
56
|
+
"vidura",
|
|
57
|
+
"krishna",
|
|
58
|
+
"sangha",
|
|
59
|
+
"dharma_steward",
|
|
60
|
+
),
|
|
61
|
+
"role_archetypes": (
|
|
62
|
+
"orchestrator",
|
|
63
|
+
"working_memory",
|
|
64
|
+
"focused_action",
|
|
65
|
+
"bridge_builder",
|
|
66
|
+
"prudent_critic",
|
|
67
|
+
"charioteer_counsel",
|
|
68
|
+
"collective_synthesis",
|
|
69
|
+
"minimal_process_guardian",
|
|
70
|
+
),
|
|
71
|
+
"path_prefixes": (
|
|
72
|
+
"multi_agent",
|
|
73
|
+
"memory",
|
|
74
|
+
"policy",
|
|
75
|
+
"security",
|
|
76
|
+
"orchestration",
|
|
77
|
+
"framework",
|
|
78
|
+
),
|
|
79
|
+
"role_path_prefixes": {
|
|
80
|
+
"sutradhara": ("multi_agent", "orchestration", "framework"),
|
|
81
|
+
"smriti": (
|
|
82
|
+
"memory",
|
|
83
|
+
"framework.memory",
|
|
84
|
+
"framework.checkpoints",
|
|
85
|
+
"framework.sessions",
|
|
86
|
+
),
|
|
87
|
+
"arjuna": ("tools", "action", "policy"),
|
|
88
|
+
"hanuman": ("multi_agent", "framework", "orchestration"),
|
|
89
|
+
"vidura": ("policy", "security", "adversarial"),
|
|
90
|
+
"krishna": ("multi_agent", "memory", "policy"),
|
|
91
|
+
"sangha": (),
|
|
92
|
+
"dharma_steward": ("policy", "security", "reliability"),
|
|
93
|
+
},
|
|
94
|
+
},
|
|
95
|
+
"council": {
|
|
96
|
+
"allocation_kind": "council_deliberation",
|
|
97
|
+
"roles": ("explorer", "critic", "synthesizer", "steward"),
|
|
98
|
+
"role_archetypes": (
|
|
99
|
+
"exploration",
|
|
100
|
+
"critique",
|
|
101
|
+
"synthesis",
|
|
102
|
+
"process_guardian",
|
|
103
|
+
),
|
|
104
|
+
"path_prefixes": ("multi_agent", "memory", "policy", "tools", "framework"),
|
|
105
|
+
"role_path_prefixes": {
|
|
106
|
+
"explorer": (),
|
|
107
|
+
"critic": ("policy", "security", "adversarial"),
|
|
108
|
+
"synthesizer": (),
|
|
109
|
+
"steward": ("policy", "reliability", "framework"),
|
|
110
|
+
},
|
|
111
|
+
},
|
|
112
|
+
"social_memory": {
|
|
113
|
+
"allocation_kind": "social_memory_credit_ledger",
|
|
114
|
+
"roles": ("smriti", "arjuna", "vidura", "sangha", "dharma_steward"),
|
|
115
|
+
"role_archetypes": (
|
|
116
|
+
"working_memory",
|
|
117
|
+
"focused_action",
|
|
118
|
+
"prudent_critic",
|
|
119
|
+
"collective_synthesis",
|
|
120
|
+
"minimal_process_guardian",
|
|
121
|
+
),
|
|
122
|
+
"path_prefixes": ("memory", "multi_agent", "policy", "framework"),
|
|
123
|
+
"role_path_prefixes": {
|
|
124
|
+
"smriti": (
|
|
125
|
+
"memory",
|
|
126
|
+
"framework.memory",
|
|
127
|
+
"framework.checkpoints",
|
|
128
|
+
"framework.sessions",
|
|
129
|
+
),
|
|
130
|
+
"arjuna": ("multi_agent", "tools", "action"),
|
|
131
|
+
"vidura": ("policy", "security", "adversarial"),
|
|
132
|
+
"sangha": (),
|
|
133
|
+
"dharma_steward": ("policy", "reliability", "framework"),
|
|
134
|
+
},
|
|
135
|
+
},
|
|
136
|
+
"curriculum": {
|
|
137
|
+
"allocation_kind": "deliberate_practice_curriculum",
|
|
138
|
+
"roles": ("teacher", "student", "coach"),
|
|
139
|
+
"role_archetypes": (
|
|
140
|
+
"staged_practice",
|
|
141
|
+
"metric_drill",
|
|
142
|
+
"remediation_coach",
|
|
143
|
+
),
|
|
144
|
+
"path_prefixes": ("objective", "planner", "memory", "policy", "framework"),
|
|
145
|
+
"role_path_prefixes": {
|
|
146
|
+
"teacher": ("objective", "evaluation", "framework"),
|
|
147
|
+
"student": (),
|
|
148
|
+
"coach": ("memory", "policy", "planner"),
|
|
149
|
+
},
|
|
150
|
+
},
|
|
151
|
+
"evolution": {
|
|
152
|
+
"allocation_kind": "evolutionary_exploration",
|
|
153
|
+
"roles": ("population_explorer", "mutation_stressor", "fitness_selector"),
|
|
154
|
+
"role_archetypes": ("variation", "stress", "selection"),
|
|
155
|
+
"path_prefixes": (),
|
|
156
|
+
"role_path_prefixes": {
|
|
157
|
+
"population_explorer": (),
|
|
158
|
+
"mutation_stressor": ("security", "policy", "tools", "framework"),
|
|
159
|
+
"fitness_selector": (),
|
|
160
|
+
},
|
|
161
|
+
},
|
|
162
|
+
"pareto": {
|
|
163
|
+
"allocation_kind": "pareto_tradeoff_search",
|
|
164
|
+
"roles": ("tradeoff_arbiter", "frontier_keeper"),
|
|
165
|
+
"role_archetypes": ("multi_objective_balance", "frontier_selection"),
|
|
166
|
+
"path_prefixes": (),
|
|
167
|
+
"role_path_prefixes": {
|
|
168
|
+
"tradeoff_arbiter": (),
|
|
169
|
+
"frontier_keeper": (),
|
|
170
|
+
},
|
|
171
|
+
},
|
|
172
|
+
"tpe": {
|
|
173
|
+
"allocation_kind": "tpe_prior_sampling",
|
|
174
|
+
"roles": ("prior_sampler", "density_estimator"),
|
|
175
|
+
"role_archetypes": ("probabilistic_prior", "expected_improvement"),
|
|
176
|
+
"path_prefixes": (),
|
|
177
|
+
"role_path_prefixes": {
|
|
178
|
+
"prior_sampler": (),
|
|
179
|
+
"density_estimator": (),
|
|
180
|
+
},
|
|
181
|
+
},
|
|
182
|
+
"bandit": {
|
|
183
|
+
"allocation_kind": "bandit_budget_allocation",
|
|
184
|
+
"roles": ("allocation_arbiter", "exploit_explore_allocator"),
|
|
185
|
+
"role_archetypes": ("budget_allocator", "adaptive_selection"),
|
|
186
|
+
"path_prefixes": (),
|
|
187
|
+
"role_path_prefixes": {
|
|
188
|
+
"allocation_arbiter": (),
|
|
189
|
+
"exploit_explore_allocator": (),
|
|
190
|
+
},
|
|
191
|
+
},
|
|
192
|
+
"agent": {
|
|
193
|
+
"allocation_kind": "deterministic_candidate_search",
|
|
194
|
+
"roles": ("deterministic_engineer",),
|
|
195
|
+
"role_archetypes": ("metric_patch_search",),
|
|
196
|
+
"path_prefixes": (),
|
|
197
|
+
"role_path_prefixes": {"deterministic_engineer": ()},
|
|
198
|
+
},
|
|
199
|
+
}
|
|
200
|
+
DeploymentLike = (
|
|
201
|
+
AgentPromotionCheck
|
|
202
|
+
| AgentDeploymentExport
|
|
203
|
+
| OptimizationResult
|
|
204
|
+
| AgentCandidate
|
|
205
|
+
| Mapping[str, Any]
|
|
206
|
+
)
|
|
207
|
+
CandidateScorer = Callable[
|
|
208
|
+
[AgentCandidate],
|
|
209
|
+
CandidateEvaluation | EvaluationResult | float,
|
|
210
|
+
]
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
class AgentFeedbackCase(BaseModel):
|
|
214
|
+
"""One production or replayed feedback observation used for re-optimization."""
|
|
215
|
+
|
|
216
|
+
index: int
|
|
217
|
+
source: str = "rollback_observation"
|
|
218
|
+
candidate_id: Optional[str] = None
|
|
219
|
+
score: float
|
|
220
|
+
passed: bool
|
|
221
|
+
failures: list[str] = Field(default_factory=list)
|
|
222
|
+
metrics: dict[str, float] = Field(default_factory=dict)
|
|
223
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
class AgentFeedbackOptimizationResult(BaseModel):
|
|
227
|
+
"""Audit record for a live-feedback-triggered optimization round."""
|
|
228
|
+
|
|
229
|
+
schema_version: str = FEEDBACK_SCHEMA_VERSION
|
|
230
|
+
optimizer: str
|
|
231
|
+
feedback_source: str
|
|
232
|
+
rollback_decision: AgentRollbackDecision
|
|
233
|
+
feedback_cases: list[AgentFeedbackCase] = Field(default_factory=list)
|
|
234
|
+
diagnoses: list[ComponentDiagnosis] = Field(default_factory=list)
|
|
235
|
+
search_paths: list[str] = Field(default_factory=list)
|
|
236
|
+
reoptimization_result: OptimizationResult
|
|
237
|
+
baseline_score: Optional[float] = None
|
|
238
|
+
feedback_score: Optional[float] = None
|
|
239
|
+
final_score: float
|
|
240
|
+
baseline_delta: Optional[float] = None
|
|
241
|
+
feedback_delta: Optional[float] = None
|
|
242
|
+
improved: bool
|
|
243
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
244
|
+
|
|
245
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
246
|
+
return self.model_dump()
|
|
247
|
+
|
|
248
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
249
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
class AgentMultiInteractionBackendPlan(BaseModel):
|
|
253
|
+
"""One deterministic backend allocation in a multi-interaction round."""
|
|
254
|
+
|
|
255
|
+
optimizer: str
|
|
256
|
+
rank: int
|
|
257
|
+
weight: float
|
|
258
|
+
reason: str
|
|
259
|
+
kwargs: dict[str, Any] = Field(default_factory=dict)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
class AgentMultiInteractionBackendRun(BaseModel):
|
|
263
|
+
"""Result from running one allocated optimizer backend."""
|
|
264
|
+
|
|
265
|
+
optimizer: str
|
|
266
|
+
rank: int
|
|
267
|
+
status: str
|
|
268
|
+
final_score: Optional[float] = None
|
|
269
|
+
improved: bool = False
|
|
270
|
+
total_evaluations: int = 0
|
|
271
|
+
failure: Optional[str] = None
|
|
272
|
+
result: Optional[AgentFeedbackOptimizationResult] = None
|
|
273
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
class AgentMultiInteractionBackendLineage(BaseModel):
|
|
277
|
+
"""Candidate contribution summary for one backend in a portfolio run."""
|
|
278
|
+
|
|
279
|
+
optimizer: str
|
|
280
|
+
rank: int
|
|
281
|
+
allocation_weight: float = 0.0
|
|
282
|
+
allocation_reason: str = ""
|
|
283
|
+
status: str
|
|
284
|
+
final_score: Optional[float] = None
|
|
285
|
+
improved: bool = False
|
|
286
|
+
total_evaluations: int = 0
|
|
287
|
+
candidate_id: Optional[str] = None
|
|
288
|
+
parent_candidate_id: Optional[str] = None
|
|
289
|
+
candidate_patch: dict[str, Any] = Field(default_factory=dict)
|
|
290
|
+
patch_paths: list[str] = Field(default_factory=list)
|
|
291
|
+
unique_candidate_patch: dict[str, Any] = Field(default_factory=dict)
|
|
292
|
+
unique_patch_paths: list[str] = Field(default_factory=list)
|
|
293
|
+
shared_candidate_patch: dict[str, Any] = Field(default_factory=dict)
|
|
294
|
+
shared_patch_paths: list[str] = Field(default_factory=list)
|
|
295
|
+
equivalent_backends: list[str] = Field(default_factory=list)
|
|
296
|
+
equivalent_backend_count: int = 0
|
|
297
|
+
selection_relation: str = "unclassified"
|
|
298
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
class AgentMultiInteractionAblationReport(BaseModel):
|
|
302
|
+
"""Leave-one-backend-out summary for the selected portfolio result."""
|
|
303
|
+
|
|
304
|
+
selected_optimizer: str
|
|
305
|
+
selected_candidate_id: Optional[str] = None
|
|
306
|
+
selected_patch: dict[str, Any] = Field(default_factory=dict)
|
|
307
|
+
selected_patch_paths: list[str] = Field(default_factory=list)
|
|
308
|
+
final_score: float
|
|
309
|
+
best_without_selected_optimizer: Optional[str] = None
|
|
310
|
+
best_without_selected_score: Optional[float] = None
|
|
311
|
+
score_delta_without_selected: Optional[float] = None
|
|
312
|
+
selected_backend_required: bool
|
|
313
|
+
dependency: str
|
|
314
|
+
dependency_reason: str
|
|
315
|
+
consensus_backends: list[str] = Field(default_factory=list)
|
|
316
|
+
consensus_backend_count: int = 0
|
|
317
|
+
shared_selected_patch_paths: list[str] = Field(default_factory=list)
|
|
318
|
+
unique_selected_patch_paths: list[str] = Field(default_factory=list)
|
|
319
|
+
selected_patch_support: dict[str, list[str]] = Field(default_factory=dict)
|
|
320
|
+
backend_scoreboard: list[dict[str, Any]] = Field(default_factory=list)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
class AgentMultiInteractionOptimizationResult(BaseModel):
|
|
324
|
+
"""Audit record for automatic multi-backend agent re-optimization."""
|
|
325
|
+
|
|
326
|
+
schema_version: str = MULTI_INTERACTION_SCHEMA_VERSION
|
|
327
|
+
selected_optimizer: str
|
|
328
|
+
feedback_source: str
|
|
329
|
+
rollback_decision: AgentRollbackDecision
|
|
330
|
+
feedback_cases: list[AgentFeedbackCase] = Field(default_factory=list)
|
|
331
|
+
diagnoses: list[ComponentDiagnosis] = Field(default_factory=list)
|
|
332
|
+
search_paths: list[str] = Field(default_factory=list)
|
|
333
|
+
backend_plan: list[AgentMultiInteractionBackendPlan] = Field(default_factory=list)
|
|
334
|
+
backend_runs: list[AgentMultiInteractionBackendRun] = Field(default_factory=list)
|
|
335
|
+
backend_lineage: list[AgentMultiInteractionBackendLineage] = Field(default_factory=list)
|
|
336
|
+
ablation_report: AgentMultiInteractionAblationReport
|
|
337
|
+
best_result: AgentFeedbackOptimizationResult
|
|
338
|
+
final_score: float
|
|
339
|
+
improved: bool
|
|
340
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
341
|
+
|
|
342
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
343
|
+
return self.model_dump()
|
|
344
|
+
|
|
345
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
346
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
class AgentFeedbackOptimizer(BaseOptimizer):
|
|
350
|
+
"""
|
|
351
|
+
Re-optimize an agent from live trace/evaluation feedback.
|
|
352
|
+
|
|
353
|
+
The optimizer first turns post-deployment rollback evidence into component
|
|
354
|
+
diagnoses and search paths, then delegates the actual search to one of the
|
|
355
|
+
existing agent optimizers (`society`, `social_memory`, `curriculum`,
|
|
356
|
+
`council`, `evolution`, `tpe`, `pareto`, `bandit`, or deterministic
|
|
357
|
+
`agent`).
|
|
358
|
+
"""
|
|
359
|
+
|
|
360
|
+
def __init__(
|
|
361
|
+
self,
|
|
362
|
+
target: Optional[OptimizationTarget] = None,
|
|
363
|
+
*,
|
|
364
|
+
deployment: Optional[DeploymentLike] = None,
|
|
365
|
+
rollback_decision: Optional[AgentRollbackDecision] = None,
|
|
366
|
+
live_evaluations: Optional[Sequence[Any]] = None,
|
|
367
|
+
evaluate_candidate: Optional[CandidateScorer] = None,
|
|
368
|
+
simulation_evaluator: Any = None,
|
|
369
|
+
optimizer: str = "society",
|
|
370
|
+
diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
|
|
371
|
+
diagnostic_score_threshold: float = 0.85,
|
|
372
|
+
optimizer_kwargs: Optional[Mapping[str, Any]] = None,
|
|
373
|
+
rollback_kwargs: Optional[Mapping[str, Any]] = None,
|
|
374
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
375
|
+
) -> None:
|
|
376
|
+
self.target = target
|
|
377
|
+
self.deployment = deployment
|
|
378
|
+
self.rollback_decision = rollback_decision
|
|
379
|
+
self.live_evaluations = (
|
|
380
|
+
list(live_evaluations) if live_evaluations is not None else None
|
|
381
|
+
)
|
|
382
|
+
self.evaluate_candidate = evaluate_candidate
|
|
383
|
+
self.simulation_evaluator = simulation_evaluator
|
|
384
|
+
self.optimizer = optimizer
|
|
385
|
+
self.diagnoses = _normalize_diagnoses(diagnoses)
|
|
386
|
+
self.diagnostic_score_threshold = diagnostic_score_threshold
|
|
387
|
+
self.optimizer_kwargs = dict(optimizer_kwargs or {})
|
|
388
|
+
self.rollback_kwargs = dict(rollback_kwargs or {})
|
|
389
|
+
self.metadata = dict(metadata or {})
|
|
390
|
+
super().__init__()
|
|
391
|
+
|
|
392
|
+
def optimize(
|
|
393
|
+
self,
|
|
394
|
+
evaluator: Any = None,
|
|
395
|
+
data_mapper: Any = None,
|
|
396
|
+
dataset: Optional[List[dict[str, Any]]] = None,
|
|
397
|
+
metric: Optional[Callable] = None,
|
|
398
|
+
*,
|
|
399
|
+
target: Optional[OptimizationTarget] = None,
|
|
400
|
+
deployment: Optional[DeploymentLike] = None,
|
|
401
|
+
rollback_decision: Optional[AgentRollbackDecision] = None,
|
|
402
|
+
live_evaluations: Optional[Sequence[Any]] = None,
|
|
403
|
+
evaluate_candidate: Optional[CandidateScorer] = None,
|
|
404
|
+
simulation_evaluator: Any = None,
|
|
405
|
+
optimizer: Optional[str] = None,
|
|
406
|
+
diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
|
|
407
|
+
diagnostic_score_threshold: Optional[float] = None,
|
|
408
|
+
optimizer_kwargs: Optional[Mapping[str, Any]] = None,
|
|
409
|
+
rollback_kwargs: Optional[Mapping[str, Any]] = None,
|
|
410
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
411
|
+
**backend_kwargs: Any,
|
|
412
|
+
) -> AgentFeedbackOptimizationResult:
|
|
413
|
+
active_target = target or self.target
|
|
414
|
+
if active_target is None:
|
|
415
|
+
raise ValueError("AgentFeedbackOptimizer requires a target.")
|
|
416
|
+
|
|
417
|
+
active_evaluator = evaluate_candidate or self.evaluate_candidate
|
|
418
|
+
active_simulation = simulation_evaluator or self.simulation_evaluator
|
|
419
|
+
if (
|
|
420
|
+
active_evaluator is None
|
|
421
|
+
and getattr(active_simulation, "evaluate_candidate", None) is None
|
|
422
|
+
):
|
|
423
|
+
raise ValueError(
|
|
424
|
+
"AgentFeedbackOptimizer requires evaluate_candidate or simulation_evaluator."
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
active_diagnostic_threshold = (
|
|
428
|
+
self.diagnostic_score_threshold
|
|
429
|
+
if diagnostic_score_threshold is None
|
|
430
|
+
else diagnostic_score_threshold
|
|
431
|
+
)
|
|
432
|
+
explicit_diagnoses = _normalize_diagnoses(diagnoses)
|
|
433
|
+
if diagnoses is None:
|
|
434
|
+
explicit_diagnoses = list(self.diagnoses)
|
|
435
|
+
|
|
436
|
+
active_rollback_decision = rollback_decision or self.rollback_decision
|
|
437
|
+
active_live_evaluations = (
|
|
438
|
+
list(live_evaluations)
|
|
439
|
+
if live_evaluations is not None
|
|
440
|
+
else self.live_evaluations
|
|
441
|
+
)
|
|
442
|
+
active_deployment = deployment or self.deployment
|
|
443
|
+
active_deployment, auto_seed_deployment = _auto_seed_deployment_for_replay(
|
|
444
|
+
target=active_target,
|
|
445
|
+
deployment=active_deployment,
|
|
446
|
+
rollback_decision=active_rollback_decision,
|
|
447
|
+
live_evaluations=active_live_evaluations,
|
|
448
|
+
simulation_evaluator=active_simulation,
|
|
449
|
+
metadata={**self.metadata, **dict(metadata or {})},
|
|
450
|
+
)
|
|
451
|
+
decision, feedback_source = _resolve_rollback_decision(
|
|
452
|
+
rollback_decision=active_rollback_decision,
|
|
453
|
+
deployment=active_deployment,
|
|
454
|
+
live_evaluations=active_live_evaluations,
|
|
455
|
+
simulation_evaluator=active_simulation,
|
|
456
|
+
rollback_kwargs={
|
|
457
|
+
**self.rollback_kwargs,
|
|
458
|
+
**dict(rollback_kwargs or {}),
|
|
459
|
+
},
|
|
460
|
+
)
|
|
461
|
+
feedback_cases = _feedback_cases_from_rollback(decision)
|
|
462
|
+
feedback_diagnoses = _diagnose_feedback_cases(
|
|
463
|
+
feedback_cases,
|
|
464
|
+
target=active_target,
|
|
465
|
+
failing_threshold=active_diagnostic_threshold,
|
|
466
|
+
)
|
|
467
|
+
active_diagnoses = _dedupe_diagnoses([*explicit_diagnoses, *feedback_diagnoses])
|
|
468
|
+
search_paths = _search_paths_for_feedback(active_target, active_diagnoses)
|
|
469
|
+
|
|
470
|
+
backend_name = optimizer or self.optimizer
|
|
471
|
+
resolved_optimizer = _resolve_feedback_optimizer(backend_name)
|
|
472
|
+
combined_backend_kwargs = {
|
|
473
|
+
**self.optimizer_kwargs,
|
|
474
|
+
**dict(optimizer_kwargs or {}),
|
|
475
|
+
**backend_kwargs,
|
|
476
|
+
}
|
|
477
|
+
backend = resolved_optimizer(
|
|
478
|
+
target=active_target,
|
|
479
|
+
evaluate_candidate=active_evaluator,
|
|
480
|
+
simulation_evaluator=active_simulation,
|
|
481
|
+
diagnoses=active_diagnoses,
|
|
482
|
+
diagnostic_score_threshold=active_diagnostic_threshold,
|
|
483
|
+
**combined_backend_kwargs,
|
|
484
|
+
)
|
|
485
|
+
reoptimization = backend.optimize()
|
|
486
|
+
baseline_score = decision.baseline_score
|
|
487
|
+
feedback_score = decision.latest_score
|
|
488
|
+
baseline_delta = (
|
|
489
|
+
reoptimization.final_score - baseline_score
|
|
490
|
+
if baseline_score is not None
|
|
491
|
+
else None
|
|
492
|
+
)
|
|
493
|
+
feedback_delta = (
|
|
494
|
+
reoptimization.final_score - feedback_score
|
|
495
|
+
if feedback_score is not None
|
|
496
|
+
else None
|
|
497
|
+
)
|
|
498
|
+
improved = (
|
|
499
|
+
reoptimization.final_score >= decision.min_score
|
|
500
|
+
and (feedback_delta is None or feedback_delta > 0)
|
|
501
|
+
)
|
|
502
|
+
result_metadata = {
|
|
503
|
+
**self.metadata,
|
|
504
|
+
**dict(metadata or {}),
|
|
505
|
+
"rollback_required": decision.rollback_required,
|
|
506
|
+
"failure_count": decision.failure_count,
|
|
507
|
+
"consecutive_failure_count": decision.consecutive_failure_count,
|
|
508
|
+
"auto_seed_deployment": auto_seed_deployment,
|
|
509
|
+
"backend_optimizer": reoptimization.metadata.get("optimizer"),
|
|
510
|
+
}
|
|
511
|
+
return AgentFeedbackOptimizationResult(
|
|
512
|
+
optimizer=_normalize_optimizer_name(backend_name),
|
|
513
|
+
feedback_source=feedback_source,
|
|
514
|
+
rollback_decision=decision,
|
|
515
|
+
feedback_cases=feedback_cases,
|
|
516
|
+
diagnoses=active_diagnoses,
|
|
517
|
+
search_paths=search_paths,
|
|
518
|
+
reoptimization_result=reoptimization,
|
|
519
|
+
baseline_score=baseline_score,
|
|
520
|
+
feedback_score=feedback_score,
|
|
521
|
+
final_score=reoptimization.final_score,
|
|
522
|
+
baseline_delta=baseline_delta,
|
|
523
|
+
feedback_delta=feedback_delta,
|
|
524
|
+
improved=improved,
|
|
525
|
+
metadata=result_metadata,
|
|
526
|
+
)
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
class AgentMultiInteractionOptimizer(BaseOptimizer):
|
|
530
|
+
"""
|
|
531
|
+
Diagnose feedback, allocate deterministic optimizer backends, and select the best.
|
|
532
|
+
|
|
533
|
+
This is the Future AGI-native portfolio layer above `AgentFeedbackOptimizer`:
|
|
534
|
+
every backend receives the same rollback/replay evidence and metric-derived
|
|
535
|
+
diagnoses, while the allocator chooses backend priority from feedback
|
|
536
|
+
metrics, target layers, and search-space shape. Social/psychological
|
|
537
|
+
inspiration stays metadata-only; candidate acceptance is numeric.
|
|
538
|
+
"""
|
|
539
|
+
|
|
540
|
+
def __init__(
|
|
541
|
+
self,
|
|
542
|
+
target: Optional[OptimizationTarget] = None,
|
|
543
|
+
*,
|
|
544
|
+
deployment: Optional[DeploymentLike] = None,
|
|
545
|
+
rollback_decision: Optional[AgentRollbackDecision] = None,
|
|
546
|
+
live_evaluations: Optional[Sequence[Any]] = None,
|
|
547
|
+
evaluate_candidate: Optional[CandidateScorer] = None,
|
|
548
|
+
simulation_evaluator: Any = None,
|
|
549
|
+
optimizer_pool: Optional[Sequence[str]] = None,
|
|
550
|
+
max_backends: Optional[int] = None,
|
|
551
|
+
diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
|
|
552
|
+
diagnostic_score_threshold: float = 0.85,
|
|
553
|
+
optimizer_kwargs: Optional[Mapping[str, Any]] = None,
|
|
554
|
+
optimizer_kwargs_by_backend: Optional[Mapping[str, Mapping[str, Any]]] = None,
|
|
555
|
+
rollback_kwargs: Optional[Mapping[str, Any]] = None,
|
|
556
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
557
|
+
) -> None:
|
|
558
|
+
self.target = target
|
|
559
|
+
self.deployment = deployment
|
|
560
|
+
self.rollback_decision = rollback_decision
|
|
561
|
+
self.live_evaluations = (
|
|
562
|
+
list(live_evaluations) if live_evaluations is not None else None
|
|
563
|
+
)
|
|
564
|
+
self.evaluate_candidate = evaluate_candidate
|
|
565
|
+
self.simulation_evaluator = simulation_evaluator
|
|
566
|
+
self.optimizer_pool = list(optimizer_pool) if optimizer_pool is not None else None
|
|
567
|
+
self.max_backends = max_backends
|
|
568
|
+
self.diagnoses = _normalize_diagnoses(diagnoses)
|
|
569
|
+
self.diagnostic_score_threshold = diagnostic_score_threshold
|
|
570
|
+
self.optimizer_kwargs = dict(optimizer_kwargs or {})
|
|
571
|
+
self.optimizer_kwargs_by_backend = {
|
|
572
|
+
_normalize_optimizer_name(key): dict(value)
|
|
573
|
+
for key, value in dict(optimizer_kwargs_by_backend or {}).items()
|
|
574
|
+
}
|
|
575
|
+
self.rollback_kwargs = dict(rollback_kwargs or {})
|
|
576
|
+
self.metadata = dict(metadata or {})
|
|
577
|
+
super().__init__()
|
|
578
|
+
|
|
579
|
+
def optimize(
|
|
580
|
+
self,
|
|
581
|
+
evaluator: Any = None,
|
|
582
|
+
data_mapper: Any = None,
|
|
583
|
+
dataset: Optional[List[dict[str, Any]]] = None,
|
|
584
|
+
metric: Optional[Callable] = None,
|
|
585
|
+
*,
|
|
586
|
+
target: Optional[OptimizationTarget] = None,
|
|
587
|
+
deployment: Optional[DeploymentLike] = None,
|
|
588
|
+
rollback_decision: Optional[AgentRollbackDecision] = None,
|
|
589
|
+
live_evaluations: Optional[Sequence[Any]] = None,
|
|
590
|
+
evaluate_candidate: Optional[CandidateScorer] = None,
|
|
591
|
+
simulation_evaluator: Any = None,
|
|
592
|
+
optimizer_pool: Optional[Sequence[str]] = None,
|
|
593
|
+
max_backends: Optional[int] = None,
|
|
594
|
+
diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
|
|
595
|
+
diagnostic_score_threshold: Optional[float] = None,
|
|
596
|
+
optimizer_kwargs: Optional[Mapping[str, Any]] = None,
|
|
597
|
+
optimizer_kwargs_by_backend: Optional[Mapping[str, Mapping[str, Any]]] = None,
|
|
598
|
+
rollback_kwargs: Optional[Mapping[str, Any]] = None,
|
|
599
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
600
|
+
**backend_kwargs: Any,
|
|
601
|
+
) -> AgentMultiInteractionOptimizationResult:
|
|
602
|
+
active_target = target or self.target
|
|
603
|
+
if active_target is None:
|
|
604
|
+
raise ValueError("AgentMultiInteractionOptimizer requires a target.")
|
|
605
|
+
|
|
606
|
+
active_evaluator = evaluate_candidate or self.evaluate_candidate or evaluator
|
|
607
|
+
active_simulation = simulation_evaluator or self.simulation_evaluator
|
|
608
|
+
if (
|
|
609
|
+
active_evaluator is None
|
|
610
|
+
and getattr(active_simulation, "evaluate_candidate", None) is None
|
|
611
|
+
):
|
|
612
|
+
raise ValueError(
|
|
613
|
+
"AgentMultiInteractionOptimizer requires evaluate_candidate or simulation_evaluator."
|
|
614
|
+
)
|
|
615
|
+
|
|
616
|
+
active_diagnostic_threshold = (
|
|
617
|
+
self.diagnostic_score_threshold
|
|
618
|
+
if diagnostic_score_threshold is None
|
|
619
|
+
else diagnostic_score_threshold
|
|
620
|
+
)
|
|
621
|
+
explicit_diagnoses = _normalize_diagnoses(diagnoses)
|
|
622
|
+
if diagnoses is None:
|
|
623
|
+
explicit_diagnoses = list(self.diagnoses)
|
|
624
|
+
|
|
625
|
+
active_rollback_decision = rollback_decision or self.rollback_decision
|
|
626
|
+
active_live_evaluations = (
|
|
627
|
+
list(live_evaluations)
|
|
628
|
+
if live_evaluations is not None
|
|
629
|
+
else self.live_evaluations
|
|
630
|
+
)
|
|
631
|
+
active_deployment = deployment or self.deployment
|
|
632
|
+
active_deployment, auto_seed_deployment = _auto_seed_deployment_for_replay(
|
|
633
|
+
target=active_target,
|
|
634
|
+
deployment=active_deployment,
|
|
635
|
+
rollback_decision=active_rollback_decision,
|
|
636
|
+
live_evaluations=active_live_evaluations,
|
|
637
|
+
simulation_evaluator=active_simulation,
|
|
638
|
+
metadata={**self.metadata, **dict(metadata or {})},
|
|
639
|
+
)
|
|
640
|
+
decision, feedback_source = _resolve_rollback_decision(
|
|
641
|
+
rollback_decision=active_rollback_decision,
|
|
642
|
+
deployment=active_deployment,
|
|
643
|
+
live_evaluations=active_live_evaluations,
|
|
644
|
+
simulation_evaluator=active_simulation,
|
|
645
|
+
rollback_kwargs={
|
|
646
|
+
**self.rollback_kwargs,
|
|
647
|
+
**dict(rollback_kwargs or {}),
|
|
648
|
+
},
|
|
649
|
+
)
|
|
650
|
+
feedback_cases = _feedback_cases_from_rollback(decision)
|
|
651
|
+
feedback_diagnoses = _diagnose_feedback_cases(
|
|
652
|
+
feedback_cases,
|
|
653
|
+
target=active_target,
|
|
654
|
+
failing_threshold=active_diagnostic_threshold,
|
|
655
|
+
)
|
|
656
|
+
active_diagnoses = _dedupe_diagnoses([*explicit_diagnoses, *feedback_diagnoses])
|
|
657
|
+
search_paths = _search_paths_for_feedback(active_target, active_diagnoses)
|
|
658
|
+
|
|
659
|
+
base_optimizer_kwargs = {
|
|
660
|
+
**self.optimizer_kwargs,
|
|
661
|
+
**dict(optimizer_kwargs or {}),
|
|
662
|
+
**backend_kwargs,
|
|
663
|
+
}
|
|
664
|
+
per_backend_kwargs = dict(self.optimizer_kwargs_by_backend)
|
|
665
|
+
for key, value in dict(optimizer_kwargs_by_backend or {}).items():
|
|
666
|
+
per_backend_kwargs[_normalize_optimizer_name(key)] = dict(value)
|
|
667
|
+
|
|
668
|
+
plan = _multi_interaction_backend_plan(
|
|
669
|
+
target=active_target,
|
|
670
|
+
feedback_cases=feedback_cases,
|
|
671
|
+
diagnoses=active_diagnoses,
|
|
672
|
+
search_paths=search_paths,
|
|
673
|
+
optimizer_pool=optimizer_pool or self.optimizer_pool,
|
|
674
|
+
max_backends=self.max_backends if max_backends is None else max_backends,
|
|
675
|
+
optimizer_kwargs=base_optimizer_kwargs,
|
|
676
|
+
optimizer_kwargs_by_backend=per_backend_kwargs,
|
|
677
|
+
)
|
|
678
|
+
if not plan:
|
|
679
|
+
raise ValueError("AgentMultiInteractionOptimizer backend plan cannot be empty.")
|
|
680
|
+
|
|
681
|
+
runs: list[AgentMultiInteractionBackendRun] = []
|
|
682
|
+
for allocation in plan:
|
|
683
|
+
try:
|
|
684
|
+
result = AgentFeedbackOptimizer(
|
|
685
|
+
target=active_target,
|
|
686
|
+
rollback_decision=decision,
|
|
687
|
+
evaluate_candidate=active_evaluator,
|
|
688
|
+
simulation_evaluator=active_simulation,
|
|
689
|
+
optimizer=allocation.optimizer,
|
|
690
|
+
diagnoses=active_diagnoses,
|
|
691
|
+
diagnostic_score_threshold=active_diagnostic_threshold,
|
|
692
|
+
optimizer_kwargs=allocation.kwargs,
|
|
693
|
+
metadata={
|
|
694
|
+
"multi_interaction_optimizer": True,
|
|
695
|
+
"backend_rank": allocation.rank,
|
|
696
|
+
"backend_weight": allocation.weight,
|
|
697
|
+
"backend_reason": allocation.reason,
|
|
698
|
+
},
|
|
699
|
+
).optimize()
|
|
700
|
+
runs.append(
|
|
701
|
+
AgentMultiInteractionBackendRun(
|
|
702
|
+
optimizer=allocation.optimizer,
|
|
703
|
+
rank=allocation.rank,
|
|
704
|
+
status="completed",
|
|
705
|
+
final_score=result.final_score,
|
|
706
|
+
improved=result.improved,
|
|
707
|
+
total_evaluations=result.reoptimization_result.total_evaluations,
|
|
708
|
+
result=result,
|
|
709
|
+
metadata={
|
|
710
|
+
"backend_optimizer": result.metadata.get("backend_optimizer"),
|
|
711
|
+
},
|
|
712
|
+
)
|
|
713
|
+
)
|
|
714
|
+
except Exception as exc:
|
|
715
|
+
runs.append(
|
|
716
|
+
AgentMultiInteractionBackendRun(
|
|
717
|
+
optimizer=allocation.optimizer,
|
|
718
|
+
rank=allocation.rank,
|
|
719
|
+
status="failed",
|
|
720
|
+
failure=str(exc),
|
|
721
|
+
)
|
|
722
|
+
)
|
|
723
|
+
|
|
724
|
+
successful_runs = [run for run in runs if run.result is not None]
|
|
725
|
+
if not successful_runs:
|
|
726
|
+
failures = "; ".join(
|
|
727
|
+
f"{run.optimizer}: {run.failure}" for run in runs if run.failure
|
|
728
|
+
)
|
|
729
|
+
raise RuntimeError(
|
|
730
|
+
"AgentMultiInteractionOptimizer did not complete any backend"
|
|
731
|
+
+ (f": {failures}" if failures else ".")
|
|
732
|
+
)
|
|
733
|
+
best_run = max(
|
|
734
|
+
successful_runs,
|
|
735
|
+
key=lambda run: (
|
|
736
|
+
run.final_score if run.final_score is not None else float("-inf"),
|
|
737
|
+
1 if run.improved else 0,
|
|
738
|
+
-run.rank,
|
|
739
|
+
-run.total_evaluations,
|
|
740
|
+
),
|
|
741
|
+
)
|
|
742
|
+
assert best_run.result is not None
|
|
743
|
+
backend_lineage = _multi_interaction_backend_lineage(
|
|
744
|
+
target=active_target,
|
|
745
|
+
plan=plan,
|
|
746
|
+
runs=runs,
|
|
747
|
+
selected_run=best_run,
|
|
748
|
+
)
|
|
749
|
+
ablation_report = _multi_interaction_ablation_report(
|
|
750
|
+
lineage=backend_lineage,
|
|
751
|
+
selected_run=best_run,
|
|
752
|
+
)
|
|
753
|
+
allocation_metadata = _multi_interaction_allocation_metadata(
|
|
754
|
+
target=active_target,
|
|
755
|
+
plan=plan,
|
|
756
|
+
feedback_cases=feedback_cases,
|
|
757
|
+
diagnoses=active_diagnoses,
|
|
758
|
+
search_paths=search_paths,
|
|
759
|
+
)
|
|
760
|
+
result_metadata = {
|
|
761
|
+
**self.metadata,
|
|
762
|
+
**dict(metadata or {}),
|
|
763
|
+
"allocator": "metric_diagnosis_backend_portfolio",
|
|
764
|
+
**allocation_metadata,
|
|
765
|
+
"auto_seed_deployment": auto_seed_deployment,
|
|
766
|
+
"backend_count": len(plan),
|
|
767
|
+
"completed_backend_count": len(successful_runs),
|
|
768
|
+
"failed_backend_count": len(runs) - len(successful_runs),
|
|
769
|
+
"optimizer_pool": [allocation.optimizer for allocation in plan],
|
|
770
|
+
"selection_rule": "highest_final_score_then_improved_then_rank",
|
|
771
|
+
"ablation_dependency": ablation_report.dependency,
|
|
772
|
+
"selected_backend_required": ablation_report.selected_backend_required,
|
|
773
|
+
"consensus_backend_count": ablation_report.consensus_backend_count,
|
|
774
|
+
"selected_patch_paths": list(ablation_report.selected_patch_paths),
|
|
775
|
+
"strategy_inspiration": (
|
|
776
|
+
"diagnostic triage, deliberate practice, council synthesis, "
|
|
777
|
+
"social memory, evolutionary exploration, Pareto tradeoff, "
|
|
778
|
+
"TPE sampling, bandit allocation, human team roles, and "
|
|
779
|
+
"Hindu-mythology-inspired society labels; labels are metadata only"
|
|
780
|
+
),
|
|
781
|
+
}
|
|
782
|
+
return AgentMultiInteractionOptimizationResult(
|
|
783
|
+
selected_optimizer=best_run.optimizer,
|
|
784
|
+
feedback_source=feedback_source,
|
|
785
|
+
rollback_decision=decision,
|
|
786
|
+
feedback_cases=feedback_cases,
|
|
787
|
+
diagnoses=active_diagnoses,
|
|
788
|
+
search_paths=search_paths,
|
|
789
|
+
backend_plan=plan,
|
|
790
|
+
backend_runs=runs,
|
|
791
|
+
backend_lineage=backend_lineage,
|
|
792
|
+
ablation_report=ablation_report,
|
|
793
|
+
best_result=best_run.result,
|
|
794
|
+
final_score=best_run.result.final_score,
|
|
795
|
+
improved=best_run.result.improved,
|
|
796
|
+
metadata=result_metadata,
|
|
797
|
+
)
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _multi_interaction_backend_plan(
|
|
801
|
+
*,
|
|
802
|
+
target: OptimizationTarget,
|
|
803
|
+
feedback_cases: Sequence[AgentFeedbackCase],
|
|
804
|
+
diagnoses: Sequence[ComponentDiagnosis],
|
|
805
|
+
search_paths: Sequence[str],
|
|
806
|
+
optimizer_pool: Optional[Sequence[str]],
|
|
807
|
+
max_backends: Optional[int],
|
|
808
|
+
optimizer_kwargs: Mapping[str, Any],
|
|
809
|
+
optimizer_kwargs_by_backend: Mapping[str, Mapping[str, Any]],
|
|
810
|
+
) -> list[AgentMultiInteractionBackendPlan]:
|
|
811
|
+
if max_backends is not None and max_backends < 1:
|
|
812
|
+
raise ValueError("max_backends must be at least 1.")
|
|
813
|
+
|
|
814
|
+
metric_names = _failed_feedback_metric_names(feedback_cases) or _feedback_metric_names(
|
|
815
|
+
feedback_cases
|
|
816
|
+
)
|
|
817
|
+
normalized_pool = _dedupe_optimizer_pool(optimizer_pool or DEFAULT_MULTI_INTERACTION_BACKENDS)
|
|
818
|
+
scored: list[tuple[float, int, str, str, dict[str, Any]]] = []
|
|
819
|
+
default_order = {
|
|
820
|
+
name: index for index, name in enumerate(DEFAULT_MULTI_INTERACTION_BACKENDS)
|
|
821
|
+
}
|
|
822
|
+
for optimizer_name in normalized_pool:
|
|
823
|
+
_resolve_feedback_optimizer(optimizer_name)
|
|
824
|
+
backend_kwargs = _backend_kwargs_for_multi_interaction(
|
|
825
|
+
optimizer_name,
|
|
826
|
+
target=target,
|
|
827
|
+
metric_names=metric_names,
|
|
828
|
+
optimizer_kwargs=optimizer_kwargs,
|
|
829
|
+
optimizer_kwargs_by_backend=optimizer_kwargs_by_backend,
|
|
830
|
+
)
|
|
831
|
+
if optimizer_name == "pareto" and not backend_kwargs.get("objective_names"):
|
|
832
|
+
continue
|
|
833
|
+
weight, reason = _backend_allocation_weight(
|
|
834
|
+
optimizer_name,
|
|
835
|
+
target=target,
|
|
836
|
+
feedback_cases=feedback_cases,
|
|
837
|
+
diagnoses=diagnoses,
|
|
838
|
+
search_paths=search_paths,
|
|
839
|
+
metric_names=metric_names,
|
|
840
|
+
)
|
|
841
|
+
scored.append(
|
|
842
|
+
(
|
|
843
|
+
weight,
|
|
844
|
+
-default_order.get(optimizer_name, len(DEFAULT_MULTI_INTERACTION_BACKENDS)),
|
|
845
|
+
optimizer_name,
|
|
846
|
+
reason,
|
|
847
|
+
backend_kwargs,
|
|
848
|
+
)
|
|
849
|
+
)
|
|
850
|
+
|
|
851
|
+
scored.sort(key=lambda item: (item[0], item[1], item[2]), reverse=True)
|
|
852
|
+
if max_backends is not None:
|
|
853
|
+
scored = scored[:max_backends]
|
|
854
|
+
return [
|
|
855
|
+
AgentMultiInteractionBackendPlan(
|
|
856
|
+
optimizer=optimizer_name,
|
|
857
|
+
rank=index,
|
|
858
|
+
weight=round(weight, 4),
|
|
859
|
+
reason=reason,
|
|
860
|
+
kwargs=backend_kwargs,
|
|
861
|
+
)
|
|
862
|
+
for index, (weight, _, optimizer_name, reason, backend_kwargs) in enumerate(
|
|
863
|
+
scored,
|
|
864
|
+
start=1,
|
|
865
|
+
)
|
|
866
|
+
]
|
|
867
|
+
|
|
868
|
+
|
|
869
|
+
def _auto_seed_deployment_for_replay(
|
|
870
|
+
*,
|
|
871
|
+
target: OptimizationTarget,
|
|
872
|
+
deployment: Optional[DeploymentLike],
|
|
873
|
+
rollback_decision: Optional[AgentRollbackDecision],
|
|
874
|
+
live_evaluations: Optional[Sequence[Any]],
|
|
875
|
+
simulation_evaluator: Any,
|
|
876
|
+
metadata: Mapping[str, Any],
|
|
877
|
+
) -> tuple[Optional[DeploymentLike], bool]:
|
|
878
|
+
if deployment is not None or rollback_decision is not None:
|
|
879
|
+
return deployment, False
|
|
880
|
+
if live_evaluations is not None:
|
|
881
|
+
return deployment, False
|
|
882
|
+
if getattr(simulation_evaluator, "evaluate_candidate", None) is None:
|
|
883
|
+
return deployment, False
|
|
884
|
+
|
|
885
|
+
seed = target.seed_candidate()
|
|
886
|
+
return (
|
|
887
|
+
export_agent_deployment(
|
|
888
|
+
seed,
|
|
889
|
+
framework="auto",
|
|
890
|
+
metadata={
|
|
891
|
+
**dict(metadata),
|
|
892
|
+
"auto_seed_deployment": True,
|
|
893
|
+
"auto_seed_deployment_source": "simulation_replay",
|
|
894
|
+
},
|
|
895
|
+
),
|
|
896
|
+
True,
|
|
897
|
+
)
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
def _multi_interaction_allocation_metadata(
|
|
901
|
+
*,
|
|
902
|
+
target: OptimizationTarget,
|
|
903
|
+
plan: Sequence[AgentMultiInteractionBackendPlan],
|
|
904
|
+
feedback_cases: Sequence[AgentFeedbackCase],
|
|
905
|
+
diagnoses: Sequence[ComponentDiagnosis],
|
|
906
|
+
search_paths: Sequence[str],
|
|
907
|
+
) -> dict[str, Any]:
|
|
908
|
+
metric_coverage = _diagnostic_metric_coverage(
|
|
909
|
+
diagnoses,
|
|
910
|
+
metric_names=_failed_feedback_metric_names(feedback_cases)
|
|
911
|
+
or _feedback_metric_names(feedback_cases),
|
|
912
|
+
)
|
|
913
|
+
active_paths = list(search_paths or target.search_space)
|
|
914
|
+
ledger: list[dict[str, Any]] = []
|
|
915
|
+
role_coverage: dict[str, int] = {}
|
|
916
|
+
archetype_coverage: dict[str, int] = {}
|
|
917
|
+
|
|
918
|
+
for allocation in plan:
|
|
919
|
+
profile = _multi_interaction_backend_profile(allocation.optimizer)
|
|
920
|
+
path_focus = _allocation_profile_path_focus(profile, active_paths)
|
|
921
|
+
role_path_focus = _allocation_role_path_focus(profile, path_focus)
|
|
922
|
+
diagnosis_focus = _allocation_diagnosis_focus(
|
|
923
|
+
profile=profile,
|
|
924
|
+
diagnoses=diagnoses,
|
|
925
|
+
active_paths=active_paths,
|
|
926
|
+
path_focus=path_focus,
|
|
927
|
+
)
|
|
928
|
+
for role in profile["roles"]:
|
|
929
|
+
role_coverage[role] = role_coverage.get(role, 0) + 1
|
|
930
|
+
for archetype in profile["role_archetypes"]:
|
|
931
|
+
archetype_coverage[archetype] = archetype_coverage.get(archetype, 0) + 1
|
|
932
|
+
|
|
933
|
+
focused_metrics = _diagnosis_focus_metric_coverage(diagnosis_focus)
|
|
934
|
+
ledger.append(
|
|
935
|
+
{
|
|
936
|
+
"optimizer": allocation.optimizer,
|
|
937
|
+
"rank": allocation.rank,
|
|
938
|
+
"weight": allocation.weight,
|
|
939
|
+
"reason": allocation.reason,
|
|
940
|
+
"allocation_kind": profile["allocation_kind"],
|
|
941
|
+
"roles": list(profile["roles"]),
|
|
942
|
+
"role_archetypes": list(profile["role_archetypes"]),
|
|
943
|
+
"path_focus": path_focus,
|
|
944
|
+
"role_path_focus": role_path_focus,
|
|
945
|
+
"diagnostic_components": _diagnosis_focus_values(
|
|
946
|
+
diagnosis_focus,
|
|
947
|
+
"component",
|
|
948
|
+
),
|
|
949
|
+
"diagnostic_failure_modes": _diagnosis_focus_values(
|
|
950
|
+
diagnosis_focus,
|
|
951
|
+
"failure_mode",
|
|
952
|
+
),
|
|
953
|
+
"diagnostic_metrics": focused_metrics or metric_coverage,
|
|
954
|
+
"diagnosis_focus": diagnosis_focus,
|
|
955
|
+
}
|
|
956
|
+
)
|
|
957
|
+
|
|
958
|
+
path_coverage = _ordered_patch_paths_for_keys(
|
|
959
|
+
_flatten_ledger_path_focus(ledger),
|
|
960
|
+
list(target.search_space),
|
|
961
|
+
)
|
|
962
|
+
return {
|
|
963
|
+
"allocation_algorithm": "deterministic_metric_diagnosis_society_agent_anchor_allocator",
|
|
964
|
+
"allocation_inspiration": (
|
|
965
|
+
"Human-team and society-role labels guide audit metadata only; "
|
|
966
|
+
"candidate acceptance remains metric-based."
|
|
967
|
+
),
|
|
968
|
+
"deterministic_agent_anchor": any(
|
|
969
|
+
allocation.optimizer == "agent"
|
|
970
|
+
and "focused deterministic diagnosis search" in allocation.reason
|
|
971
|
+
for allocation in plan
|
|
972
|
+
),
|
|
973
|
+
"society_allocation_ledger": ledger,
|
|
974
|
+
"allocation_role_coverage": dict(sorted(role_coverage.items())),
|
|
975
|
+
"allocation_archetype_coverage": dict(sorted(archetype_coverage.items())),
|
|
976
|
+
"allocation_metric_coverage": metric_coverage,
|
|
977
|
+
"allocation_search_path_coverage": path_coverage,
|
|
978
|
+
"allocation_diagnosis_coverage": _diagnosis_coverage_keys(diagnoses),
|
|
979
|
+
}
|
|
980
|
+
|
|
981
|
+
|
|
982
|
+
def _multi_interaction_backend_profile(optimizer_name: str) -> dict[str, Any]:
|
|
983
|
+
profile = MULTI_INTERACTION_BACKEND_PROFILES.get(optimizer_name)
|
|
984
|
+
if profile is not None:
|
|
985
|
+
return profile
|
|
986
|
+
return {
|
|
987
|
+
"allocation_kind": "custom_backend_search",
|
|
988
|
+
"roles": (optimizer_name,),
|
|
989
|
+
"role_archetypes": ("custom_optimizer",),
|
|
990
|
+
"path_prefixes": (),
|
|
991
|
+
"role_path_prefixes": {optimizer_name: ()},
|
|
992
|
+
}
|
|
993
|
+
|
|
994
|
+
|
|
995
|
+
def _allocation_profile_path_focus(
|
|
996
|
+
profile: Mapping[str, Any],
|
|
997
|
+
active_paths: Sequence[str],
|
|
998
|
+
) -> list[str]:
|
|
999
|
+
path_focus = _path_prefix_focus(active_paths, profile.get("path_prefixes", ()))
|
|
1000
|
+
return path_focus or list(dict.fromkeys(active_paths))
|
|
1001
|
+
|
|
1002
|
+
|
|
1003
|
+
def _allocation_role_path_focus(
|
|
1004
|
+
profile: Mapping[str, Any],
|
|
1005
|
+
path_focus: Sequence[str],
|
|
1006
|
+
) -> dict[str, list[str]]:
|
|
1007
|
+
role_path_prefixes = dict(profile.get("role_path_prefixes", {}))
|
|
1008
|
+
role_focus: dict[str, list[str]] = {}
|
|
1009
|
+
for role in profile.get("roles", ()):
|
|
1010
|
+
prefixes = role_path_prefixes.get(role, ())
|
|
1011
|
+
focused = _path_prefix_focus(path_focus, prefixes)
|
|
1012
|
+
role_focus[str(role)] = focused or list(path_focus)
|
|
1013
|
+
return role_focus
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
def _allocation_diagnosis_focus(
|
|
1017
|
+
*,
|
|
1018
|
+
profile: Mapping[str, Any],
|
|
1019
|
+
diagnoses: Sequence[ComponentDiagnosis],
|
|
1020
|
+
active_paths: Sequence[str],
|
|
1021
|
+
path_focus: Sequence[str],
|
|
1022
|
+
) -> list[dict[str, Any]]:
|
|
1023
|
+
path_focus_set = set(path_focus)
|
|
1024
|
+
profile_prefixes = tuple(str(prefix) for prefix in profile.get("path_prefixes", ()))
|
|
1025
|
+
rows: list[dict[str, Any]] = []
|
|
1026
|
+
for diagnosis in diagnoses:
|
|
1027
|
+
diagnosis_paths = _diagnosis_search_path_focus(diagnosis, active_paths)
|
|
1028
|
+
if (
|
|
1029
|
+
profile_prefixes
|
|
1030
|
+
and diagnosis_paths
|
|
1031
|
+
and path_focus_set
|
|
1032
|
+
and not path_focus_set.intersection(diagnosis_paths)
|
|
1033
|
+
):
|
|
1034
|
+
continue
|
|
1035
|
+
metrics = _diagnosis_metric_names(diagnosis)
|
|
1036
|
+
row: dict[str, Any] = {
|
|
1037
|
+
"component": diagnosis.component,
|
|
1038
|
+
"failure_mode": diagnosis.failure_mode,
|
|
1039
|
+
"confidence": round(float(diagnosis.confidence), 4),
|
|
1040
|
+
}
|
|
1041
|
+
if metrics:
|
|
1042
|
+
row["metrics"] = metrics
|
|
1043
|
+
if diagnosis_paths:
|
|
1044
|
+
row["paths"] = diagnosis_paths
|
|
1045
|
+
if diagnosis.patch_strategy:
|
|
1046
|
+
row["patch_strategy"] = diagnosis.patch_strategy
|
|
1047
|
+
if diagnosis.evidence:
|
|
1048
|
+
row["evidence"] = diagnosis.evidence
|
|
1049
|
+
rows.append(row)
|
|
1050
|
+
return rows
|
|
1051
|
+
|
|
1052
|
+
|
|
1053
|
+
def _diagnosis_search_path_focus(
|
|
1054
|
+
diagnosis: ComponentDiagnosis,
|
|
1055
|
+
active_paths: Sequence[str],
|
|
1056
|
+
) -> list[str]:
|
|
1057
|
+
prefixes = [str(path) for path in diagnosis.suggested_paths]
|
|
1058
|
+
prefixes.append(str(diagnosis.component))
|
|
1059
|
+
return _path_prefix_focus(active_paths, prefixes)
|
|
1060
|
+
|
|
1061
|
+
|
|
1062
|
+
def _path_prefix_focus(
|
|
1063
|
+
paths: Sequence[str],
|
|
1064
|
+
prefixes: Sequence[Any],
|
|
1065
|
+
) -> list[str]:
|
|
1066
|
+
unique_paths = list(dict.fromkeys(str(path) for path in paths))
|
|
1067
|
+
unique_prefixes = [str(prefix) for prefix in prefixes if str(prefix)]
|
|
1068
|
+
if not unique_prefixes:
|
|
1069
|
+
return unique_paths
|
|
1070
|
+
return [
|
|
1071
|
+
path
|
|
1072
|
+
for path in unique_paths
|
|
1073
|
+
if any(
|
|
1074
|
+
path == prefix or path.startswith(f"{prefix}.")
|
|
1075
|
+
for prefix in unique_prefixes
|
|
1076
|
+
)
|
|
1077
|
+
]
|
|
1078
|
+
|
|
1079
|
+
|
|
1080
|
+
def _diagnostic_metric_coverage(
|
|
1081
|
+
diagnoses: Sequence[ComponentDiagnosis],
|
|
1082
|
+
*,
|
|
1083
|
+
metric_names: Sequence[str],
|
|
1084
|
+
) -> list[str]:
|
|
1085
|
+
metrics = {str(metric) for metric in metric_names}
|
|
1086
|
+
for diagnosis in diagnoses:
|
|
1087
|
+
metrics.update(_diagnosis_metric_names(diagnosis))
|
|
1088
|
+
return sorted(metrics)
|
|
1089
|
+
|
|
1090
|
+
|
|
1091
|
+
def _diagnosis_metric_names(diagnosis: ComponentDiagnosis) -> list[str]:
|
|
1092
|
+
metrics: set[str] = set()
|
|
1093
|
+
metadata = dict(diagnosis.metadata or {})
|
|
1094
|
+
for key in ("metric", "metric_name", "name"):
|
|
1095
|
+
value = metadata.get(key)
|
|
1096
|
+
if value:
|
|
1097
|
+
metrics.add(str(value))
|
|
1098
|
+
for key in ("metric_result", "finding"):
|
|
1099
|
+
value = metadata.get(key)
|
|
1100
|
+
if isinstance(value, Mapping):
|
|
1101
|
+
for nested_key in ("metric", "metric_name", "name"):
|
|
1102
|
+
nested_value = value.get(nested_key)
|
|
1103
|
+
if nested_value:
|
|
1104
|
+
metrics.add(str(nested_value))
|
|
1105
|
+
return sorted(metrics)
|
|
1106
|
+
|
|
1107
|
+
|
|
1108
|
+
def _diagnosis_focus_metric_coverage(
|
|
1109
|
+
diagnosis_focus: Sequence[Mapping[str, Any]],
|
|
1110
|
+
) -> list[str]:
|
|
1111
|
+
metrics: set[str] = set()
|
|
1112
|
+
for row in diagnosis_focus:
|
|
1113
|
+
metrics.update(str(metric) for metric in row.get("metrics", ()))
|
|
1114
|
+
return sorted(metrics)
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
def _diagnosis_focus_values(
|
|
1118
|
+
diagnosis_focus: Sequence[Mapping[str, Any]],
|
|
1119
|
+
key: str,
|
|
1120
|
+
) -> list[str]:
|
|
1121
|
+
return sorted({str(row[key]) for row in diagnosis_focus if key in row})
|
|
1122
|
+
|
|
1123
|
+
|
|
1124
|
+
def _flatten_ledger_path_focus(ledger: Sequence[Mapping[str, Any]]) -> list[str]:
|
|
1125
|
+
paths: list[str] = []
|
|
1126
|
+
for entry in ledger:
|
|
1127
|
+
paths.extend(str(path) for path in entry.get("path_focus", ()))
|
|
1128
|
+
for role_paths in dict(entry.get("role_path_focus", {})).values():
|
|
1129
|
+
paths.extend(str(path) for path in role_paths)
|
|
1130
|
+
return list(dict.fromkeys(paths))
|
|
1131
|
+
|
|
1132
|
+
|
|
1133
|
+
def _diagnosis_coverage_keys(diagnoses: Sequence[ComponentDiagnosis]) -> list[str]:
|
|
1134
|
+
return sorted(
|
|
1135
|
+
{
|
|
1136
|
+
f"{diagnosis.component}:{diagnosis.failure_mode}"
|
|
1137
|
+
for diagnosis in diagnoses
|
|
1138
|
+
}
|
|
1139
|
+
)
|
|
1140
|
+
|
|
1141
|
+
|
|
1142
|
+
def _multi_interaction_backend_lineage(
|
|
1143
|
+
*,
|
|
1144
|
+
target: OptimizationTarget,
|
|
1145
|
+
plan: Sequence[AgentMultiInteractionBackendPlan],
|
|
1146
|
+
runs: Sequence[AgentMultiInteractionBackendRun],
|
|
1147
|
+
selected_run: AgentMultiInteractionBackendRun,
|
|
1148
|
+
) -> list[AgentMultiInteractionBackendLineage]:
|
|
1149
|
+
plan_by_optimizer = {allocation.optimizer: allocation for allocation in plan}
|
|
1150
|
+
selected_patch_signature = _patch_signature(
|
|
1151
|
+
_backend_run_candidate_patch(selected_run, target)
|
|
1152
|
+
)
|
|
1153
|
+
rows: list[AgentMultiInteractionBackendLineage] = []
|
|
1154
|
+
for run in runs:
|
|
1155
|
+
allocation = plan_by_optimizer.get(run.optimizer)
|
|
1156
|
+
reoptimization_result = (
|
|
1157
|
+
run.result.reoptimization_result if run.result is not None else None
|
|
1158
|
+
)
|
|
1159
|
+
candidate = (
|
|
1160
|
+
getattr(reoptimization_result, "best_candidate", None)
|
|
1161
|
+
if reoptimization_result is not None
|
|
1162
|
+
else None
|
|
1163
|
+
)
|
|
1164
|
+
candidate_patch = _candidate_contribution_patch(candidate, target)
|
|
1165
|
+
rows.append(
|
|
1166
|
+
AgentMultiInteractionBackendLineage(
|
|
1167
|
+
optimizer=run.optimizer,
|
|
1168
|
+
rank=run.rank,
|
|
1169
|
+
allocation_weight=allocation.weight if allocation else 0.0,
|
|
1170
|
+
allocation_reason=allocation.reason if allocation else "",
|
|
1171
|
+
status=run.status,
|
|
1172
|
+
final_score=run.final_score,
|
|
1173
|
+
improved=run.improved,
|
|
1174
|
+
total_evaluations=run.total_evaluations,
|
|
1175
|
+
candidate_id=getattr(candidate, "id", None),
|
|
1176
|
+
parent_candidate_id=getattr(candidate, "parent_id", None),
|
|
1177
|
+
candidate_patch=candidate_patch,
|
|
1178
|
+
patch_paths=_ordered_patch_paths(target, candidate_patch),
|
|
1179
|
+
metadata={
|
|
1180
|
+
"backend_strategy": (
|
|
1181
|
+
reoptimization_result.metadata.get("strategy")
|
|
1182
|
+
if reoptimization_result is not None
|
|
1183
|
+
else None
|
|
1184
|
+
),
|
|
1185
|
+
"backend_optimizer": (
|
|
1186
|
+
run.result.metadata.get("backend_optimizer")
|
|
1187
|
+
if run.result is not None
|
|
1188
|
+
else None
|
|
1189
|
+
),
|
|
1190
|
+
},
|
|
1191
|
+
)
|
|
1192
|
+
)
|
|
1193
|
+
|
|
1194
|
+
completed_rows = [row for row in rows if row.status == "completed"]
|
|
1195
|
+
patch_backends: dict[str, list[str]] = {}
|
|
1196
|
+
patch_value_backends: dict[tuple[str, str], list[str]] = {}
|
|
1197
|
+
for row in completed_rows:
|
|
1198
|
+
patch_backends.setdefault(_patch_signature(row.candidate_patch), []).append(
|
|
1199
|
+
row.optimizer
|
|
1200
|
+
)
|
|
1201
|
+
for path, value in row.candidate_patch.items():
|
|
1202
|
+
patch_value_backends.setdefault(
|
|
1203
|
+
(path, _value_signature(value)),
|
|
1204
|
+
[],
|
|
1205
|
+
).append(row.optimizer)
|
|
1206
|
+
|
|
1207
|
+
selected_patch = _backend_run_candidate_patch(selected_run, target)
|
|
1208
|
+
for row in rows:
|
|
1209
|
+
if row.status != "completed":
|
|
1210
|
+
row.selection_relation = "failed"
|
|
1211
|
+
continue
|
|
1212
|
+
|
|
1213
|
+
patch_signature = _patch_signature(row.candidate_patch)
|
|
1214
|
+
equivalent_backends = patch_backends.get(patch_signature, [])
|
|
1215
|
+
unique_patch: dict[str, Any] = {}
|
|
1216
|
+
shared_patch: dict[str, Any] = {}
|
|
1217
|
+
for path, value in row.candidate_patch.items():
|
|
1218
|
+
supporters = patch_value_backends.get((path, _value_signature(value)), [])
|
|
1219
|
+
if len(supporters) == 1:
|
|
1220
|
+
unique_patch[path] = value
|
|
1221
|
+
else:
|
|
1222
|
+
shared_patch[path] = value
|
|
1223
|
+
|
|
1224
|
+
row.equivalent_backends = list(equivalent_backends)
|
|
1225
|
+
row.equivalent_backend_count = len(equivalent_backends)
|
|
1226
|
+
row.unique_candidate_patch = unique_patch
|
|
1227
|
+
row.unique_patch_paths = _ordered_patch_paths(target, unique_patch)
|
|
1228
|
+
row.shared_candidate_patch = shared_patch
|
|
1229
|
+
row.shared_patch_paths = _ordered_patch_paths(target, shared_patch)
|
|
1230
|
+
if row.optimizer == selected_run.optimizer:
|
|
1231
|
+
row.selection_relation = "selected"
|
|
1232
|
+
elif patch_signature == selected_patch_signature:
|
|
1233
|
+
row.selection_relation = "consensus_peer"
|
|
1234
|
+
elif _patches_share_values(row.candidate_patch, selected_patch):
|
|
1235
|
+
row.selection_relation = "partial_support"
|
|
1236
|
+
else:
|
|
1237
|
+
row.selection_relation = "divergent"
|
|
1238
|
+
|
|
1239
|
+
return rows
|
|
1240
|
+
|
|
1241
|
+
|
|
1242
|
+
def _multi_interaction_ablation_report(
|
|
1243
|
+
*,
|
|
1244
|
+
lineage: Sequence[AgentMultiInteractionBackendLineage],
|
|
1245
|
+
selected_run: AgentMultiInteractionBackendRun,
|
|
1246
|
+
) -> AgentMultiInteractionAblationReport:
|
|
1247
|
+
selected_lineage = next(
|
|
1248
|
+
(row for row in lineage if row.optimizer == selected_run.optimizer),
|
|
1249
|
+
None,
|
|
1250
|
+
)
|
|
1251
|
+
selected_patch = selected_lineage.candidate_patch if selected_lineage else {}
|
|
1252
|
+
selected_signature = _patch_signature(selected_patch)
|
|
1253
|
+
final_score = float(selected_run.final_score or 0.0)
|
|
1254
|
+
completed = [row for row in lineage if row.status == "completed"]
|
|
1255
|
+
peers = [row for row in completed if row.optimizer != selected_run.optimizer]
|
|
1256
|
+
best_without_selected = (
|
|
1257
|
+
max(peers, key=_lineage_selection_key) if peers else None
|
|
1258
|
+
)
|
|
1259
|
+
score_delta_without_selected: Optional[float] = None
|
|
1260
|
+
if best_without_selected and best_without_selected.final_score is not None:
|
|
1261
|
+
score_delta_without_selected = round(
|
|
1262
|
+
final_score - best_without_selected.final_score,
|
|
1263
|
+
8,
|
|
1264
|
+
)
|
|
1265
|
+
|
|
1266
|
+
score_tolerance = 1e-9
|
|
1267
|
+
consensus_backends = [
|
|
1268
|
+
row.optimizer
|
|
1269
|
+
for row in completed
|
|
1270
|
+
if _patch_signature(row.candidate_patch) == selected_signature
|
|
1271
|
+
and row.final_score is not None
|
|
1272
|
+
and abs(row.final_score - final_score) <= score_tolerance
|
|
1273
|
+
]
|
|
1274
|
+
peer_reproduced_selected = any(
|
|
1275
|
+
optimizer != selected_run.optimizer for optimizer in consensus_backends
|
|
1276
|
+
)
|
|
1277
|
+
selected_backend_required = not peer_reproduced_selected
|
|
1278
|
+
|
|
1279
|
+
selected_patch_support: dict[str, list[str]] = {}
|
|
1280
|
+
for path, value in selected_patch.items():
|
|
1281
|
+
selected_patch_support[path] = [
|
|
1282
|
+
row.optimizer
|
|
1283
|
+
for row in completed
|
|
1284
|
+
if path in row.candidate_patch
|
|
1285
|
+
and _value_signature(row.candidate_patch[path]) == _value_signature(value)
|
|
1286
|
+
]
|
|
1287
|
+
shared_selected_patch_paths = [
|
|
1288
|
+
path for path, supporters in selected_patch_support.items() if len(supporters) > 1
|
|
1289
|
+
]
|
|
1290
|
+
unique_selected_patch_paths = [
|
|
1291
|
+
path for path, supporters in selected_patch_support.items() if len(supporters) == 1
|
|
1292
|
+
]
|
|
1293
|
+
|
|
1294
|
+
if best_without_selected is None:
|
|
1295
|
+
dependency = "single_backend_only"
|
|
1296
|
+
dependency_reason = "No other backend completed, so no leave-one-out comparison exists."
|
|
1297
|
+
elif peer_reproduced_selected:
|
|
1298
|
+
dependency = "backend_consensus"
|
|
1299
|
+
dependency_reason = (
|
|
1300
|
+
"At least one other backend reproduced the selected patch at the same score."
|
|
1301
|
+
)
|
|
1302
|
+
elif (
|
|
1303
|
+
best_without_selected.final_score is not None
|
|
1304
|
+
and abs(final_score - best_without_selected.final_score) <= score_tolerance
|
|
1305
|
+
):
|
|
1306
|
+
dependency = "score_tie_different_patch"
|
|
1307
|
+
dependency_reason = (
|
|
1308
|
+
"Removing the selected backend preserves the score, but the best peer "
|
|
1309
|
+
"uses a different patch."
|
|
1310
|
+
)
|
|
1311
|
+
else:
|
|
1312
|
+
dependency = "selected_backend_dependent"
|
|
1313
|
+
dependency_reason = (
|
|
1314
|
+
"Removing the selected backend lowers the best observed portfolio score."
|
|
1315
|
+
)
|
|
1316
|
+
|
|
1317
|
+
return AgentMultiInteractionAblationReport(
|
|
1318
|
+
selected_optimizer=selected_run.optimizer,
|
|
1319
|
+
selected_candidate_id=selected_lineage.candidate_id if selected_lineage else None,
|
|
1320
|
+
selected_patch=selected_patch,
|
|
1321
|
+
selected_patch_paths=(
|
|
1322
|
+
list(selected_lineage.patch_paths) if selected_lineage else []
|
|
1323
|
+
),
|
|
1324
|
+
final_score=final_score,
|
|
1325
|
+
best_without_selected_optimizer=(
|
|
1326
|
+
best_without_selected.optimizer if best_without_selected else None
|
|
1327
|
+
),
|
|
1328
|
+
best_without_selected_score=(
|
|
1329
|
+
best_without_selected.final_score if best_without_selected else None
|
|
1330
|
+
),
|
|
1331
|
+
score_delta_without_selected=score_delta_without_selected,
|
|
1332
|
+
selected_backend_required=selected_backend_required,
|
|
1333
|
+
dependency=dependency,
|
|
1334
|
+
dependency_reason=dependency_reason,
|
|
1335
|
+
consensus_backends=consensus_backends,
|
|
1336
|
+
consensus_backend_count=len(consensus_backends),
|
|
1337
|
+
shared_selected_patch_paths=_ordered_patch_paths_for_keys(
|
|
1338
|
+
shared_selected_patch_paths,
|
|
1339
|
+
list(selected_patch),
|
|
1340
|
+
),
|
|
1341
|
+
unique_selected_patch_paths=_ordered_patch_paths_for_keys(
|
|
1342
|
+
unique_selected_patch_paths,
|
|
1343
|
+
list(selected_patch),
|
|
1344
|
+
),
|
|
1345
|
+
selected_patch_support={
|
|
1346
|
+
path: selected_patch_support[path]
|
|
1347
|
+
for path in _ordered_patch_paths_for_keys(
|
|
1348
|
+
selected_patch_support,
|
|
1349
|
+
list(selected_patch),
|
|
1350
|
+
)
|
|
1351
|
+
},
|
|
1352
|
+
backend_scoreboard=[
|
|
1353
|
+
{
|
|
1354
|
+
"optimizer": row.optimizer,
|
|
1355
|
+
"rank": row.rank,
|
|
1356
|
+
"status": row.status,
|
|
1357
|
+
"final_score": row.final_score,
|
|
1358
|
+
"improved": row.improved,
|
|
1359
|
+
"candidate_id": row.candidate_id,
|
|
1360
|
+
"patch_paths": list(row.patch_paths),
|
|
1361
|
+
"selection_relation": row.selection_relation,
|
|
1362
|
+
}
|
|
1363
|
+
for row in sorted(completed, key=_lineage_selection_key, reverse=True)
|
|
1364
|
+
],
|
|
1365
|
+
)
|
|
1366
|
+
|
|
1367
|
+
|
|
1368
|
+
def _backend_run_candidate_patch(
|
|
1369
|
+
run: AgentMultiInteractionBackendRun,
|
|
1370
|
+
target: OptimizationTarget,
|
|
1371
|
+
) -> dict[str, Any]:
|
|
1372
|
+
if run.result is None:
|
|
1373
|
+
return {}
|
|
1374
|
+
return _candidate_contribution_patch(
|
|
1375
|
+
run.result.reoptimization_result.best_candidate,
|
|
1376
|
+
target,
|
|
1377
|
+
)
|
|
1378
|
+
|
|
1379
|
+
|
|
1380
|
+
def _candidate_contribution_patch(
|
|
1381
|
+
candidate: Any,
|
|
1382
|
+
target: OptimizationTarget,
|
|
1383
|
+
) -> dict[str, Any]:
|
|
1384
|
+
if candidate is None:
|
|
1385
|
+
return {}
|
|
1386
|
+
base_candidate = target.seed_candidate()
|
|
1387
|
+
changed = {
|
|
1388
|
+
path: candidate.get_path(path)
|
|
1389
|
+
for path in target.search_space
|
|
1390
|
+
if candidate.get_path(path) != base_candidate.get_path(path)
|
|
1391
|
+
}
|
|
1392
|
+
if changed:
|
|
1393
|
+
return changed
|
|
1394
|
+
raw_patch = getattr(candidate, "patch", None)
|
|
1395
|
+
return dict(raw_patch or {})
|
|
1396
|
+
|
|
1397
|
+
|
|
1398
|
+
def _patch_signature(patch: Mapping[str, Any]) -> str:
|
|
1399
|
+
return json.dumps(dict(patch), sort_keys=True, default=str, separators=(",", ":"))
|
|
1400
|
+
|
|
1401
|
+
|
|
1402
|
+
def _value_signature(value: Any) -> str:
|
|
1403
|
+
return json.dumps(value, sort_keys=True, default=str, separators=(",", ":"))
|
|
1404
|
+
|
|
1405
|
+
|
|
1406
|
+
def _patches_share_values(
|
|
1407
|
+
first: Mapping[str, Any],
|
|
1408
|
+
second: Mapping[str, Any],
|
|
1409
|
+
) -> bool:
|
|
1410
|
+
return any(
|
|
1411
|
+
path in second and _value_signature(value) == _value_signature(second[path])
|
|
1412
|
+
for path, value in first.items()
|
|
1413
|
+
)
|
|
1414
|
+
|
|
1415
|
+
|
|
1416
|
+
def _ordered_patch_paths(
|
|
1417
|
+
target: OptimizationTarget,
|
|
1418
|
+
patch: Mapping[str, Any],
|
|
1419
|
+
) -> list[str]:
|
|
1420
|
+
ordered = [path for path in target.search_space if path in patch]
|
|
1421
|
+
ordered.extend(path for path in patch if path not in target.search_space)
|
|
1422
|
+
return ordered
|
|
1423
|
+
|
|
1424
|
+
|
|
1425
|
+
def _ordered_patch_paths_for_keys(
|
|
1426
|
+
paths: Iterable[str],
|
|
1427
|
+
order: Sequence[str],
|
|
1428
|
+
) -> list[str]:
|
|
1429
|
+
path_set = set(paths)
|
|
1430
|
+
ordered = [path for path in order if path in path_set]
|
|
1431
|
+
ordered.extend(sorted(path for path in path_set if path not in order))
|
|
1432
|
+
return ordered
|
|
1433
|
+
|
|
1434
|
+
|
|
1435
|
+
def _lineage_selection_key(
|
|
1436
|
+
row: AgentMultiInteractionBackendLineage,
|
|
1437
|
+
) -> tuple[float, int, int, int]:
|
|
1438
|
+
return (
|
|
1439
|
+
row.final_score if row.final_score is not None else float("-inf"),
|
|
1440
|
+
1 if row.improved else 0,
|
|
1441
|
+
-row.rank,
|
|
1442
|
+
-row.total_evaluations,
|
|
1443
|
+
)
|
|
1444
|
+
|
|
1445
|
+
|
|
1446
|
+
def _dedupe_optimizer_pool(pool: Sequence[str]) -> list[str]:
|
|
1447
|
+
seen: set[str] = set()
|
|
1448
|
+
names: list[str] = []
|
|
1449
|
+
for item in pool:
|
|
1450
|
+
normalized = _normalize_optimizer_name(str(item))
|
|
1451
|
+
if normalized in seen:
|
|
1452
|
+
continue
|
|
1453
|
+
seen.add(normalized)
|
|
1454
|
+
names.append(normalized)
|
|
1455
|
+
return names
|
|
1456
|
+
|
|
1457
|
+
|
|
1458
|
+
def _backend_kwargs_for_multi_interaction(
|
|
1459
|
+
optimizer_name: str,
|
|
1460
|
+
*,
|
|
1461
|
+
target: OptimizationTarget,
|
|
1462
|
+
metric_names: Sequence[str],
|
|
1463
|
+
optimizer_kwargs: Mapping[str, Any],
|
|
1464
|
+
optimizer_kwargs_by_backend: Mapping[str, Mapping[str, Any]],
|
|
1465
|
+
) -> dict[str, Any]:
|
|
1466
|
+
defaults: dict[str, Any] = {}
|
|
1467
|
+
target_score = float(optimizer_kwargs.get("target_score", 0.99))
|
|
1468
|
+
if optimizer_name == "agent":
|
|
1469
|
+
defaults.update({"max_candidates": 16})
|
|
1470
|
+
elif optimizer_name in {"council", "society"}:
|
|
1471
|
+
defaults.update(
|
|
1472
|
+
{
|
|
1473
|
+
"max_rounds": 2,
|
|
1474
|
+
"beam_width": 4,
|
|
1475
|
+
"max_proposals_per_round": 16,
|
|
1476
|
+
"target_score": target_score,
|
|
1477
|
+
}
|
|
1478
|
+
)
|
|
1479
|
+
elif optimizer_name == "social_memory":
|
|
1480
|
+
defaults.update(
|
|
1481
|
+
{
|
|
1482
|
+
"max_rounds": 2,
|
|
1483
|
+
"beam_width": 4,
|
|
1484
|
+
"max_proposals_per_round": 16,
|
|
1485
|
+
"target_score": target_score,
|
|
1486
|
+
}
|
|
1487
|
+
)
|
|
1488
|
+
elif optimizer_name == "curriculum":
|
|
1489
|
+
defaults.update({"max_candidates_per_stage": 8, "target_score": target_score})
|
|
1490
|
+
elif optimizer_name == "evolution":
|
|
1491
|
+
defaults.update(
|
|
1492
|
+
{
|
|
1493
|
+
"population_size": min(10, max(4, len(target.search_space) * 2)),
|
|
1494
|
+
"generations": 2,
|
|
1495
|
+
"elite_count": 2,
|
|
1496
|
+
"seed": 42,
|
|
1497
|
+
"target_score": target_score,
|
|
1498
|
+
}
|
|
1499
|
+
)
|
|
1500
|
+
elif optimizer_name == "tpe":
|
|
1501
|
+
defaults.update({"n_trials": 8, "seed": 42, "target_score": target_score})
|
|
1502
|
+
elif optimizer_name == "pareto":
|
|
1503
|
+
defaults.update({"n_trials": 8, "seed": 42, "target_score": target_score})
|
|
1504
|
+
if metric_names:
|
|
1505
|
+
defaults["objective_names"] = list(metric_names[:4])
|
|
1506
|
+
elif optimizer_name == "bandit":
|
|
1507
|
+
defaults.update(
|
|
1508
|
+
{
|
|
1509
|
+
"max_candidates": 8,
|
|
1510
|
+
"total_budget": 12,
|
|
1511
|
+
"selection": "best",
|
|
1512
|
+
"target_score": target_score,
|
|
1513
|
+
}
|
|
1514
|
+
)
|
|
1515
|
+
shared_keys = {"include_seed", "auto_diagnose", "diagnostic_score_threshold"}
|
|
1516
|
+
shared_kwargs = {
|
|
1517
|
+
key: value
|
|
1518
|
+
for key, value in dict(optimizer_kwargs).items()
|
|
1519
|
+
if key in shared_keys
|
|
1520
|
+
}
|
|
1521
|
+
if optimizer_name != "agent" and "target_score" in optimizer_kwargs:
|
|
1522
|
+
shared_kwargs["target_score"] = optimizer_kwargs["target_score"]
|
|
1523
|
+
combined = {
|
|
1524
|
+
**defaults,
|
|
1525
|
+
**shared_kwargs,
|
|
1526
|
+
**dict(optimizer_kwargs_by_backend.get(optimizer_name, {})),
|
|
1527
|
+
}
|
|
1528
|
+
return combined
|
|
1529
|
+
|
|
1530
|
+
|
|
1531
|
+
def _backend_allocation_weight(
|
|
1532
|
+
optimizer_name: str,
|
|
1533
|
+
*,
|
|
1534
|
+
target: OptimizationTarget,
|
|
1535
|
+
feedback_cases: Sequence[AgentFeedbackCase],
|
|
1536
|
+
diagnoses: Sequence[ComponentDiagnosis],
|
|
1537
|
+
search_paths: Sequence[str],
|
|
1538
|
+
metric_names: Sequence[str],
|
|
1539
|
+
) -> tuple[float, str]:
|
|
1540
|
+
layers = set(target.layers)
|
|
1541
|
+
text = " ".join(
|
|
1542
|
+
[
|
|
1543
|
+
" ".join(layers),
|
|
1544
|
+
" ".join(search_paths),
|
|
1545
|
+
" ".join(metric_names),
|
|
1546
|
+
" ".join(diagnosis.component for diagnosis in diagnoses),
|
|
1547
|
+
" ".join(diagnosis.failure_mode for diagnosis in diagnoses),
|
|
1548
|
+
]
|
|
1549
|
+
).lower()
|
|
1550
|
+
path_count = len(search_paths) if search_paths else len(target.search_space)
|
|
1551
|
+
metric_count = len(metric_names)
|
|
1552
|
+
failed_count = sum(1 for case in feedback_cases if not case.passed)
|
|
1553
|
+
candidate_space_size = _target_search_space_cardinality(target)
|
|
1554
|
+
architecture_config_signal = _architecture_config_signal(text)
|
|
1555
|
+
|
|
1556
|
+
weights = {
|
|
1557
|
+
"agent": 0.25,
|
|
1558
|
+
"curriculum": 0.55,
|
|
1559
|
+
"council": 0.6,
|
|
1560
|
+
"society": 0.65,
|
|
1561
|
+
"social_memory": 0.55,
|
|
1562
|
+
"evolution": 0.5,
|
|
1563
|
+
"pareto": 0.45,
|
|
1564
|
+
"tpe": 0.4,
|
|
1565
|
+
"bandit": 0.4,
|
|
1566
|
+
}
|
|
1567
|
+
reasons: list[str] = []
|
|
1568
|
+
weight = weights.get(optimizer_name, 0.1)
|
|
1569
|
+
if failed_count:
|
|
1570
|
+
weight += 0.1
|
|
1571
|
+
reasons.append(f"{failed_count} failing feedback case(s)")
|
|
1572
|
+
if optimizer_name == "agent":
|
|
1573
|
+
if 0 < path_count <= 3:
|
|
1574
|
+
weight += 0.45
|
|
1575
|
+
reasons.append("focused deterministic diagnosis search")
|
|
1576
|
+
if target.search_space and candidate_space_size <= 32:
|
|
1577
|
+
weight += 0.25
|
|
1578
|
+
reasons.append(f"exact categorical search space size {candidate_space_size}")
|
|
1579
|
+
if architecture_config_signal:
|
|
1580
|
+
weight += 0.35
|
|
1581
|
+
reasons.append("architecture/config signal")
|
|
1582
|
+
if path_count > 1:
|
|
1583
|
+
if optimizer_name in {"council", "society", "evolution"}:
|
|
1584
|
+
weight += 0.3
|
|
1585
|
+
if optimizer_name in {"curriculum", "social_memory"}:
|
|
1586
|
+
weight += 0.15
|
|
1587
|
+
reasons.append(f"{path_count} diagnosed search paths")
|
|
1588
|
+
if metric_count > 1:
|
|
1589
|
+
if optimizer_name in {"curriculum", "pareto"}:
|
|
1590
|
+
weight += 0.35
|
|
1591
|
+
if optimizer_name in {"society", "council", "social_memory"}:
|
|
1592
|
+
weight += 0.15
|
|
1593
|
+
reasons.append(f"{metric_count} failed metrics")
|
|
1594
|
+
if any(token in text for token in ("multi_agent", "handoff", "coordination", "review")):
|
|
1595
|
+
if optimizer_name in {"society", "council"}:
|
|
1596
|
+
weight += 0.45
|
|
1597
|
+
if optimizer_name == "social_memory":
|
|
1598
|
+
weight += 0.15
|
|
1599
|
+
reasons.append("multi-agent coordination signal")
|
|
1600
|
+
if "memory" in text or "cross_trial" in text:
|
|
1601
|
+
if optimizer_name == "social_memory":
|
|
1602
|
+
weight += 0.45
|
|
1603
|
+
if optimizer_name in {"society", "council"}:
|
|
1604
|
+
weight += 0.1
|
|
1605
|
+
reasons.append("memory/history signal")
|
|
1606
|
+
if "policy" in text or "security" in text:
|
|
1607
|
+
if optimizer_name in {"agent", "evolution", "bandit"}:
|
|
1608
|
+
weight += 0.15
|
|
1609
|
+
reasons.append("policy/security signal")
|
|
1610
|
+
if len(target.search_space) >= 6:
|
|
1611
|
+
if optimizer_name in {"tpe", "evolution"}:
|
|
1612
|
+
weight += 0.3
|
|
1613
|
+
if optimizer_name == "bandit":
|
|
1614
|
+
weight += 0.15
|
|
1615
|
+
reasons.append("larger categorical search space")
|
|
1616
|
+
if len(feedback_cases) > 1:
|
|
1617
|
+
if optimizer_name in {"bandit", "social_memory", "curriculum"}:
|
|
1618
|
+
weight += 0.2
|
|
1619
|
+
reasons.append("multi-observation replay window")
|
|
1620
|
+
if not reasons:
|
|
1621
|
+
reasons.append("deterministic fallback allocation")
|
|
1622
|
+
return weight, "; ".join(dict.fromkeys(reasons))
|
|
1623
|
+
|
|
1624
|
+
|
|
1625
|
+
def _target_search_space_cardinality(target: OptimizationTarget) -> int:
|
|
1626
|
+
total = 1
|
|
1627
|
+
for values in target.search_space.values():
|
|
1628
|
+
if isinstance(values, (list, tuple, set)):
|
|
1629
|
+
total *= max(1, len(values))
|
|
1630
|
+
else:
|
|
1631
|
+
total *= 1
|
|
1632
|
+
if total > 1_000_000:
|
|
1633
|
+
return total
|
|
1634
|
+
return total
|
|
1635
|
+
|
|
1636
|
+
|
|
1637
|
+
def _architecture_config_signal(text: str) -> bool:
|
|
1638
|
+
return any(
|
|
1639
|
+
token in text
|
|
1640
|
+
for token in (
|
|
1641
|
+
"architecture",
|
|
1642
|
+
"config",
|
|
1643
|
+
"framework",
|
|
1644
|
+
"adapter",
|
|
1645
|
+
"trace",
|
|
1646
|
+
"event_stream",
|
|
1647
|
+
"streaming",
|
|
1648
|
+
"orchestration",
|
|
1649
|
+
"workflow",
|
|
1650
|
+
"runtime",
|
|
1651
|
+
"instrumentation",
|
|
1652
|
+
"otel",
|
|
1653
|
+
"langchain",
|
|
1654
|
+
"langgraph",
|
|
1655
|
+
"openai_agents",
|
|
1656
|
+
"pipecat",
|
|
1657
|
+
"livekit",
|
|
1658
|
+
)
|
|
1659
|
+
)
|
|
1660
|
+
|
|
1661
|
+
|
|
1662
|
+
def _feedback_metric_names(feedback_cases: Sequence[AgentFeedbackCase]) -> list[str]:
|
|
1663
|
+
names: set[str] = set()
|
|
1664
|
+
for case in feedback_cases:
|
|
1665
|
+
names.update(str(key) for key in case.metrics.keys())
|
|
1666
|
+
for failure in case.failures:
|
|
1667
|
+
names.update(_METRIC_NAME_RE.findall(failure))
|
|
1668
|
+
return sorted(names)
|
|
1669
|
+
|
|
1670
|
+
|
|
1671
|
+
def _failed_feedback_metric_names(feedback_cases: Sequence[AgentFeedbackCase]) -> list[str]:
|
|
1672
|
+
names: set[str] = set()
|
|
1673
|
+
for case in feedback_cases:
|
|
1674
|
+
for failure in case.failures:
|
|
1675
|
+
names.update(_METRIC_NAME_RE.findall(failure))
|
|
1676
|
+
return sorted(names)
|
|
1677
|
+
|
|
1678
|
+
|
|
1679
|
+
def _resolve_rollback_decision(
|
|
1680
|
+
*,
|
|
1681
|
+
rollback_decision: Optional[AgentRollbackDecision],
|
|
1682
|
+
deployment: Optional[DeploymentLike],
|
|
1683
|
+
live_evaluations: Optional[Sequence[Any]],
|
|
1684
|
+
simulation_evaluator: Any,
|
|
1685
|
+
rollback_kwargs: Mapping[str, Any],
|
|
1686
|
+
) -> tuple[AgentRollbackDecision, str]:
|
|
1687
|
+
if rollback_decision is not None:
|
|
1688
|
+
return rollback_decision, "rollback_decision"
|
|
1689
|
+
if deployment is None:
|
|
1690
|
+
raise ValueError(
|
|
1691
|
+
"AgentFeedbackOptimizer requires deployment or rollback_decision."
|
|
1692
|
+
)
|
|
1693
|
+
decision = check_agent_deployment_rollback(
|
|
1694
|
+
deployment,
|
|
1695
|
+
live_evaluations=(
|
|
1696
|
+
list(live_evaluations) if live_evaluations is not None else None
|
|
1697
|
+
),
|
|
1698
|
+
simulation_evaluator=simulation_evaluator,
|
|
1699
|
+
**dict(rollback_kwargs),
|
|
1700
|
+
)
|
|
1701
|
+
source = "live_evaluations" if live_evaluations is not None else "simulation_replay"
|
|
1702
|
+
return decision, source
|
|
1703
|
+
|
|
1704
|
+
|
|
1705
|
+
def _feedback_cases_from_rollback(
|
|
1706
|
+
decision: AgentRollbackDecision,
|
|
1707
|
+
) -> list[AgentFeedbackCase]:
|
|
1708
|
+
return [
|
|
1709
|
+
AgentFeedbackCase(
|
|
1710
|
+
index=observation.index,
|
|
1711
|
+
candidate_id=observation.candidate_id,
|
|
1712
|
+
score=observation.score,
|
|
1713
|
+
passed=observation.passed,
|
|
1714
|
+
failures=list(observation.failures),
|
|
1715
|
+
metrics=dict(observation.metrics),
|
|
1716
|
+
metadata={
|
|
1717
|
+
**dict(observation.metadata),
|
|
1718
|
+
"rollback_required": decision.rollback_required,
|
|
1719
|
+
},
|
|
1720
|
+
)
|
|
1721
|
+
for observation in decision.observations
|
|
1722
|
+
]
|
|
1723
|
+
|
|
1724
|
+
|
|
1725
|
+
def _diagnose_feedback_cases(
|
|
1726
|
+
feedback_cases: Sequence[AgentFeedbackCase],
|
|
1727
|
+
*,
|
|
1728
|
+
target: OptimizationTarget,
|
|
1729
|
+
failing_threshold: float,
|
|
1730
|
+
) -> list[ComponentDiagnosis]:
|
|
1731
|
+
diagnostics: list[ComponentDiagnosis] = []
|
|
1732
|
+
failed_cases = [case for case in feedback_cases if not case.passed]
|
|
1733
|
+
for case in failed_cases:
|
|
1734
|
+
diagnostics.extend(
|
|
1735
|
+
diagnose_agent_report_evaluation(
|
|
1736
|
+
_agent_report_from_feedback_case(case),
|
|
1737
|
+
failing_threshold=failing_threshold,
|
|
1738
|
+
confidence=0.9,
|
|
1739
|
+
)
|
|
1740
|
+
)
|
|
1741
|
+
for failure in case.failures:
|
|
1742
|
+
diagnostics.extend(diagnose_text(failure, confidence=0.75))
|
|
1743
|
+
|
|
1744
|
+
if not diagnostics and failed_cases:
|
|
1745
|
+
diagnostics.append(
|
|
1746
|
+
ComponentDiagnosis(
|
|
1747
|
+
component="custom",
|
|
1748
|
+
failure_mode="unknown",
|
|
1749
|
+
confidence=0.5,
|
|
1750
|
+
evidence="Live feedback score regression without metric-specific diagnosis.",
|
|
1751
|
+
suggested_paths=list(target.search_space),
|
|
1752
|
+
metadata={"failed_feedback_cases": len(failed_cases)},
|
|
1753
|
+
)
|
|
1754
|
+
)
|
|
1755
|
+
return _dedupe_diagnoses(diagnostics)
|
|
1756
|
+
|
|
1757
|
+
|
|
1758
|
+
def _agent_report_from_feedback_case(case: AgentFeedbackCase) -> dict[str, Any]:
|
|
1759
|
+
metrics = dict(case.metrics)
|
|
1760
|
+
for failure in case.failures:
|
|
1761
|
+
for metric_name in _METRIC_NAME_RE.findall(failure):
|
|
1762
|
+
metrics.setdefault(metric_name, 0.0)
|
|
1763
|
+
return {
|
|
1764
|
+
"summary": {"metric_averages": metrics},
|
|
1765
|
+
"cases": [
|
|
1766
|
+
{
|
|
1767
|
+
"id": f"feedback-{case.index}",
|
|
1768
|
+
"metrics": [
|
|
1769
|
+
{
|
|
1770
|
+
"name": name,
|
|
1771
|
+
"score": score,
|
|
1772
|
+
"reason": "; ".join(case.failures),
|
|
1773
|
+
}
|
|
1774
|
+
for name, score in metrics.items()
|
|
1775
|
+
],
|
|
1776
|
+
"findings": [
|
|
1777
|
+
{
|
|
1778
|
+
"metric": name,
|
|
1779
|
+
"score": score,
|
|
1780
|
+
"evidence": "; ".join(case.failures),
|
|
1781
|
+
}
|
|
1782
|
+
for name, score in metrics.items()
|
|
1783
|
+
],
|
|
1784
|
+
}
|
|
1785
|
+
],
|
|
1786
|
+
}
|
|
1787
|
+
|
|
1788
|
+
|
|
1789
|
+
_METRIC_NAME_RE = re.compile(r"metric '([^']+)'")
|
|
1790
|
+
|
|
1791
|
+
|
|
1792
|
+
def _search_paths_for_feedback(
|
|
1793
|
+
target: OptimizationTarget,
|
|
1794
|
+
diagnoses: Sequence[ComponentDiagnosis],
|
|
1795
|
+
) -> list[str]:
|
|
1796
|
+
allowed_paths = relevant_search_paths(target.search_space, diagnoses)
|
|
1797
|
+
return [path for path in target.search_space if path in allowed_paths]
|
|
1798
|
+
|
|
1799
|
+
|
|
1800
|
+
def _resolve_feedback_optimizer(name: str) -> type:
|
|
1801
|
+
normalized = _normalize_optimizer_name(name)
|
|
1802
|
+
optimizers = {
|
|
1803
|
+
"agent": AgentOptimizer,
|
|
1804
|
+
"deterministic": AgentOptimizer,
|
|
1805
|
+
"council": CouncilAgentOptimizer,
|
|
1806
|
+
"society": SocietyAgentOptimizer,
|
|
1807
|
+
"social_memory": AgentSocialMemoryOptimizer,
|
|
1808
|
+
"curriculum": AgentCurriculumOptimizer,
|
|
1809
|
+
"evolution": AgentEvolutionOptimizer,
|
|
1810
|
+
"tpe": AgentTPEOptimizer,
|
|
1811
|
+
"pareto": AgentParetoOptimizer,
|
|
1812
|
+
"bandit": AgentBanditOptimizer,
|
|
1813
|
+
}
|
|
1814
|
+
if normalized not in optimizers:
|
|
1815
|
+
raise ValueError(
|
|
1816
|
+
"optimizer must be one of: agent, deterministic, council, society, "
|
|
1817
|
+
"social_memory, curriculum, evolution, tpe, pareto, or bandit."
|
|
1818
|
+
)
|
|
1819
|
+
return optimizers[normalized]
|
|
1820
|
+
|
|
1821
|
+
|
|
1822
|
+
def _normalize_optimizer_name(name: str) -> str:
|
|
1823
|
+
normalized = name.strip().lower().replace("-", "_")
|
|
1824
|
+
aliases = {
|
|
1825
|
+
"agentoptimizer": "agent",
|
|
1826
|
+
"agent_optimizer": "agent",
|
|
1827
|
+
"deterministic_agent_optimizer": "deterministic",
|
|
1828
|
+
"councilagentoptimizer": "council",
|
|
1829
|
+
"council_agent_optimizer": "council",
|
|
1830
|
+
"societyagentoptimizer": "society",
|
|
1831
|
+
"society_agent_optimizer": "society",
|
|
1832
|
+
"agentsocialmemoryoptimizer": "social_memory",
|
|
1833
|
+
"agent_social_memory_optimizer": "social_memory",
|
|
1834
|
+
"socialmemory": "social_memory",
|
|
1835
|
+
"social_memory_optimizer": "social_memory",
|
|
1836
|
+
"futureagi_social_memory": "social_memory",
|
|
1837
|
+
"futureagi_social_memory_optimizer": "social_memory",
|
|
1838
|
+
"agentcurriculumoptimizer": "curriculum",
|
|
1839
|
+
"agent_curriculum_optimizer": "curriculum",
|
|
1840
|
+
"curriculum_optimizer": "curriculum",
|
|
1841
|
+
"deliberate_practice": "curriculum",
|
|
1842
|
+
"deliberate_practice_curriculum": "curriculum",
|
|
1843
|
+
"agentevolutionoptimizer": "evolution",
|
|
1844
|
+
"agent_evolution_optimizer": "evolution",
|
|
1845
|
+
"agenttpeoptimizer": "tpe",
|
|
1846
|
+
"agent_tpe_optimizer": "tpe",
|
|
1847
|
+
"agentparetooptimizer": "pareto",
|
|
1848
|
+
"agent_pareto_optimizer": "pareto",
|
|
1849
|
+
"agentbanditoptimizer": "bandit",
|
|
1850
|
+
"agent_bandit_optimizer": "bandit",
|
|
1851
|
+
"agentmultiinteractionoptimizer": "multi_interaction",
|
|
1852
|
+
"agent_multi_interaction_optimizer": "multi_interaction",
|
|
1853
|
+
"multiinteraction": "multi_interaction",
|
|
1854
|
+
"multi_interaction_optimizer": "multi_interaction",
|
|
1855
|
+
"portfolio": "multi_interaction",
|
|
1856
|
+
"portfolio_optimizer": "multi_interaction",
|
|
1857
|
+
"auto": "multi_interaction",
|
|
1858
|
+
"auto_backend": "multi_interaction",
|
|
1859
|
+
}
|
|
1860
|
+
return aliases.get(
|
|
1861
|
+
normalized,
|
|
1862
|
+
normalized.replace("agent_", "").replace("_optimizer", "").replace("optimizer", ""),
|
|
1863
|
+
)
|