agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,1313 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import inspect
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Callable, Mapping, Optional, Sequence, Type
|
|
8
|
+
|
|
9
|
+
from ..optimizers.agent import AgentOptimizer
|
|
10
|
+
from ..optimizers.agent_evolution import AgentEvolutionOptimizer
|
|
11
|
+
from ..optimizers.agent_social_memory import AgentSocialMemoryOptimizer
|
|
12
|
+
from ..evidence import score_simulation_evidence
|
|
13
|
+
from ..simulation import _coerce_score, _iter_report_scores, _run_sync
|
|
14
|
+
from ..targets import (
|
|
15
|
+
AgentCandidate,
|
|
16
|
+
CandidateEvaluation,
|
|
17
|
+
OptimizationLayer,
|
|
18
|
+
OptimizationTarget,
|
|
19
|
+
set_path,
|
|
20
|
+
)
|
|
21
|
+
from ..types import EvaluationResult, OptimizationResult
|
|
22
|
+
|
|
23
|
+
ManifestRunner = Callable[[Mapping[str, Any], AgentCandidate], Any]
|
|
24
|
+
ManifestScorer = Callable[[Mapping[str, Any], Any, AgentCandidate], Any]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class SimulateManifestOptimizationProblem:
|
|
29
|
+
"""
|
|
30
|
+
Bridge portable simulation manifests into AgentOptimizer-style config search.
|
|
31
|
+
|
|
32
|
+
`base_manifest` is the runnable manifest without its `optimization` block.
|
|
33
|
+
Candidate configs are deep-merged into it, then `evaluate_manifest` runs the
|
|
34
|
+
real simulator/world/eval stack. The returned report, or the optional
|
|
35
|
+
`score_manifest` result, is normalized into `CandidateEvaluation`.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
base_manifest: Mapping[str, Any]
|
|
39
|
+
target: OptimizationTarget
|
|
40
|
+
evaluate_manifest: ManifestRunner
|
|
41
|
+
score_manifest: Optional[ManifestScorer] = None
|
|
42
|
+
evidence_scorer_config: Optional[Mapping[str, Any]] = None
|
|
43
|
+
threshold: float = 0.7
|
|
44
|
+
optimizer_kwargs: Mapping[str, Any] = field(default_factory=dict)
|
|
45
|
+
optimizer_cls: Type[Any] = AgentOptimizer
|
|
46
|
+
metadata: Mapping[str, Any] = field(default_factory=dict)
|
|
47
|
+
|
|
48
|
+
@classmethod
|
|
49
|
+
def from_manifest(
|
|
50
|
+
cls,
|
|
51
|
+
manifest: Mapping[str, Any],
|
|
52
|
+
*,
|
|
53
|
+
evaluate_manifest: ManifestRunner,
|
|
54
|
+
score_manifest: Optional[ManifestScorer] = None,
|
|
55
|
+
name: Optional[str] = None,
|
|
56
|
+
threshold: Optional[float] = None,
|
|
57
|
+
) -> "SimulateManifestOptimizationProblem":
|
|
58
|
+
optimization = _require_mapping(
|
|
59
|
+
manifest.get("optimization"),
|
|
60
|
+
"manifest.optimization",
|
|
61
|
+
)
|
|
62
|
+
target_config = _target_config(optimization)
|
|
63
|
+
optimizer_kwargs = _optimizer_kwargs(
|
|
64
|
+
_optional_mapping(optimization.get("optimizer"))
|
|
65
|
+
)
|
|
66
|
+
optimizer_cls = _optimizer_cls(_optional_mapping(optimization.get("optimizer")))
|
|
67
|
+
evidence_scorer_config = _evidence_scorer_config(
|
|
68
|
+
optimization,
|
|
69
|
+
target_config,
|
|
70
|
+
base_manifest=manifest,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
base_manifest = copy.deepcopy(dict(manifest))
|
|
74
|
+
base_manifest.pop("optimization", None)
|
|
75
|
+
|
|
76
|
+
manifest_name = str(name or manifest.get("name") or "agent-simulate-manifest")
|
|
77
|
+
target_metadata = copy.deepcopy(dict(target_config.get("metadata") or {}))
|
|
78
|
+
target_metadata.setdefault("source", "simulate_manifest")
|
|
79
|
+
target_metadata.setdefault("manifest_name", manifest_name)
|
|
80
|
+
|
|
81
|
+
target = OptimizationTarget(
|
|
82
|
+
name=str(target_config.get("name") or manifest_name),
|
|
83
|
+
layers=_layers(target_config.get("layers")),
|
|
84
|
+
base_config=copy.deepcopy(dict(target_config["base_config"])),
|
|
85
|
+
search_space=_search_space(target_config["search_space"]),
|
|
86
|
+
metadata=target_metadata,
|
|
87
|
+
)
|
|
88
|
+
return cls(
|
|
89
|
+
base_manifest=base_manifest,
|
|
90
|
+
target=target,
|
|
91
|
+
evaluate_manifest=evaluate_manifest,
|
|
92
|
+
score_manifest=score_manifest,
|
|
93
|
+
evidence_scorer_config=evidence_scorer_config,
|
|
94
|
+
threshold=float(
|
|
95
|
+
threshold
|
|
96
|
+
if threshold is not None
|
|
97
|
+
else optimization.get("threshold", 0.7)
|
|
98
|
+
),
|
|
99
|
+
optimizer_kwargs=optimizer_kwargs,
|
|
100
|
+
optimizer_cls=optimizer_cls,
|
|
101
|
+
metadata={
|
|
102
|
+
"source": "simulate_manifest",
|
|
103
|
+
"manifest_name": manifest_name,
|
|
104
|
+
"optimizer_algorithm": _optimizer_algorithm_name(optimizer_cls),
|
|
105
|
+
},
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
def candidate_manifest(self, candidate: AgentCandidate) -> dict[str, Any]:
|
|
109
|
+
merged = deep_merge(
|
|
110
|
+
copy.deepcopy(dict(self.base_manifest)),
|
|
111
|
+
copy.deepcopy(candidate.config),
|
|
112
|
+
)
|
|
113
|
+
_apply_candidate_patch_replacements(merged, candidate)
|
|
114
|
+
return merged
|
|
115
|
+
|
|
116
|
+
def evaluate_candidate(self, candidate: AgentCandidate) -> CandidateEvaluation:
|
|
117
|
+
candidate_manifest = self.candidate_manifest(candidate)
|
|
118
|
+
report = _run_sync(self.evaluate_manifest(candidate_manifest, candidate))
|
|
119
|
+
score_source = report
|
|
120
|
+
if self.score_manifest is not None:
|
|
121
|
+
score_source = _run_sync(
|
|
122
|
+
self.score_manifest(candidate_manifest, report, candidate)
|
|
123
|
+
)
|
|
124
|
+
elif self.evidence_scorer_config is not None and (
|
|
125
|
+
not self.evidence_scorer_config.get("_auto")
|
|
126
|
+
or not _report_has_score(report)
|
|
127
|
+
):
|
|
128
|
+
score_source = score_simulation_evidence(
|
|
129
|
+
report,
|
|
130
|
+
manifest=candidate_manifest,
|
|
131
|
+
candidate=candidate,
|
|
132
|
+
config=self.evidence_scorer_config,
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
metadata = {
|
|
136
|
+
**dict(self.metadata),
|
|
137
|
+
"candidate_manifest": copy.deepcopy(candidate_manifest),
|
|
138
|
+
"candidate_patch": copy.deepcopy(candidate.patch),
|
|
139
|
+
"patch": copy.deepcopy(candidate.patch),
|
|
140
|
+
"search_paths": list(candidate.metadata.get("search_paths", [])),
|
|
141
|
+
}
|
|
142
|
+
evaluation = _candidate_evaluation_from_value(
|
|
143
|
+
score_source,
|
|
144
|
+
candidate,
|
|
145
|
+
report=report,
|
|
146
|
+
metadata=metadata,
|
|
147
|
+
)
|
|
148
|
+
return evaluation
|
|
149
|
+
|
|
150
|
+
def build_optimizer(
|
|
151
|
+
self,
|
|
152
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
153
|
+
**optimizer_kwargs: Any,
|
|
154
|
+
) -> Any:
|
|
155
|
+
optimizer_cls = optimizer_cls or self.optimizer_cls
|
|
156
|
+
kwargs = {**dict(self.optimizer_kwargs), **optimizer_kwargs}
|
|
157
|
+
kwargs = _filter_optimizer_kwargs(optimizer_cls, kwargs)
|
|
158
|
+
return optimizer_cls(
|
|
159
|
+
target=self.target,
|
|
160
|
+
evaluate_candidate=self.evaluate_candidate,
|
|
161
|
+
**kwargs,
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
def optimize(
|
|
165
|
+
self,
|
|
166
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
167
|
+
**optimizer_kwargs: Any,
|
|
168
|
+
) -> OptimizationResult:
|
|
169
|
+
return _as_optimization_result(
|
|
170
|
+
self.build_optimizer(optimizer_cls, **optimizer_kwargs).optimize()
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
ManifestOptimizationProblem = SimulateManifestOptimizationProblem
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@dataclass
|
|
178
|
+
class SimulateEvalSuiteOptimizationProblem:
|
|
179
|
+
"""
|
|
180
|
+
Bridge promptfoo-style simulate-sdk eval suites into AgentOptimizer search.
|
|
181
|
+
|
|
182
|
+
Candidate configs are deep-merged into the eval-suite JSON/YAML contract,
|
|
183
|
+
then scored by simulate-sdk's public `run_eval_suite` API. This gives
|
|
184
|
+
optimizer users a local prompt/provider/test/assertion loop without writing
|
|
185
|
+
adapter glue.
|
|
186
|
+
"""
|
|
187
|
+
|
|
188
|
+
base_suite: Mapping[str, Any]
|
|
189
|
+
target: OptimizationTarget
|
|
190
|
+
run_suite: Callable[[Mapping[str, Any], AgentCandidate], Any]
|
|
191
|
+
threshold: float = 1.0
|
|
192
|
+
optimizer_kwargs: Mapping[str, Any] = field(default_factory=dict)
|
|
193
|
+
optimizer_cls: Type[Any] = AgentOptimizer
|
|
194
|
+
metadata: Mapping[str, Any] = field(default_factory=dict)
|
|
195
|
+
|
|
196
|
+
@classmethod
|
|
197
|
+
def from_suite(
|
|
198
|
+
cls,
|
|
199
|
+
suite: Mapping[str, Any],
|
|
200
|
+
*,
|
|
201
|
+
run_suite: Optional[Callable[[Mapping[str, Any], AgentCandidate], Any]] = None,
|
|
202
|
+
name: Optional[str] = None,
|
|
203
|
+
threshold: Optional[float] = None,
|
|
204
|
+
) -> "SimulateEvalSuiteOptimizationProblem":
|
|
205
|
+
optimization = _require_mapping(
|
|
206
|
+
suite.get("optimization"),
|
|
207
|
+
"suite.optimization",
|
|
208
|
+
)
|
|
209
|
+
target_config = _target_config(optimization)
|
|
210
|
+
optimizer_kwargs = _optimizer_kwargs(
|
|
211
|
+
_optional_mapping(optimization.get("optimizer"))
|
|
212
|
+
)
|
|
213
|
+
optimizer_cls = _optimizer_cls(_optional_mapping(optimization.get("optimizer")))
|
|
214
|
+
base_suite = copy.deepcopy(dict(suite))
|
|
215
|
+
base_suite.pop("optimization", None)
|
|
216
|
+
suite_name = str(name or suite.get("name") or "agent-simulate-eval-suite")
|
|
217
|
+
target_metadata = copy.deepcopy(dict(target_config.get("metadata") or {}))
|
|
218
|
+
target_metadata.setdefault("source", "simulate_eval_suite")
|
|
219
|
+
target_metadata.setdefault("suite_name", suite_name)
|
|
220
|
+
return cls(
|
|
221
|
+
base_suite=base_suite,
|
|
222
|
+
target=OptimizationTarget(
|
|
223
|
+
name=str(target_config.get("name") or suite_name),
|
|
224
|
+
layers=_layers(target_config.get("layers")),
|
|
225
|
+
base_config=copy.deepcopy(dict(target_config["base_config"])),
|
|
226
|
+
search_space=_search_space(target_config["search_space"]),
|
|
227
|
+
metadata=target_metadata,
|
|
228
|
+
),
|
|
229
|
+
run_suite=run_suite or _public_eval_suite_runner(),
|
|
230
|
+
threshold=float(
|
|
231
|
+
threshold
|
|
232
|
+
if threshold is not None
|
|
233
|
+
else optimization.get("threshold", 1.0)
|
|
234
|
+
),
|
|
235
|
+
optimizer_kwargs=optimizer_kwargs,
|
|
236
|
+
optimizer_cls=optimizer_cls,
|
|
237
|
+
metadata={
|
|
238
|
+
"source": "simulate_eval_suite",
|
|
239
|
+
"suite_name": suite_name,
|
|
240
|
+
"optimizer_algorithm": _optimizer_algorithm_name(optimizer_cls),
|
|
241
|
+
},
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
def candidate_suite(self, candidate: AgentCandidate) -> dict[str, Any]:
|
|
245
|
+
merged = deep_merge(
|
|
246
|
+
copy.deepcopy(dict(self.base_suite)),
|
|
247
|
+
copy.deepcopy(candidate.config),
|
|
248
|
+
)
|
|
249
|
+
_apply_candidate_patch_replacements(merged, candidate)
|
|
250
|
+
return merged
|
|
251
|
+
|
|
252
|
+
def evaluate_candidate(self, candidate: AgentCandidate) -> CandidateEvaluation:
|
|
253
|
+
candidate_suite = self.candidate_suite(candidate)
|
|
254
|
+
result = _run_sync(self.run_suite(candidate_suite, candidate))
|
|
255
|
+
metadata = {
|
|
256
|
+
**dict(self.metadata),
|
|
257
|
+
"candidate_suite": copy.deepcopy(candidate_suite),
|
|
258
|
+
"candidate_patch": copy.deepcopy(candidate.patch),
|
|
259
|
+
"patch": copy.deepcopy(candidate.patch),
|
|
260
|
+
"report": copy.deepcopy(result),
|
|
261
|
+
"search_paths": list(candidate.metadata.get("search_paths", [])),
|
|
262
|
+
}
|
|
263
|
+
return _candidate_evaluation_from_value(
|
|
264
|
+
result,
|
|
265
|
+
candidate,
|
|
266
|
+
report=result,
|
|
267
|
+
metadata=metadata,
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
def build_optimizer(
|
|
271
|
+
self,
|
|
272
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
273
|
+
**optimizer_kwargs: Any,
|
|
274
|
+
) -> Any:
|
|
275
|
+
optimizer_cls = optimizer_cls or self.optimizer_cls
|
|
276
|
+
kwargs = {**dict(self.optimizer_kwargs), **optimizer_kwargs}
|
|
277
|
+
kwargs = _filter_optimizer_kwargs(optimizer_cls, kwargs)
|
|
278
|
+
return optimizer_cls(
|
|
279
|
+
target=self.target,
|
|
280
|
+
evaluate_candidate=self.evaluate_candidate,
|
|
281
|
+
**kwargs,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
def optimize(
|
|
285
|
+
self,
|
|
286
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
287
|
+
**optimizer_kwargs: Any,
|
|
288
|
+
) -> OptimizationResult:
|
|
289
|
+
return _as_optimization_result(
|
|
290
|
+
self.build_optimizer(optimizer_cls, **optimizer_kwargs).optimize()
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
EvalSuiteOptimizationProblem = SimulateEvalSuiteOptimizationProblem
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
@dataclass
|
|
298
|
+
class SimulateSuiteOptimizationProblem:
|
|
299
|
+
"""
|
|
300
|
+
Bridge promptfoo-style Agent Learning suites into AgentOptimizer search.
|
|
301
|
+
|
|
302
|
+
This is the suite-level counterpart to ``SimulateManifestOptimizationProblem``
|
|
303
|
+
and ``SimulateEvalSuiteOptimizationProblem``: candidate configs are merged
|
|
304
|
+
into a full Agent Learning suite, then the whole mixed workflow can be
|
|
305
|
+
scored across simulation, eval, red-team, nested suites, and optimization
|
|
306
|
+
children. It is the optimizer primitive for trajectory-level trinity gates,
|
|
307
|
+
not isolated prompt/provider edits.
|
|
308
|
+
"""
|
|
309
|
+
|
|
310
|
+
base_suite: Mapping[str, Any]
|
|
311
|
+
target: OptimizationTarget
|
|
312
|
+
run_suite: Callable[[Mapping[str, Any], AgentCandidate], Any]
|
|
313
|
+
score_suite: Optional[ManifestScorer] = None
|
|
314
|
+
threshold: float = 1.0
|
|
315
|
+
optimizer_kwargs: Mapping[str, Any] = field(default_factory=dict)
|
|
316
|
+
optimizer_cls: Type[Any] = AgentOptimizer
|
|
317
|
+
metadata: Mapping[str, Any] = field(default_factory=dict)
|
|
318
|
+
|
|
319
|
+
@classmethod
|
|
320
|
+
def from_suite(
|
|
321
|
+
cls,
|
|
322
|
+
suite: Mapping[str, Any],
|
|
323
|
+
*,
|
|
324
|
+
run_suite: Callable[[Mapping[str, Any], AgentCandidate], Any],
|
|
325
|
+
score_suite: Optional[ManifestScorer] = None,
|
|
326
|
+
name: Optional[str] = None,
|
|
327
|
+
threshold: Optional[float] = None,
|
|
328
|
+
) -> "SimulateSuiteOptimizationProblem":
|
|
329
|
+
optimization = _require_mapping(
|
|
330
|
+
suite.get("optimization"),
|
|
331
|
+
"suite.optimization",
|
|
332
|
+
)
|
|
333
|
+
target_config = _target_config(optimization)
|
|
334
|
+
optimizer_kwargs = _optimizer_kwargs(
|
|
335
|
+
_optional_mapping(optimization.get("optimizer"))
|
|
336
|
+
)
|
|
337
|
+
optimizer_cls = _optimizer_cls(_optional_mapping(optimization.get("optimizer")))
|
|
338
|
+
base_suite = copy.deepcopy(dict(suite))
|
|
339
|
+
base_suite.pop("optimization", None)
|
|
340
|
+
suite_name = str(name or suite.get("name") or "agent-learning-suite")
|
|
341
|
+
target_metadata = copy.deepcopy(dict(target_config.get("metadata") or {}))
|
|
342
|
+
target_metadata.setdefault("source", "agent_learning_suite")
|
|
343
|
+
target_metadata.setdefault("suite_name", suite_name)
|
|
344
|
+
return cls(
|
|
345
|
+
base_suite=base_suite,
|
|
346
|
+
target=OptimizationTarget(
|
|
347
|
+
name=str(target_config.get("name") or suite_name),
|
|
348
|
+
layers=_layers(target_config.get("layers")),
|
|
349
|
+
base_config=copy.deepcopy(dict(target_config["base_config"])),
|
|
350
|
+
search_space=_search_space(target_config["search_space"]),
|
|
351
|
+
metadata=target_metadata,
|
|
352
|
+
),
|
|
353
|
+
run_suite=run_suite,
|
|
354
|
+
score_suite=score_suite or _score_agent_learning_suite,
|
|
355
|
+
threshold=float(
|
|
356
|
+
threshold
|
|
357
|
+
if threshold is not None
|
|
358
|
+
else optimization.get("threshold", 1.0)
|
|
359
|
+
),
|
|
360
|
+
optimizer_kwargs=optimizer_kwargs,
|
|
361
|
+
optimizer_cls=optimizer_cls,
|
|
362
|
+
metadata={
|
|
363
|
+
"source": "agent_learning_suite",
|
|
364
|
+
"suite_name": suite_name,
|
|
365
|
+
"optimizer_algorithm": _optimizer_algorithm_name(optimizer_cls),
|
|
366
|
+
},
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
def candidate_suite(self, candidate: AgentCandidate) -> dict[str, Any]:
|
|
370
|
+
merged = deep_merge(
|
|
371
|
+
copy.deepcopy(dict(self.base_suite)),
|
|
372
|
+
copy.deepcopy(candidate.config),
|
|
373
|
+
)
|
|
374
|
+
_apply_candidate_patch_replacements(merged, candidate)
|
|
375
|
+
return merged
|
|
376
|
+
|
|
377
|
+
def evaluate_candidate(self, candidate: AgentCandidate) -> CandidateEvaluation:
|
|
378
|
+
candidate_suite = self.candidate_suite(candidate)
|
|
379
|
+
result = _run_sync(self.run_suite(candidate_suite, candidate))
|
|
380
|
+
score_source = result
|
|
381
|
+
if self.score_suite is not None:
|
|
382
|
+
score_source = _run_sync(
|
|
383
|
+
self.score_suite(candidate_suite, result, candidate)
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
metadata = {
|
|
387
|
+
**dict(self.metadata),
|
|
388
|
+
"candidate_suite": copy.deepcopy(candidate_suite),
|
|
389
|
+
"candidate_patch": copy.deepcopy(candidate.patch),
|
|
390
|
+
"patch": copy.deepcopy(candidate.patch),
|
|
391
|
+
"report": copy.deepcopy(result),
|
|
392
|
+
"report_summary": copy.deepcopy(_mapping_summary(result)),
|
|
393
|
+
"search_paths": list(candidate.metadata.get("search_paths", [])),
|
|
394
|
+
}
|
|
395
|
+
return _candidate_evaluation_from_value(
|
|
396
|
+
score_source,
|
|
397
|
+
candidate,
|
|
398
|
+
report=result,
|
|
399
|
+
metadata=metadata,
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
def build_optimizer(
|
|
403
|
+
self,
|
|
404
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
405
|
+
**optimizer_kwargs: Any,
|
|
406
|
+
) -> Any:
|
|
407
|
+
optimizer_cls = optimizer_cls or self.optimizer_cls
|
|
408
|
+
kwargs = {**dict(self.optimizer_kwargs), **optimizer_kwargs}
|
|
409
|
+
kwargs = _filter_optimizer_kwargs(optimizer_cls, kwargs)
|
|
410
|
+
return optimizer_cls(
|
|
411
|
+
target=self.target,
|
|
412
|
+
evaluate_candidate=self.evaluate_candidate,
|
|
413
|
+
**kwargs,
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
def optimize(
|
|
417
|
+
self,
|
|
418
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
419
|
+
**optimizer_kwargs: Any,
|
|
420
|
+
) -> OptimizationResult:
|
|
421
|
+
return _as_optimization_result(
|
|
422
|
+
self.build_optimizer(optimizer_cls, **optimizer_kwargs).optimize()
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
SuiteOptimizationProblem = SimulateSuiteOptimizationProblem
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def problem_from_simulate_manifest(
|
|
430
|
+
manifest: Mapping[str, Any],
|
|
431
|
+
*,
|
|
432
|
+
manifest_path: str | Path = ".",
|
|
433
|
+
name: Optional[str] = None,
|
|
434
|
+
) -> SimulateManifestOptimizationProblem:
|
|
435
|
+
"""Build a manifest optimization problem using simulate-sdk's public runtime."""
|
|
436
|
+
|
|
437
|
+
build_problem = _simulate_sdk_attr("build_manifest_optimization_problem")
|
|
438
|
+
return build_problem(
|
|
439
|
+
manifest,
|
|
440
|
+
manifest_path=Path(manifest_path).expanduser().resolve(),
|
|
441
|
+
name=name,
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
def problem_from_simulate_manifest_file(
|
|
446
|
+
path: str | Path,
|
|
447
|
+
*,
|
|
448
|
+
name: Optional[str] = None,
|
|
449
|
+
) -> SimulateManifestOptimizationProblem:
|
|
450
|
+
"""Load an agent-simulate manifest file and build an optimization problem."""
|
|
451
|
+
|
|
452
|
+
load_manifest = _simulate_sdk_attr("load_manifest")
|
|
453
|
+
manifest_path = Path(path).expanduser().resolve()
|
|
454
|
+
return problem_from_simulate_manifest(
|
|
455
|
+
load_manifest(manifest_path),
|
|
456
|
+
manifest_path=manifest_path,
|
|
457
|
+
name=name,
|
|
458
|
+
)
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def optimize_simulate_manifest(
|
|
462
|
+
manifest: Mapping[str, Any],
|
|
463
|
+
*,
|
|
464
|
+
manifest_path: str | Path = ".",
|
|
465
|
+
name: Optional[str] = None,
|
|
466
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
467
|
+
**optimizer_kwargs: Any,
|
|
468
|
+
) -> OptimizationResult:
|
|
469
|
+
"""Optimize an in-memory agent-simulate manifest through simulate-sdk."""
|
|
470
|
+
|
|
471
|
+
return problem_from_simulate_manifest(
|
|
472
|
+
manifest,
|
|
473
|
+
manifest_path=manifest_path,
|
|
474
|
+
name=name,
|
|
475
|
+
).optimize(optimizer_cls=optimizer_cls, **optimizer_kwargs)
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def optimize_simulate_manifest_file(
|
|
479
|
+
path: str | Path,
|
|
480
|
+
*,
|
|
481
|
+
name: Optional[str] = None,
|
|
482
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
483
|
+
**optimizer_kwargs: Any,
|
|
484
|
+
) -> OptimizationResult:
|
|
485
|
+
"""Optimize an agent-simulate manifest file through simulate-sdk."""
|
|
486
|
+
|
|
487
|
+
return problem_from_simulate_manifest_file(path, name=name).optimize(
|
|
488
|
+
optimizer_cls=optimizer_cls,
|
|
489
|
+
**optimizer_kwargs,
|
|
490
|
+
)
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def problem_from_eval_suite(
|
|
494
|
+
suite: Mapping[str, Any],
|
|
495
|
+
*,
|
|
496
|
+
suite_path: str | Path = ".",
|
|
497
|
+
name: Optional[str] = None,
|
|
498
|
+
) -> SimulateEvalSuiteOptimizationProblem:
|
|
499
|
+
"""Build an eval-suite optimization problem using simulate-sdk's runtime."""
|
|
500
|
+
|
|
501
|
+
run_eval_suite = _simulate_sdk_attr("run_eval_suite")
|
|
502
|
+
suite_path = _suite_file_like_path(suite_path)
|
|
503
|
+
|
|
504
|
+
def run_suite(candidate_suite: Mapping[str, Any], candidate: AgentCandidate) -> Any:
|
|
505
|
+
return run_eval_suite(candidate_suite, suite_path=suite_path)
|
|
506
|
+
|
|
507
|
+
return SimulateEvalSuiteOptimizationProblem.from_suite(
|
|
508
|
+
suite,
|
|
509
|
+
run_suite=run_suite,
|
|
510
|
+
name=name,
|
|
511
|
+
)
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def problem_from_eval_suite_file(
|
|
515
|
+
path: str | Path,
|
|
516
|
+
*,
|
|
517
|
+
name: Optional[str] = None,
|
|
518
|
+
) -> SimulateEvalSuiteOptimizationProblem:
|
|
519
|
+
"""Load a simulate-sdk eval suite file and build an optimization problem."""
|
|
520
|
+
|
|
521
|
+
load_eval_suite_file = _simulate_sdk_attr("load_eval_suite_file")
|
|
522
|
+
suite_path = Path(path).expanduser().resolve()
|
|
523
|
+
return problem_from_eval_suite(
|
|
524
|
+
load_eval_suite_file(suite_path),
|
|
525
|
+
suite_path=suite_path,
|
|
526
|
+
name=name,
|
|
527
|
+
)
|
|
528
|
+
|
|
529
|
+
|
|
530
|
+
def optimize_eval_suite(
|
|
531
|
+
suite: Mapping[str, Any],
|
|
532
|
+
*,
|
|
533
|
+
suite_path: str | Path = ".",
|
|
534
|
+
name: Optional[str] = None,
|
|
535
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
536
|
+
**optimizer_kwargs: Any,
|
|
537
|
+
) -> OptimizationResult:
|
|
538
|
+
"""Optimize an in-memory simulate-sdk eval suite."""
|
|
539
|
+
|
|
540
|
+
return problem_from_eval_suite(
|
|
541
|
+
suite,
|
|
542
|
+
suite_path=suite_path,
|
|
543
|
+
name=name,
|
|
544
|
+
).optimize(optimizer_cls=optimizer_cls, **optimizer_kwargs)
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
def optimize_eval_suite_file(
|
|
548
|
+
path: str | Path,
|
|
549
|
+
*,
|
|
550
|
+
name: Optional[str] = None,
|
|
551
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
552
|
+
**optimizer_kwargs: Any,
|
|
553
|
+
) -> OptimizationResult:
|
|
554
|
+
"""Optimize a simulate-sdk eval suite file."""
|
|
555
|
+
|
|
556
|
+
return problem_from_eval_suite_file(path, name=name).optimize(
|
|
557
|
+
optimizer_cls=optimizer_cls,
|
|
558
|
+
**optimizer_kwargs,
|
|
559
|
+
)
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def problem_from_agent_learning_suite(
|
|
563
|
+
suite: Mapping[str, Any],
|
|
564
|
+
*,
|
|
565
|
+
suite_path: str | Path = ".",
|
|
566
|
+
name: Optional[str] = None,
|
|
567
|
+
) -> SimulateSuiteOptimizationProblem:
|
|
568
|
+
"""Build a full Agent Learning suite optimization problem."""
|
|
569
|
+
|
|
570
|
+
run_agent_learning_suite = _agent_learning_suite_attr("run_suite")
|
|
571
|
+
suite_path = _agent_learning_suite_file_like_path(suite_path)
|
|
572
|
+
|
|
573
|
+
def run_suite(candidate_suite: Mapping[str, Any], candidate: AgentCandidate) -> Any:
|
|
574
|
+
return run_agent_learning_suite(candidate_suite, suite_path=suite_path)
|
|
575
|
+
|
|
576
|
+
return SimulateSuiteOptimizationProblem.from_suite(
|
|
577
|
+
suite,
|
|
578
|
+
run_suite=run_suite,
|
|
579
|
+
score_suite=_score_agent_learning_suite,
|
|
580
|
+
name=name,
|
|
581
|
+
)
|
|
582
|
+
|
|
583
|
+
|
|
584
|
+
def problem_from_agent_learning_suite_file(
|
|
585
|
+
path: str | Path,
|
|
586
|
+
*,
|
|
587
|
+
name: Optional[str] = None,
|
|
588
|
+
) -> SimulateSuiteOptimizationProblem:
|
|
589
|
+
"""Load an Agent Learning suite file and build an optimization problem."""
|
|
590
|
+
|
|
591
|
+
load_suite_file = _agent_learning_suite_attr("load_suite_file")
|
|
592
|
+
suite_path = Path(path).expanduser().resolve()
|
|
593
|
+
return problem_from_agent_learning_suite(
|
|
594
|
+
load_suite_file(suite_path),
|
|
595
|
+
suite_path=suite_path,
|
|
596
|
+
name=name,
|
|
597
|
+
)
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
def optimize_agent_learning_suite(
|
|
601
|
+
suite: Mapping[str, Any],
|
|
602
|
+
*,
|
|
603
|
+
suite_path: str | Path = ".",
|
|
604
|
+
name: Optional[str] = None,
|
|
605
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
606
|
+
**optimizer_kwargs: Any,
|
|
607
|
+
) -> OptimizationResult:
|
|
608
|
+
"""Optimize an in-memory Agent Learning suite."""
|
|
609
|
+
|
|
610
|
+
return problem_from_agent_learning_suite(
|
|
611
|
+
suite,
|
|
612
|
+
suite_path=suite_path,
|
|
613
|
+
name=name,
|
|
614
|
+
).optimize(optimizer_cls=optimizer_cls, **optimizer_kwargs)
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
def optimize_agent_learning_suite_file(
|
|
618
|
+
path: str | Path,
|
|
619
|
+
*,
|
|
620
|
+
name: Optional[str] = None,
|
|
621
|
+
optimizer_cls: Optional[Type[Any]] = None,
|
|
622
|
+
**optimizer_kwargs: Any,
|
|
623
|
+
) -> OptimizationResult:
|
|
624
|
+
"""Optimize an Agent Learning suite file."""
|
|
625
|
+
|
|
626
|
+
return problem_from_agent_learning_suite_file(path, name=name).optimize(
|
|
627
|
+
optimizer_cls=optimizer_cls,
|
|
628
|
+
**optimizer_kwargs,
|
|
629
|
+
)
|
|
630
|
+
|
|
631
|
+
|
|
632
|
+
def deep_merge(base: Any, patch: Any) -> Any:
|
|
633
|
+
if isinstance(base, dict) and isinstance(patch, Mapping):
|
|
634
|
+
for key, value in patch.items():
|
|
635
|
+
base[key] = deep_merge(base.get(key), value)
|
|
636
|
+
return base
|
|
637
|
+
if isinstance(base, list) and isinstance(patch, list):
|
|
638
|
+
merged = list(base)
|
|
639
|
+
for index, value in enumerate(patch):
|
|
640
|
+
if index < len(merged):
|
|
641
|
+
merged[index] = deep_merge(merged[index], value)
|
|
642
|
+
else:
|
|
643
|
+
merged.append(copy.deepcopy(value))
|
|
644
|
+
return merged
|
|
645
|
+
return copy.deepcopy(patch)
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _apply_candidate_patch_replacements(
|
|
649
|
+
payload: dict[str, Any],
|
|
650
|
+
candidate: AgentCandidate,
|
|
651
|
+
) -> None:
|
|
652
|
+
"""Reapply exact search-path patches after deep-merge candidate assembly."""
|
|
653
|
+
|
|
654
|
+
for path, value in candidate.patch.items():
|
|
655
|
+
set_path(payload, str(path), copy.deepcopy(value))
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
def _candidate_evaluation_from_value(
|
|
659
|
+
value: Any,
|
|
660
|
+
candidate: AgentCandidate,
|
|
661
|
+
*,
|
|
662
|
+
report: Any,
|
|
663
|
+
metadata: Mapping[str, Any],
|
|
664
|
+
) -> CandidateEvaluation:
|
|
665
|
+
if isinstance(value, CandidateEvaluation):
|
|
666
|
+
return CandidateEvaluation(
|
|
667
|
+
candidate=candidate,
|
|
668
|
+
score=float(value.score),
|
|
669
|
+
reason=value.reason,
|
|
670
|
+
individual_results=list(value.individual_results or []),
|
|
671
|
+
report=value.report if value.report is not None else report,
|
|
672
|
+
metadata={**dict(metadata), **dict(value.metadata or {})},
|
|
673
|
+
)
|
|
674
|
+
if isinstance(value, EvaluationResult):
|
|
675
|
+
return CandidateEvaluation(
|
|
676
|
+
candidate=candidate,
|
|
677
|
+
score=float(value.score),
|
|
678
|
+
reason=value.reason,
|
|
679
|
+
individual_results=[value],
|
|
680
|
+
report=report,
|
|
681
|
+
metadata={**dict(metadata), **dict(value.metadata or {})},
|
|
682
|
+
)
|
|
683
|
+
|
|
684
|
+
score = _score_from_value(value)
|
|
685
|
+
reason = _reason_from_value(value)
|
|
686
|
+
individual_results = _individual_results_from_value(value)
|
|
687
|
+
report_value = _report_from_value(value, report)
|
|
688
|
+
extra_metadata = _metadata_from_value(value)
|
|
689
|
+
if score is None:
|
|
690
|
+
score = _score_from_value(report)
|
|
691
|
+
if score is None:
|
|
692
|
+
scores = list(_iter_report_scores(report))
|
|
693
|
+
if scores:
|
|
694
|
+
score = sum(scores) / len(scores)
|
|
695
|
+
if score is None:
|
|
696
|
+
raise ValueError(
|
|
697
|
+
"Manifest evaluation returned no score. Return a numeric score, "
|
|
698
|
+
"EvaluationResult, CandidateEvaluation, score-bearing mapping/object, "
|
|
699
|
+
"or provide score_manifest."
|
|
700
|
+
)
|
|
701
|
+
return CandidateEvaluation(
|
|
702
|
+
candidate=candidate,
|
|
703
|
+
score=score,
|
|
704
|
+
reason=reason,
|
|
705
|
+
individual_results=individual_results,
|
|
706
|
+
report=report_value,
|
|
707
|
+
metadata={**dict(metadata), **extra_metadata},
|
|
708
|
+
)
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
def _declared_anchor_objective(value: Any) -> Optional[Mapping[str, Any]]:
|
|
712
|
+
"""Return a REAL declared objective (with ``evals`` carrying >=1 ``anchor``
|
|
713
|
+
term) if the candidate value carries one — searched in the manifest/result
|
|
714
|
+
locations only. NEVER synthesized from config (that over-reaches and regresses
|
|
715
|
+
structural/hook manifests). Bug #2: only opted-in declared-anchor objectives
|
|
716
|
+
get objective-anchored scoring; everything else keeps the engine score."""
|
|
717
|
+
from fi.opt._objective_scoring import has_declared_anchor_objective
|
|
718
|
+
|
|
719
|
+
if not isinstance(value, Mapping):
|
|
720
|
+
return None
|
|
721
|
+
candidates = [
|
|
722
|
+
value.get("objective"),
|
|
723
|
+
(value.get("evaluation") or {}).get("objective") if isinstance(value.get("evaluation"), Mapping) else None,
|
|
724
|
+
((value.get("simulation") or {}).get("inline") or {}).get("objective")
|
|
725
|
+
if isinstance(value.get("simulation"), Mapping) else None,
|
|
726
|
+
(value.get("scenario") or {}).get("objective") if isinstance(value.get("scenario"), Mapping) else None,
|
|
727
|
+
]
|
|
728
|
+
for obj in candidates:
|
|
729
|
+
if has_declared_anchor_objective(obj):
|
|
730
|
+
return obj
|
|
731
|
+
return None
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def _candidate_metric_averages(value: Any) -> Optional[Mapping[str, Any]]:
|
|
735
|
+
if not isinstance(value, Mapping):
|
|
736
|
+
return None
|
|
737
|
+
if isinstance(value.get("metric_averages"), Mapping):
|
|
738
|
+
return value["metric_averages"]
|
|
739
|
+
summary = value.get("summary")
|
|
740
|
+
if isinstance(summary, Mapping) and isinstance(summary.get("metric_averages"), Mapping):
|
|
741
|
+
return summary["metric_averages"]
|
|
742
|
+
return None
|
|
743
|
+
|
|
744
|
+
|
|
745
|
+
def _objective_anchored_score(value: Any) -> Optional[float]:
|
|
746
|
+
"""Bug #2: score a candidate on its DECLARED anchor objective (real dynamic
|
|
747
|
+
range) instead of the all-metrics-mean ``evaluation_score``. Returns None
|
|
748
|
+
unless BOTH a declared-anchor objective and metric_averages are present, so
|
|
749
|
+
legacy/structural manifests fall through to the existing score unchanged."""
|
|
750
|
+
objective = _declared_anchor_objective(value)
|
|
751
|
+
if objective is None:
|
|
752
|
+
return None
|
|
753
|
+
metrics = _candidate_metric_averages(value)
|
|
754
|
+
if not metrics:
|
|
755
|
+
return None
|
|
756
|
+
from fi.opt._objective_scoring import objective_score
|
|
757
|
+
|
|
758
|
+
return _coerce_score(objective_score(metrics, objective).get("score"))
|
|
759
|
+
|
|
760
|
+
|
|
761
|
+
def _score_from_value(value: Any) -> Optional[float]:
|
|
762
|
+
anchored = _objective_anchored_score(value)
|
|
763
|
+
if anchored is not None:
|
|
764
|
+
return anchored
|
|
765
|
+
direct = _coerce_score(value)
|
|
766
|
+
if direct is not None:
|
|
767
|
+
return direct
|
|
768
|
+
if isinstance(value, Mapping):
|
|
769
|
+
for key in ("score", "final_score", "average_score", "optimization_score"):
|
|
770
|
+
score = _coerce_score(value.get(key))
|
|
771
|
+
if score is not None:
|
|
772
|
+
return score
|
|
773
|
+
summary = value.get("summary")
|
|
774
|
+
if isinstance(summary, Mapping):
|
|
775
|
+
for key in ("score", "final_score", "optimization_score"):
|
|
776
|
+
score = _coerce_score(summary.get(key))
|
|
777
|
+
if score is not None:
|
|
778
|
+
return score
|
|
779
|
+
for key in ("score", "final_score", "average_score", "optimization_score"):
|
|
780
|
+
score = _coerce_score(getattr(value, key, None))
|
|
781
|
+
if score is not None:
|
|
782
|
+
return score
|
|
783
|
+
return None
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
def _reason_from_value(value: Any) -> str:
|
|
787
|
+
if isinstance(value, Mapping):
|
|
788
|
+
return str(value.get("reason") or value.get("status") or "")
|
|
789
|
+
return str(getattr(value, "reason", "") or "")
|
|
790
|
+
|
|
791
|
+
|
|
792
|
+
def _individual_results_from_value(value: Any) -> list[Any]:
|
|
793
|
+
if isinstance(value, Mapping):
|
|
794
|
+
results = value.get("individual_results")
|
|
795
|
+
return list(results or [])
|
|
796
|
+
return list(getattr(value, "individual_results", []) or [])
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
def _report_from_value(value: Any, fallback: Any) -> Any:
|
|
800
|
+
if isinstance(value, Mapping) and "report" in value:
|
|
801
|
+
return value["report"]
|
|
802
|
+
report = getattr(value, "report", None)
|
|
803
|
+
return fallback if report is None else report
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _metadata_from_value(value: Any) -> dict[str, Any]:
|
|
807
|
+
if isinstance(value, Mapping):
|
|
808
|
+
return copy.deepcopy(dict(value.get("metadata") or {}))
|
|
809
|
+
metadata = getattr(value, "metadata", None)
|
|
810
|
+
if isinstance(metadata, Mapping):
|
|
811
|
+
return copy.deepcopy(dict(metadata))
|
|
812
|
+
return {}
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def _mapping_summary(value: Any) -> dict[str, Any]:
|
|
816
|
+
if isinstance(value, Mapping):
|
|
817
|
+
summary = value.get("summary")
|
|
818
|
+
if isinstance(summary, Mapping):
|
|
819
|
+
return copy.deepcopy(dict(summary))
|
|
820
|
+
return {}
|
|
821
|
+
|
|
822
|
+
|
|
823
|
+
def _score_agent_learning_suite(
|
|
824
|
+
candidate_suite: Mapping[str, Any],
|
|
825
|
+
result: Any,
|
|
826
|
+
candidate: AgentCandidate,
|
|
827
|
+
) -> dict[str, Any]:
|
|
828
|
+
summary = _mapping_summary(result)
|
|
829
|
+
raw_score = _score_from_value(result)
|
|
830
|
+
score = float(raw_score if raw_score is not None else 0.0)
|
|
831
|
+
action_run_score: Optional[float] = None
|
|
832
|
+
if isinstance(result, Mapping):
|
|
833
|
+
status = str(result.get("status") or "")
|
|
834
|
+
exit_code = int(result.get("exit_code", 1) or 0)
|
|
835
|
+
capability_gate = bool(
|
|
836
|
+
summary.get("capability_gate_passed")
|
|
837
|
+
if "capability_gate_passed" in summary
|
|
838
|
+
else True
|
|
839
|
+
)
|
|
840
|
+
executed = float(summary.get("executed_count") or 0.0)
|
|
841
|
+
job_count = float(summary.get("job_count") or executed or 1.0)
|
|
842
|
+
execution_score = executed / job_count if job_count else score
|
|
843
|
+
if status != "passed" or exit_code != 0:
|
|
844
|
+
score = min(score, execution_score)
|
|
845
|
+
if not capability_gate:
|
|
846
|
+
score = min(score, 0.5)
|
|
847
|
+
action_run_score = _action_run_suite_score(result)
|
|
848
|
+
if action_run_score is not None:
|
|
849
|
+
score = min(score, action_run_score)
|
|
850
|
+
return {
|
|
851
|
+
"score": round(score, 4),
|
|
852
|
+
"reason": str(result.get("status") if isinstance(result, Mapping) else ""),
|
|
853
|
+
"metadata": {
|
|
854
|
+
"suite_summary": summary,
|
|
855
|
+
"action_run_score": action_run_score,
|
|
856
|
+
"candidate_suite_name": candidate_suite.get("name"),
|
|
857
|
+
"candidate_id": candidate.id,
|
|
858
|
+
},
|
|
859
|
+
}
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
def _action_run_suite_score(result: Mapping[str, Any]) -> Optional[float]:
|
|
863
|
+
children = [
|
|
864
|
+
child
|
|
865
|
+
for child in result.get("children") or result.get("jobs") or []
|
|
866
|
+
if isinstance(child, Mapping)
|
|
867
|
+
]
|
|
868
|
+
action_children = [
|
|
869
|
+
child
|
|
870
|
+
for child in children
|
|
871
|
+
if str(child.get("command") or "").replace("-", "_") == "action_run"
|
|
872
|
+
]
|
|
873
|
+
if not action_children:
|
|
874
|
+
return None
|
|
875
|
+
scores: list[float] = []
|
|
876
|
+
for child in action_children:
|
|
877
|
+
exit_code = child.get("exit_code", 1)
|
|
878
|
+
if int(exit_code if exit_code is not None else 1) != 0:
|
|
879
|
+
scores.append(0.0)
|
|
880
|
+
continue
|
|
881
|
+
child_summary = _mapping_summary(child.get("result"))
|
|
882
|
+
output_count = float(child_summary.get("output_count") or 0.0)
|
|
883
|
+
written_count = float(child_summary.get("outputs_written_count") or 0.0)
|
|
884
|
+
completion = (
|
|
885
|
+
float(child_summary.get("output_completion_rate"))
|
|
886
|
+
if child_summary.get("output_completion_rate") is not None
|
|
887
|
+
else (written_count / output_count if output_count else 1.0)
|
|
888
|
+
)
|
|
889
|
+
evidence_depth = min(written_count / 4.0, 1.0)
|
|
890
|
+
scores.append((0.8 * completion) + (0.2 * evidence_depth))
|
|
891
|
+
return round(sum(scores) / len(scores), 4) if scores else None
|
|
892
|
+
|
|
893
|
+
|
|
894
|
+
def _target_config(optimization: Mapping[str, Any]) -> Mapping[str, Any]:
|
|
895
|
+
target = _require_mapping(optimization.get("target"), "optimization.target")
|
|
896
|
+
_require_mapping(target.get("base_config"), "optimization.target.base_config")
|
|
897
|
+
search_space = _require_mapping(
|
|
898
|
+
target.get("search_space"),
|
|
899
|
+
"optimization.target.search_space",
|
|
900
|
+
)
|
|
901
|
+
if not search_space:
|
|
902
|
+
raise ValueError("optimization.target.search_space must not be empty.")
|
|
903
|
+
return target
|
|
904
|
+
|
|
905
|
+
|
|
906
|
+
def _optimizer_kwargs(config: Optional[Mapping[str, Any]]) -> dict[str, Any]:
|
|
907
|
+
if not config:
|
|
908
|
+
return {}
|
|
909
|
+
allowed = {
|
|
910
|
+
"max_candidates",
|
|
911
|
+
"max_rounds",
|
|
912
|
+
"beam_width",
|
|
913
|
+
"max_proposals_per_round",
|
|
914
|
+
"include_seed",
|
|
915
|
+
"auto_diagnose",
|
|
916
|
+
"diagnoses",
|
|
917
|
+
"diagnostic_score_threshold",
|
|
918
|
+
"total_budget",
|
|
919
|
+
"min_pulls_per_candidate",
|
|
920
|
+
"exploration",
|
|
921
|
+
"target_score",
|
|
922
|
+
"selection",
|
|
923
|
+
"population_size",
|
|
924
|
+
"generations",
|
|
925
|
+
"elite_count",
|
|
926
|
+
"mutation_rate",
|
|
927
|
+
"crossover_rate",
|
|
928
|
+
"max_mutations_per_candidate",
|
|
929
|
+
"tournament_size",
|
|
930
|
+
"seed",
|
|
931
|
+
"layer_path_bias",
|
|
932
|
+
"mutation_library",
|
|
933
|
+
"max_library_candidates",
|
|
934
|
+
# Phase 4 (extend-only): declared budgets, Elo selection knobs,
|
|
935
|
+
# two-chamber budgets, society ledger, strategy declaration, TPE trials.
|
|
936
|
+
"eval_budget",
|
|
937
|
+
"elo_k_factor",
|
|
938
|
+
"elo_initial_rating",
|
|
939
|
+
"samiti_budget",
|
|
940
|
+
"sabha_budget",
|
|
941
|
+
"society_ledger",
|
|
942
|
+
"search_strategy",
|
|
943
|
+
"n_trials",
|
|
944
|
+
# Phase 4 (extend-only): regression-replay backend inputs — a local
|
|
945
|
+
# AgentRegressionDataset (mapping coerced below) plus the delegated
|
|
946
|
+
# repair backend selector consumed by FutureAGIRegressionReplayOptimizer.
|
|
947
|
+
"dataset",
|
|
948
|
+
"optimizer",
|
|
949
|
+
}
|
|
950
|
+
kwargs = {key: copy.deepcopy(config[key]) for key in allowed if key in config}
|
|
951
|
+
dataset = kwargs.get("dataset")
|
|
952
|
+
if isinstance(dataset, Mapping) and "cases" in dataset:
|
|
953
|
+
from ..observability import AgentRegressionDataset
|
|
954
|
+
|
|
955
|
+
kwargs["dataset"] = AgentRegressionDataset.model_validate(dict(dataset))
|
|
956
|
+
return kwargs
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
def _optimizer_cls(config: Optional[Mapping[str, Any]]) -> Type[Any]:
|
|
960
|
+
if not config:
|
|
961
|
+
return AgentOptimizer
|
|
962
|
+
raw = (
|
|
963
|
+
config.get("algorithm")
|
|
964
|
+
or config.get("type")
|
|
965
|
+
or config.get("name")
|
|
966
|
+
or config.get("strategy")
|
|
967
|
+
or "agent"
|
|
968
|
+
)
|
|
969
|
+
normalized = str(raw or "agent").strip().lower().replace("-", "_").replace(" ", "_")
|
|
970
|
+
if normalized in {
|
|
971
|
+
"agent",
|
|
972
|
+
"agent_optimizer",
|
|
973
|
+
"deterministic",
|
|
974
|
+
"candidate_search",
|
|
975
|
+
"deterministic_candidate_search",
|
|
976
|
+
"grid",
|
|
977
|
+
}:
|
|
978
|
+
return AgentOptimizer
|
|
979
|
+
if normalized in {
|
|
980
|
+
"evolution",
|
|
981
|
+
"agent_evolution",
|
|
982
|
+
"agent_evolution_optimizer",
|
|
983
|
+
"domain_aware_evolution",
|
|
984
|
+
"mutation",
|
|
985
|
+
"mutation_library",
|
|
986
|
+
}:
|
|
987
|
+
return AgentEvolutionOptimizer
|
|
988
|
+
if normalized in {
|
|
989
|
+
"social_memory",
|
|
990
|
+
"society",
|
|
991
|
+
"agent_social_memory",
|
|
992
|
+
"agent_social_memory_optimizer",
|
|
993
|
+
"futureagi_social_memory",
|
|
994
|
+
"futureagi_social_memory_optimizer",
|
|
995
|
+
"multi_interaction",
|
|
996
|
+
"multi_interaction_social_memory",
|
|
997
|
+
}:
|
|
998
|
+
return AgentSocialMemoryOptimizer
|
|
999
|
+
# Phase 4 (extend-only): additional target-contract backends for the
|
|
1000
|
+
# optimizer profile matrix; legacy tokens above are untouched.
|
|
1001
|
+
if normalized in {"council", "council_agent", "council_agent_optimizer"}:
|
|
1002
|
+
from ..optimizers.council import CouncilAgentOptimizer
|
|
1003
|
+
|
|
1004
|
+
return CouncilAgentOptimizer
|
|
1005
|
+
if normalized in {
|
|
1006
|
+
"society_agent",
|
|
1007
|
+
"society_agent_optimizer",
|
|
1008
|
+
"role_graph_society",
|
|
1009
|
+
"society_role_graph",
|
|
1010
|
+
}:
|
|
1011
|
+
from ..optimizers.council import SocietyAgentOptimizer
|
|
1012
|
+
|
|
1013
|
+
return SocietyAgentOptimizer
|
|
1014
|
+
if normalized in {"tpe", "agent_tpe", "agent_tpe_optimizer"}:
|
|
1015
|
+
from ..optimizers.agent_tpe import AgentTPEOptimizer
|
|
1016
|
+
|
|
1017
|
+
return AgentTPEOptimizer
|
|
1018
|
+
if normalized in {"bandit", "agent_bandit", "agent_bandit_optimizer", "ucb"}:
|
|
1019
|
+
from ..optimizers.agent_bandit import AgentBanditOptimizer
|
|
1020
|
+
|
|
1021
|
+
return AgentBanditOptimizer
|
|
1022
|
+
if normalized in {
|
|
1023
|
+
"regression_replay",
|
|
1024
|
+
"futureagi_regression_replay",
|
|
1025
|
+
"futureagi_replay",
|
|
1026
|
+
"regression_replay_optimizer",
|
|
1027
|
+
}:
|
|
1028
|
+
from ..optimizers.futureagi_replay import FutureAGIRegressionReplayOptimizer
|
|
1029
|
+
|
|
1030
|
+
return FutureAGIRegressionReplayOptimizer
|
|
1031
|
+
if normalized in {
|
|
1032
|
+
"curriculum",
|
|
1033
|
+
"agent_curriculum",
|
|
1034
|
+
"agent_curriculum_optimizer",
|
|
1035
|
+
"staged",
|
|
1036
|
+
}:
|
|
1037
|
+
from ..optimizers.agent_curriculum import AgentCurriculumOptimizer
|
|
1038
|
+
|
|
1039
|
+
return AgentCurriculumOptimizer
|
|
1040
|
+
if normalized in {
|
|
1041
|
+
"pareto",
|
|
1042
|
+
"agent_pareto",
|
|
1043
|
+
"agent_pareto_optimizer",
|
|
1044
|
+
"multi_objective",
|
|
1045
|
+
"multi_objective_pareto",
|
|
1046
|
+
}:
|
|
1047
|
+
from ..optimizers.agent_pareto import AgentParetoOptimizer
|
|
1048
|
+
|
|
1049
|
+
return AgentParetoOptimizer
|
|
1050
|
+
if normalized in {
|
|
1051
|
+
"feedback",
|
|
1052
|
+
"agent_feedback",
|
|
1053
|
+
"agent_feedback_optimizer",
|
|
1054
|
+
"diagnostic_feedback",
|
|
1055
|
+
}:
|
|
1056
|
+
from ..optimizers.agent_feedback import AgentFeedbackOptimizer
|
|
1057
|
+
|
|
1058
|
+
return AgentFeedbackOptimizer
|
|
1059
|
+
raise ValueError(
|
|
1060
|
+
"optimization.optimizer.algorithm must be one of: agent, evolution, "
|
|
1061
|
+
"social_memory, council, society_role_graph, tpe, bandit, "
|
|
1062
|
+
"regression_replay, curriculum, pareto, feedback (deterministic agent "
|
|
1063
|
+
"backends), or a generative token routed through the generative "
|
|
1064
|
+
"eval-suite bridge: gepa, protegi, metaprompt, promptwizard, "
|
|
1065
|
+
"random_search, bayesian_search"
|
|
1066
|
+
)
|
|
1067
|
+
|
|
1068
|
+
|
|
1069
|
+
def _optimizer_algorithm_name(optimizer_cls: Type[Any]) -> str:
|
|
1070
|
+
if optimizer_cls is AgentEvolutionOptimizer:
|
|
1071
|
+
return "evolution"
|
|
1072
|
+
if optimizer_cls is AgentSocialMemoryOptimizer:
|
|
1073
|
+
return "social_memory"
|
|
1074
|
+
name = getattr(optimizer_cls, "__name__", "")
|
|
1075
|
+
if name == "CouncilAgentOptimizer":
|
|
1076
|
+
return "council"
|
|
1077
|
+
if name == "SocietyAgentOptimizer":
|
|
1078
|
+
return "society_role_graph"
|
|
1079
|
+
if name == "AgentTPEOptimizer":
|
|
1080
|
+
return "tpe"
|
|
1081
|
+
if name == "AgentBanditOptimizer":
|
|
1082
|
+
return "bandit"
|
|
1083
|
+
if name == "FutureAGIRegressionReplayOptimizer":
|
|
1084
|
+
return "regression_replay"
|
|
1085
|
+
if name == "AgentCurriculumOptimizer":
|
|
1086
|
+
return "curriculum"
|
|
1087
|
+
if name == "AgentParetoOptimizer":
|
|
1088
|
+
return "pareto"
|
|
1089
|
+
if name == "AgentFeedbackOptimizer":
|
|
1090
|
+
return "feedback"
|
|
1091
|
+
return "agent"
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
def _as_optimization_result(result: Any) -> OptimizationResult:
|
|
1095
|
+
"""Coerce backend audit records onto the OptimizationResult contract.
|
|
1096
|
+
|
|
1097
|
+
Phase 4 (extend-only): ``FutureAGIRegressionReplayOptimizer`` returns an
|
|
1098
|
+
``AgentFeedbackOptimizationResult`` audit record wrapping the inner
|
|
1099
|
+
``reoptimization_result``; the manifest pipeline consumes the inner
|
|
1100
|
+
result with the replay audit carried in its metadata. Every existing
|
|
1101
|
+
backend already returns ``OptimizationResult`` and passes through.
|
|
1102
|
+
"""
|
|
1103
|
+
|
|
1104
|
+
inner = getattr(result, "reoptimization_result", None)
|
|
1105
|
+
if inner is None:
|
|
1106
|
+
return result
|
|
1107
|
+
metadata = dict(getattr(inner, "metadata", {}) or {})
|
|
1108
|
+
metadata.setdefault(
|
|
1109
|
+
"regression_replay",
|
|
1110
|
+
{
|
|
1111
|
+
"optimizer": getattr(result, "optimizer", None),
|
|
1112
|
+
"feedback_source": getattr(result, "feedback_source", None),
|
|
1113
|
+
"baseline_score": getattr(result, "baseline_score", None),
|
|
1114
|
+
"final_score": getattr(result, "final_score", None),
|
|
1115
|
+
"improved": getattr(result, "improved", None),
|
|
1116
|
+
"feedback_case_count": len(getattr(result, "feedback_cases", []) or []),
|
|
1117
|
+
},
|
|
1118
|
+
)
|
|
1119
|
+
inner.metadata = metadata
|
|
1120
|
+
return inner
|
|
1121
|
+
|
|
1122
|
+
|
|
1123
|
+
def _evidence_scorer_config(
|
|
1124
|
+
optimization: Mapping[str, Any],
|
|
1125
|
+
target_config: Mapping[str, Any],
|
|
1126
|
+
*,
|
|
1127
|
+
base_manifest: Mapping[str, Any],
|
|
1128
|
+
) -> Optional[dict[str, Any]]:
|
|
1129
|
+
raw = (
|
|
1130
|
+
optimization.get("simulation_evidence")
|
|
1131
|
+
or optimization.get("evidence_scorer")
|
|
1132
|
+
or optimization.get("scoring")
|
|
1133
|
+
)
|
|
1134
|
+
if raw is False:
|
|
1135
|
+
return None
|
|
1136
|
+
if isinstance(raw, str):
|
|
1137
|
+
normalized = raw.strip().lower().replace("-", "_").replace(" ", "_")
|
|
1138
|
+
if normalized in {"simulation_evidence", "evidence", "environment_evidence"}:
|
|
1139
|
+
return {"enabled": True, "method": "simulation_evidence"}
|
|
1140
|
+
return None
|
|
1141
|
+
if isinstance(raw, Mapping):
|
|
1142
|
+
method = str(
|
|
1143
|
+
raw.get("method")
|
|
1144
|
+
or raw.get("type")
|
|
1145
|
+
or raw.get("name")
|
|
1146
|
+
or raw.get("strategy")
|
|
1147
|
+
or "simulation_evidence"
|
|
1148
|
+
).strip().lower().replace("-", "_").replace(" ", "_")
|
|
1149
|
+
enabled = bool(raw.get("enabled", True))
|
|
1150
|
+
if enabled and method in {
|
|
1151
|
+
"simulation_evidence",
|
|
1152
|
+
"evidence",
|
|
1153
|
+
"environment_evidence",
|
|
1154
|
+
"trace_evidence",
|
|
1155
|
+
}:
|
|
1156
|
+
config = copy.deepcopy(dict(raw))
|
|
1157
|
+
config["method"] = "simulation_evidence"
|
|
1158
|
+
return config
|
|
1159
|
+
return None
|
|
1160
|
+
|
|
1161
|
+
if raw is True:
|
|
1162
|
+
return {"enabled": True, "method": "simulation_evidence"}
|
|
1163
|
+
|
|
1164
|
+
layers = {str(layer).lower() for layer in target_config.get("layers", [])}
|
|
1165
|
+
should_auto_score = bool(layers & {"framework", "world", "orchestration"}) and not (
|
|
1166
|
+
_optional_mapping(
|
|
1167
|
+
_optional_mapping(base_manifest.get("evaluation")) or {}
|
|
1168
|
+
)
|
|
1169
|
+
and _optional_mapping(
|
|
1170
|
+
(_optional_mapping(base_manifest.get("evaluation")) or {}).get(
|
|
1171
|
+
"agent_report"
|
|
1172
|
+
)
|
|
1173
|
+
)
|
|
1174
|
+
)
|
|
1175
|
+
if should_auto_score:
|
|
1176
|
+
return {"enabled": True, "method": "simulation_evidence", "_auto": True}
|
|
1177
|
+
return None
|
|
1178
|
+
|
|
1179
|
+
|
|
1180
|
+
def _report_has_score(report: Any) -> bool:
|
|
1181
|
+
if _score_from_value(report) is not None:
|
|
1182
|
+
return True
|
|
1183
|
+
return bool(list(_iter_report_scores(report)))
|
|
1184
|
+
|
|
1185
|
+
|
|
1186
|
+
def _filter_optimizer_kwargs(
|
|
1187
|
+
optimizer_cls: Type[Any],
|
|
1188
|
+
kwargs: Mapping[str, Any],
|
|
1189
|
+
) -> dict[str, Any]:
|
|
1190
|
+
try:
|
|
1191
|
+
signature = inspect.signature(optimizer_cls)
|
|
1192
|
+
except (TypeError, ValueError):
|
|
1193
|
+
return dict(kwargs)
|
|
1194
|
+
parameters = signature.parameters
|
|
1195
|
+
if any(
|
|
1196
|
+
parameter.kind is inspect.Parameter.VAR_KEYWORD
|
|
1197
|
+
for parameter in parameters.values()
|
|
1198
|
+
):
|
|
1199
|
+
return dict(kwargs)
|
|
1200
|
+
allowed = set(parameters)
|
|
1201
|
+
return {key: value for key, value in kwargs.items() if key in allowed}
|
|
1202
|
+
|
|
1203
|
+
|
|
1204
|
+
def _layers(value: Any) -> list[OptimizationLayer]:
|
|
1205
|
+
return list(value or ["harness", "evaluator"])
|
|
1206
|
+
|
|
1207
|
+
|
|
1208
|
+
def _search_space(value: Mapping[str, Any]) -> dict[str, list[Any]]:
|
|
1209
|
+
search_space: dict[str, list[Any]] = {}
|
|
1210
|
+
for path, choices in value.items():
|
|
1211
|
+
if isinstance(choices, (str, bytes)) or not isinstance(choices, Sequence):
|
|
1212
|
+
raise ValueError(
|
|
1213
|
+
f"optimization.target.search_space.{path} must be a sequence."
|
|
1214
|
+
)
|
|
1215
|
+
if not choices:
|
|
1216
|
+
raise ValueError(
|
|
1217
|
+
f"optimization.target.search_space.{path} must not be empty."
|
|
1218
|
+
)
|
|
1219
|
+
search_space[str(path)] = copy.deepcopy(list(choices))
|
|
1220
|
+
return search_space
|
|
1221
|
+
|
|
1222
|
+
|
|
1223
|
+
def _require_mapping(value: Any, name: str) -> Mapping[str, Any]:
|
|
1224
|
+
if not isinstance(value, Mapping):
|
|
1225
|
+
raise ValueError(f"{name} must be an object.")
|
|
1226
|
+
return value
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def _optional_mapping(value: Any) -> Optional[Mapping[str, Any]]:
|
|
1230
|
+
if value is None:
|
|
1231
|
+
return None
|
|
1232
|
+
return _require_mapping(value, "optimization.optimizer")
|
|
1233
|
+
|
|
1234
|
+
|
|
1235
|
+
def _simulate_sdk_attr(name: str) -> Any:
|
|
1236
|
+
try:
|
|
1237
|
+
from fi import simulate as simulate_sdk
|
|
1238
|
+
except Exception as exc: # pragma: no cover - optional dependency clarity
|
|
1239
|
+
raise RuntimeError(
|
|
1240
|
+
"agent-simulate is required for simulate-sdk manifest helpers. "
|
|
1241
|
+
"Install simulate-sdk or call ManifestOptimizationProblem.from_manifest "
|
|
1242
|
+
"with explicit evaluate_manifest/score_manifest callbacks."
|
|
1243
|
+
) from exc
|
|
1244
|
+
try:
|
|
1245
|
+
return getattr(simulate_sdk, name)
|
|
1246
|
+
except AttributeError as exc: # pragma: no cover - version clarity
|
|
1247
|
+
raise RuntimeError(
|
|
1248
|
+
f"agent-simulate with `{name}` is required; upgrade simulate-sdk."
|
|
1249
|
+
) from exc
|
|
1250
|
+
|
|
1251
|
+
|
|
1252
|
+
def _suite_file_like_path(path: str | Path) -> Path:
|
|
1253
|
+
resolved = Path(path).expanduser().resolve()
|
|
1254
|
+
if resolved.is_dir():
|
|
1255
|
+
return resolved / "eval_suite.json"
|
|
1256
|
+
return resolved
|
|
1257
|
+
|
|
1258
|
+
|
|
1259
|
+
def _agent_learning_suite_file_like_path(path: str | Path) -> Path:
|
|
1260
|
+
resolved = Path(path).expanduser().resolve()
|
|
1261
|
+
if resolved.is_dir():
|
|
1262
|
+
return resolved / "agent_learning_suite.json"
|
|
1263
|
+
return resolved
|
|
1264
|
+
|
|
1265
|
+
|
|
1266
|
+
def _public_eval_suite_runner() -> Callable[[Mapping[str, Any], AgentCandidate], Any]:
|
|
1267
|
+
run_eval_suite = _simulate_sdk_attr("run_eval_suite")
|
|
1268
|
+
suite_path = Path.cwd() / "eval_suite.json"
|
|
1269
|
+
|
|
1270
|
+
def run_suite(candidate_suite: Mapping[str, Any], candidate: AgentCandidate) -> Any:
|
|
1271
|
+
return run_eval_suite(candidate_suite, suite_path=suite_path)
|
|
1272
|
+
|
|
1273
|
+
return run_suite
|
|
1274
|
+
|
|
1275
|
+
|
|
1276
|
+
def _agent_learning_suite_attr(name: str) -> Any:
|
|
1277
|
+
try:
|
|
1278
|
+
from fi.alk import suite as agent_learning_suite
|
|
1279
|
+
except Exception as exc: # pragma: no cover - optional dependency clarity
|
|
1280
|
+
raise RuntimeError(
|
|
1281
|
+
"agent-learning-kit is required for Agent Learning suite optimization."
|
|
1282
|
+
) from exc
|
|
1283
|
+
try:
|
|
1284
|
+
return getattr(agent_learning_suite, name)
|
|
1285
|
+
except AttributeError as exc: # pragma: no cover - version clarity
|
|
1286
|
+
raise RuntimeError(
|
|
1287
|
+
f"agent-learning-kit with `fi.alk.suite.{name}` is required."
|
|
1288
|
+
) from exc
|
|
1289
|
+
|
|
1290
|
+
|
|
1291
|
+
__all__ = [
|
|
1292
|
+
"EvalSuiteOptimizationProblem",
|
|
1293
|
+
"ManifestOptimizationProblem",
|
|
1294
|
+
"ManifestRunner",
|
|
1295
|
+
"ManifestScorer",
|
|
1296
|
+
"SimulateEvalSuiteOptimizationProblem",
|
|
1297
|
+
"SimulateManifestOptimizationProblem",
|
|
1298
|
+
"SimulateSuiteOptimizationProblem",
|
|
1299
|
+
"SuiteOptimizationProblem",
|
|
1300
|
+
"deep_merge",
|
|
1301
|
+
"optimize_agent_learning_suite",
|
|
1302
|
+
"optimize_agent_learning_suite_file",
|
|
1303
|
+
"optimize_eval_suite",
|
|
1304
|
+
"optimize_eval_suite_file",
|
|
1305
|
+
"optimize_simulate_manifest",
|
|
1306
|
+
"optimize_simulate_manifest_file",
|
|
1307
|
+
"problem_from_agent_learning_suite",
|
|
1308
|
+
"problem_from_agent_learning_suite_file",
|
|
1309
|
+
"problem_from_eval_suite",
|
|
1310
|
+
"problem_from_eval_suite_file",
|
|
1311
|
+
"problem_from_simulate_manifest",
|
|
1312
|
+
"problem_from_simulate_manifest_file",
|
|
1313
|
+
]
|