agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/opt/simulation.py
ADDED
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import inspect
|
|
5
|
+
from typing import Any, Callable, Dict, Iterable, Mapping, Optional
|
|
6
|
+
|
|
7
|
+
from .evidence import score_simulation_evidence
|
|
8
|
+
from .targets import AgentCandidate, CandidateEvaluation
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class SimulationEvaluator:
|
|
12
|
+
"""
|
|
13
|
+
Evaluates an AgentCandidate through simulate-sdk and optionally ai-evaluation.
|
|
14
|
+
|
|
15
|
+
The bridge is dependency-light by design:
|
|
16
|
+
- pass fake runner/evaluate functions in tests,
|
|
17
|
+
- or install `agent-simulate` and `ai-evaluation` for real runs.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
def __init__(
|
|
21
|
+
self,
|
|
22
|
+
*,
|
|
23
|
+
agent_factory: Optional[Callable[[AgentCandidate], Any]] = None,
|
|
24
|
+
scenario: Any = None,
|
|
25
|
+
topic: Optional[str] = None,
|
|
26
|
+
runner: Any = None,
|
|
27
|
+
runner_cls: Any = None,
|
|
28
|
+
runner_kwargs: Optional[
|
|
29
|
+
Mapping[str, Any] | Callable[[AgentCandidate], Mapping[str, Any]]
|
|
30
|
+
] = None,
|
|
31
|
+
eval_specs: Optional[Iterable[Dict[str, Any]]] = None,
|
|
32
|
+
eval_templates: Optional[Iterable[str]] = None,
|
|
33
|
+
evaluate_report_fn: Optional[Callable[..., Any]] = None,
|
|
34
|
+
evaluate_report_kwargs: Optional[Dict[str, Any]] = None,
|
|
35
|
+
agent_report_config: Optional[
|
|
36
|
+
Mapping[str, Any] | Callable[[AgentCandidate], Mapping[str, Any]]
|
|
37
|
+
] = None,
|
|
38
|
+
agent_report_threshold: float = 0.7,
|
|
39
|
+
use_agent_report_evaluator: bool = False,
|
|
40
|
+
report_scorer: Optional[Callable[[Any, AgentCandidate], float]] = None,
|
|
41
|
+
evidence_scorer_config: Optional[Mapping[str, Any]] = None,
|
|
42
|
+
) -> None:
|
|
43
|
+
self.agent_factory = agent_factory
|
|
44
|
+
self.scenario = scenario
|
|
45
|
+
self.topic = topic
|
|
46
|
+
self.runner = runner
|
|
47
|
+
self.runner_cls = runner_cls
|
|
48
|
+
self.runner_kwargs = runner_kwargs or {}
|
|
49
|
+
self.eval_specs = list(eval_specs) if eval_specs is not None else None
|
|
50
|
+
self.eval_templates = list(eval_templates) if eval_templates is not None else None
|
|
51
|
+
self.evaluate_report_fn = evaluate_report_fn
|
|
52
|
+
self.evaluate_report_kwargs = evaluate_report_kwargs or {}
|
|
53
|
+
self.agent_report_config = agent_report_config
|
|
54
|
+
self.agent_report_threshold = agent_report_threshold
|
|
55
|
+
self.use_agent_report_evaluator = use_agent_report_evaluator
|
|
56
|
+
self.report_scorer = report_scorer
|
|
57
|
+
self.evidence_scorer_config = (
|
|
58
|
+
dict(evidence_scorer_config) if evidence_scorer_config is not None else None
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
def evaluate_candidate(self, candidate: AgentCandidate) -> CandidateEvaluation:
|
|
62
|
+
agent = self._build_agent(candidate)
|
|
63
|
+
runner = self._get_runner()
|
|
64
|
+
run_kwargs = {
|
|
65
|
+
**self._runner_kwargs(candidate),
|
|
66
|
+
**candidate.config.get("simulation", {}),
|
|
67
|
+
}
|
|
68
|
+
if self.scenario is not None:
|
|
69
|
+
run_kwargs["scenario"] = self.scenario
|
|
70
|
+
if self.topic is not None:
|
|
71
|
+
run_kwargs["topic"] = self.topic
|
|
72
|
+
|
|
73
|
+
report = _run_sync(
|
|
74
|
+
runner.run_test(
|
|
75
|
+
agent_callback=agent,
|
|
76
|
+
**run_kwargs,
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
if self.eval_specs is not None or self.eval_templates is not None:
|
|
81
|
+
report = self._evaluate_report(report)
|
|
82
|
+
|
|
83
|
+
agent_report_evaluation = None
|
|
84
|
+
if self.use_agent_report_evaluator or self.agent_report_config is not None:
|
|
85
|
+
agent_report_evaluation = self._evaluate_agent_report(report, candidate)
|
|
86
|
+
|
|
87
|
+
evidence_evaluation = None
|
|
88
|
+
if self.evidence_scorer_config is not None and agent_report_evaluation is None:
|
|
89
|
+
evidence_evaluation = score_simulation_evidence(
|
|
90
|
+
report,
|
|
91
|
+
candidate=candidate,
|
|
92
|
+
config=self.evidence_scorer_config,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
score_source = (
|
|
96
|
+
agent_report_evaluation
|
|
97
|
+
if agent_report_evaluation is not None
|
|
98
|
+
else evidence_evaluation
|
|
99
|
+
if evidence_evaluation is not None
|
|
100
|
+
else report
|
|
101
|
+
)
|
|
102
|
+
score = self._score_report(score_source, candidate)
|
|
103
|
+
metadata = {"source": "simulate-sdk"}
|
|
104
|
+
if agent_report_evaluation is not None:
|
|
105
|
+
metadata["agent_report_evaluation"] = _dump_model(agent_report_evaluation)
|
|
106
|
+
if evidence_evaluation is not None:
|
|
107
|
+
metadata["simulation_evidence_score"] = evidence_evaluation.metadata.get(
|
|
108
|
+
"simulation_evidence_score"
|
|
109
|
+
)
|
|
110
|
+
return CandidateEvaluation(
|
|
111
|
+
candidate=candidate,
|
|
112
|
+
score=score,
|
|
113
|
+
report=report,
|
|
114
|
+
metadata=metadata,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
def _build_agent(self, candidate: AgentCandidate) -> Any:
|
|
118
|
+
if self.agent_factory is not None:
|
|
119
|
+
return self.agent_factory(candidate)
|
|
120
|
+
agent = candidate.config.get("agent_callback") or candidate.config.get("agent")
|
|
121
|
+
if agent is None:
|
|
122
|
+
raise ValueError(
|
|
123
|
+
"SimulationEvaluator needs an agent_factory or candidate.config['agent_callback']."
|
|
124
|
+
)
|
|
125
|
+
return agent
|
|
126
|
+
|
|
127
|
+
def _get_runner(self) -> Any:
|
|
128
|
+
if self.runner is not None:
|
|
129
|
+
return self.runner
|
|
130
|
+
if self.runner_cls is not None:
|
|
131
|
+
return self.runner_cls()
|
|
132
|
+
try:
|
|
133
|
+
from fi.simulate import TestRunner
|
|
134
|
+
except Exception as exc: # pragma: no cover - import clarity
|
|
135
|
+
raise RuntimeError(
|
|
136
|
+
"agent-simulate is required for SimulationEvaluator unless runner/runner_cls is provided."
|
|
137
|
+
) from exc
|
|
138
|
+
return TestRunner()
|
|
139
|
+
|
|
140
|
+
def _evaluate_report(self, report: Any) -> Any:
|
|
141
|
+
evaluate_report = self.evaluate_report_fn
|
|
142
|
+
if evaluate_report is None:
|
|
143
|
+
try:
|
|
144
|
+
from fi.simulate.evaluation import evaluate_report as imported
|
|
145
|
+
except Exception as exc: # pragma: no cover - import clarity
|
|
146
|
+
raise RuntimeError(
|
|
147
|
+
"simulate-sdk evaluate_report is required unless evaluate_report_fn is provided."
|
|
148
|
+
) from exc
|
|
149
|
+
evaluate_report = imported
|
|
150
|
+
|
|
151
|
+
kwargs = dict(self.evaluate_report_kwargs)
|
|
152
|
+
if self.eval_specs is not None:
|
|
153
|
+
kwargs["eval_specs"] = self.eval_specs
|
|
154
|
+
if self.eval_templates is not None:
|
|
155
|
+
kwargs["eval_templates"] = self.eval_templates
|
|
156
|
+
return evaluate_report(report, **kwargs)
|
|
157
|
+
|
|
158
|
+
def _evaluate_agent_report(
|
|
159
|
+
self,
|
|
160
|
+
report: Any,
|
|
161
|
+
candidate: AgentCandidate,
|
|
162
|
+
) -> Any:
|
|
163
|
+
config = self._agent_report_config(candidate)
|
|
164
|
+
try:
|
|
165
|
+
from fi.simulate.evaluation import evaluate_agent_report
|
|
166
|
+
except Exception:
|
|
167
|
+
try:
|
|
168
|
+
from fi.evals.metrics.agents import evaluate_agent_report
|
|
169
|
+
except Exception as exc: # pragma: no cover - import clarity
|
|
170
|
+
raise RuntimeError(
|
|
171
|
+
"SimulationEvaluator local agent report scoring requires "
|
|
172
|
+
"simulate-sdk with evaluate_agent_report or ai-evaluation>=1.1."
|
|
173
|
+
) from exc
|
|
174
|
+
|
|
175
|
+
return evaluate_agent_report(
|
|
176
|
+
report,
|
|
177
|
+
config=config,
|
|
178
|
+
threshold=self.agent_report_threshold,
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
def _agent_report_config(self, candidate: AgentCandidate) -> Dict[str, Any]:
|
|
182
|
+
config = self.agent_report_config
|
|
183
|
+
if callable(config):
|
|
184
|
+
return dict(config(candidate))
|
|
185
|
+
return dict(config or {})
|
|
186
|
+
|
|
187
|
+
def _runner_kwargs(self, candidate: AgentCandidate) -> Dict[str, Any]:
|
|
188
|
+
config = self.runner_kwargs
|
|
189
|
+
if callable(config):
|
|
190
|
+
return dict(config(candidate))
|
|
191
|
+
return dict(config or {})
|
|
192
|
+
|
|
193
|
+
def _score_report(self, report: Any, candidate: AgentCandidate) -> float:
|
|
194
|
+
if self.report_scorer is not None:
|
|
195
|
+
return float(self.report_scorer(report, candidate))
|
|
196
|
+
direct_score = _coerce_score(getattr(report, "score", None))
|
|
197
|
+
if direct_score is not None:
|
|
198
|
+
return direct_score
|
|
199
|
+
scores = list(_iter_report_scores(report))
|
|
200
|
+
if not scores:
|
|
201
|
+
return 0.0
|
|
202
|
+
return sum(scores) / len(scores)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _run_sync(value: Any) -> Any:
|
|
206
|
+
if not inspect.isawaitable(value):
|
|
207
|
+
return value
|
|
208
|
+
try:
|
|
209
|
+
asyncio.get_running_loop()
|
|
210
|
+
except RuntimeError:
|
|
211
|
+
return asyncio.run(value)
|
|
212
|
+
raise RuntimeError(
|
|
213
|
+
"SimulationEvaluator.evaluate_candidate() was called from a running event loop. "
|
|
214
|
+
"Pass a synchronous runner or call it outside the event loop for now."
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _iter_report_scores(report: Any):
|
|
219
|
+
for result in getattr(report, "results", []) or []:
|
|
220
|
+
evaluation = getattr(result, "evaluation", None)
|
|
221
|
+
if not isinstance(evaluation, dict):
|
|
222
|
+
continue
|
|
223
|
+
for item in evaluation.values():
|
|
224
|
+
if isinstance(item, dict):
|
|
225
|
+
for key in ("score", "output", "value"):
|
|
226
|
+
value = item.get(key)
|
|
227
|
+
score = _coerce_score(value)
|
|
228
|
+
if score is not None:
|
|
229
|
+
yield score
|
|
230
|
+
break
|
|
231
|
+
else:
|
|
232
|
+
score = _coerce_score(item)
|
|
233
|
+
if score is not None:
|
|
234
|
+
yield score
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _coerce_score(value: Any) -> Optional[float]:
|
|
238
|
+
if isinstance(value, bool):
|
|
239
|
+
return 1.0 if value else 0.0
|
|
240
|
+
if isinstance(value, (int, float)):
|
|
241
|
+
return max(0.0, min(1.0, float(value)))
|
|
242
|
+
if isinstance(value, str):
|
|
243
|
+
lowered = value.strip().lower()
|
|
244
|
+
if lowered in {"pass", "passed", "true", "yes"}:
|
|
245
|
+
return 1.0
|
|
246
|
+
if lowered in {"fail", "failed", "false", "no"}:
|
|
247
|
+
return 0.0
|
|
248
|
+
try:
|
|
249
|
+
return max(0.0, min(1.0, float(lowered)))
|
|
250
|
+
except ValueError:
|
|
251
|
+
return None
|
|
252
|
+
return None
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _dump_model(value: Any) -> Any:
|
|
256
|
+
if hasattr(value, "model_dump"):
|
|
257
|
+
return value.model_dump()
|
|
258
|
+
if hasattr(value, "dict"):
|
|
259
|
+
return value.dict()
|
|
260
|
+
return value
|
fi/opt/targets.py
ADDED
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import hashlib
|
|
5
|
+
import itertools
|
|
6
|
+
import json
|
|
7
|
+
from typing import Any, Dict, Iterable, List, Literal, Optional
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel, Field
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
OptimizationLayer = Literal[
|
|
13
|
+
"objective",
|
|
14
|
+
"harness",
|
|
15
|
+
"integration",
|
|
16
|
+
"framework",
|
|
17
|
+
"streaming",
|
|
18
|
+
"world",
|
|
19
|
+
"security",
|
|
20
|
+
"perception",
|
|
21
|
+
"prompt",
|
|
22
|
+
"planner",
|
|
23
|
+
"autonomy",
|
|
24
|
+
"policy",
|
|
25
|
+
"tools",
|
|
26
|
+
"memory",
|
|
27
|
+
"router",
|
|
28
|
+
"graph",
|
|
29
|
+
"retrieval",
|
|
30
|
+
"retriever",
|
|
31
|
+
"model",
|
|
32
|
+
"voice",
|
|
33
|
+
"browser",
|
|
34
|
+
"cua",
|
|
35
|
+
"multi_agent",
|
|
36
|
+
"orchestration",
|
|
37
|
+
"action",
|
|
38
|
+
"environment",
|
|
39
|
+
"implementation",
|
|
40
|
+
"evaluator",
|
|
41
|
+
"custom",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AgentCandidate(BaseModel):
|
|
46
|
+
"""
|
|
47
|
+
A concrete agent/workflow configuration to evaluate.
|
|
48
|
+
|
|
49
|
+
`config` is intentionally framework-neutral. It can represent a LangGraph
|
|
50
|
+
graph config, CrewAI crew inputs, LiveKit voice settings, Pipecat pipeline
|
|
51
|
+
parameters, browser/CUA policy, tool schemas, memory settings, or a plain
|
|
52
|
+
prompt template.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
id: str
|
|
56
|
+
config: Dict[str, Any]
|
|
57
|
+
target_name: Optional[str] = None
|
|
58
|
+
layers: List[OptimizationLayer] = Field(default_factory=list)
|
|
59
|
+
parent_id: Optional[str] = None
|
|
60
|
+
patch: Dict[str, Any] = Field(default_factory=dict)
|
|
61
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
62
|
+
|
|
63
|
+
@classmethod
|
|
64
|
+
def from_config(
|
|
65
|
+
cls,
|
|
66
|
+
config: Dict[str, Any],
|
|
67
|
+
*,
|
|
68
|
+
target_name: Optional[str] = None,
|
|
69
|
+
layers: Optional[List[OptimizationLayer]] = None,
|
|
70
|
+
parent_id: Optional[str] = None,
|
|
71
|
+
patch: Optional[Dict[str, Any]] = None,
|
|
72
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
73
|
+
) -> "AgentCandidate":
|
|
74
|
+
payload = {
|
|
75
|
+
"target_name": target_name,
|
|
76
|
+
"config": config,
|
|
77
|
+
"patch": patch or {},
|
|
78
|
+
}
|
|
79
|
+
digest = hashlib.sha256(
|
|
80
|
+
json.dumps(payload, sort_keys=True, default=str).encode("utf-8")
|
|
81
|
+
).hexdigest()[:16]
|
|
82
|
+
return cls(
|
|
83
|
+
id=f"candidate_{digest}",
|
|
84
|
+
config=copy.deepcopy(config),
|
|
85
|
+
target_name=target_name,
|
|
86
|
+
layers=list(layers or []),
|
|
87
|
+
parent_id=parent_id,
|
|
88
|
+
patch=copy.deepcopy(patch or {}),
|
|
89
|
+
metadata=copy.deepcopy(metadata or {}),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def get_path(self, path: str, default: Any = None) -> Any:
|
|
93
|
+
current: Any = self.config
|
|
94
|
+
for part in _split_path(path):
|
|
95
|
+
if isinstance(current, dict) and part in current:
|
|
96
|
+
current = current[part]
|
|
97
|
+
elif isinstance(current, list) and part.isdigit() and int(part) < len(current):
|
|
98
|
+
current = current[int(part)]
|
|
99
|
+
else:
|
|
100
|
+
return default
|
|
101
|
+
return current
|
|
102
|
+
|
|
103
|
+
def with_patch(
|
|
104
|
+
self,
|
|
105
|
+
patch: Dict[str, Any],
|
|
106
|
+
*,
|
|
107
|
+
layers: Optional[List[OptimizationLayer]] = None,
|
|
108
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
109
|
+
) -> "AgentCandidate":
|
|
110
|
+
new_config = copy.deepcopy(self.config)
|
|
111
|
+
for path, value in patch.items():
|
|
112
|
+
set_path(new_config, path, value)
|
|
113
|
+
merged_metadata = {**self.metadata, **(metadata or {})}
|
|
114
|
+
return AgentCandidate.from_config(
|
|
115
|
+
new_config,
|
|
116
|
+
target_name=self.target_name,
|
|
117
|
+
layers=layers or self.layers,
|
|
118
|
+
parent_id=self.id,
|
|
119
|
+
patch=patch,
|
|
120
|
+
metadata=merged_metadata,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class OptimizationTarget(BaseModel):
|
|
125
|
+
"""
|
|
126
|
+
Framework-neutral optimization target.
|
|
127
|
+
|
|
128
|
+
`search_space` maps dot paths to candidate values. Examples:
|
|
129
|
+
- `prompt.system`: ["Be concise", "Ask one clarifying question first"]
|
|
130
|
+
- `tools.0.description`: ["Search orders by id", "Search orders by id and email"]
|
|
131
|
+
- `memory.strategy`: ["buffer", "summary", "vector"]
|
|
132
|
+
- `router.default_model`: ["gpt-4o-mini", "claude-haiku"]
|
|
133
|
+
- `voice.vad.min_silence_duration`: [0.1, 0.3, 0.5]
|
|
134
|
+
- `browser.policy.allow_cross_origin`: [False, True]
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
name: str
|
|
138
|
+
base_config: Dict[str, Any]
|
|
139
|
+
layers: List[OptimizationLayer] = Field(default_factory=lambda: ["prompt"])
|
|
140
|
+
search_space: Dict[str, List[Any]] = Field(default_factory=dict)
|
|
141
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
142
|
+
|
|
143
|
+
def seed_candidate(self) -> AgentCandidate:
|
|
144
|
+
return AgentCandidate.from_config(
|
|
145
|
+
self.base_config,
|
|
146
|
+
target_name=self.name,
|
|
147
|
+
layers=self.layers,
|
|
148
|
+
metadata={"kind": "seed", **self.metadata},
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
def iter_candidates(
|
|
152
|
+
self,
|
|
153
|
+
*,
|
|
154
|
+
include_seed: bool = True,
|
|
155
|
+
max_candidates: Optional[int] = None,
|
|
156
|
+
) -> Iterable[AgentCandidate]:
|
|
157
|
+
count = 0
|
|
158
|
+
if include_seed:
|
|
159
|
+
yield self.seed_candidate()
|
|
160
|
+
count += 1
|
|
161
|
+
if max_candidates is not None and count >= max_candidates:
|
|
162
|
+
return
|
|
163
|
+
|
|
164
|
+
if not self.search_space:
|
|
165
|
+
return
|
|
166
|
+
|
|
167
|
+
paths = list(self.search_space.keys())
|
|
168
|
+
value_lists = [self.search_space[path] for path in paths]
|
|
169
|
+
for values in itertools.product(*value_lists):
|
|
170
|
+
patch = dict(zip(paths, values))
|
|
171
|
+
# Avoid duplicating the seed when every patch value equals base config.
|
|
172
|
+
if all(self.seed_candidate().get_path(path) == value for path, value in patch.items()):
|
|
173
|
+
continue
|
|
174
|
+
yield self.seed_candidate().with_patch(
|
|
175
|
+
patch,
|
|
176
|
+
metadata={"kind": "search", "search_paths": paths, **self.metadata},
|
|
177
|
+
)
|
|
178
|
+
count += 1
|
|
179
|
+
if max_candidates is not None and count >= max_candidates:
|
|
180
|
+
return
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
class CandidateEvaluation(BaseModel):
|
|
184
|
+
"""Score and evidence for one evaluated candidate."""
|
|
185
|
+
|
|
186
|
+
candidate: AgentCandidate
|
|
187
|
+
score: float
|
|
188
|
+
reason: str = ""
|
|
189
|
+
individual_results: List[Any] = Field(default_factory=list)
|
|
190
|
+
report: Any = None
|
|
191
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def set_path(config: Dict[str, Any], path: str, value: Any) -> None:
|
|
195
|
+
parts = _split_path(path)
|
|
196
|
+
if not parts:
|
|
197
|
+
raise ValueError("Path cannot be empty.")
|
|
198
|
+
|
|
199
|
+
current: Any = config
|
|
200
|
+
for index, part in enumerate(parts[:-1]):
|
|
201
|
+
next_part = parts[index + 1]
|
|
202
|
+
if isinstance(current, list):
|
|
203
|
+
if not part.isdigit():
|
|
204
|
+
raise ValueError(f"Expected numeric list index in path '{path}'.")
|
|
205
|
+
list_index = int(part)
|
|
206
|
+
_ensure_list_size(current, list_index)
|
|
207
|
+
if current[list_index] is None:
|
|
208
|
+
current[list_index] = [] if next_part.isdigit() else {}
|
|
209
|
+
current = current[list_index]
|
|
210
|
+
else:
|
|
211
|
+
if part not in current or current[part] is None:
|
|
212
|
+
current[part] = [] if next_part.isdigit() else {}
|
|
213
|
+
current = current[part]
|
|
214
|
+
|
|
215
|
+
final = parts[-1]
|
|
216
|
+
if isinstance(current, list):
|
|
217
|
+
if not final.isdigit():
|
|
218
|
+
raise ValueError(f"Expected numeric list index in path '{path}'.")
|
|
219
|
+
list_index = int(final)
|
|
220
|
+
_ensure_list_size(current, list_index)
|
|
221
|
+
current[list_index] = value
|
|
222
|
+
else:
|
|
223
|
+
current[final] = value
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _split_path(path: str) -> List[str]:
|
|
227
|
+
return [part for part in path.split(".") if part]
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _ensure_list_size(items: List[Any], index: int) -> None:
|
|
231
|
+
while len(items) <= index:
|
|
232
|
+
items.append(None)
|
fi/opt/types.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
from pydantic import BaseModel, Field
|
|
2
|
+
from typing import Any, List, Dict, Optional
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class LLMMessage(BaseModel):
|
|
6
|
+
"""Every message sent and received by the LLM MUST follow this format."""
|
|
7
|
+
|
|
8
|
+
role: str
|
|
9
|
+
content: str
|
|
10
|
+
name: Optional[str] = None
|
|
11
|
+
function_call: Optional[str] = None
|
|
12
|
+
tool_call_id: Optional[str] = None
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class EvaluationResult(BaseModel):
|
|
16
|
+
"""
|
|
17
|
+
A standardized result from a single evaluation.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
score: float = Field(..., description="The normalized score (0.0 to 1.0).")
|
|
21
|
+
reason: str = Field("", description="The explanation for the score.")
|
|
22
|
+
metadata: Dict[str, Any] = Field(
|
|
23
|
+
default_factory=dict, description="Any other metadata from the evaluator."
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class IterationHistory(BaseModel):
|
|
28
|
+
"""
|
|
29
|
+
A detailed record of a single optimization iteration.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
prompt: str
|
|
33
|
+
average_score: float
|
|
34
|
+
individual_results: List[EvaluationResult]
|
|
35
|
+
candidate_id: Optional[str] = None
|
|
36
|
+
candidate_config: Optional[Dict[str, Any]] = None
|
|
37
|
+
layers: List[str] = Field(default_factory=list)
|
|
38
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class OptimizationResult(BaseModel):
|
|
42
|
+
"""Output Model to hold the results of an optimization run."""
|
|
43
|
+
|
|
44
|
+
best_generator: Any
|
|
45
|
+
best_candidate: Any = None
|
|
46
|
+
history: List[IterationHistory]
|
|
47
|
+
final_score: float = 0.0
|
|
48
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
49
|
+
|
|
50
|
+
# Early stopping metadata
|
|
51
|
+
early_stopped: bool = Field(
|
|
52
|
+
default=False,
|
|
53
|
+
description="Whether optimization was terminated early by a stopping criterion"
|
|
54
|
+
)
|
|
55
|
+
stop_reason: Optional[str] = Field(
|
|
56
|
+
default=None,
|
|
57
|
+
description="Explanation for early stopping (if applicable)"
|
|
58
|
+
)
|
|
59
|
+
total_iterations: int = Field(
|
|
60
|
+
default=0,
|
|
61
|
+
description="Total number of iterations completed"
|
|
62
|
+
)
|
|
63
|
+
total_evaluations: int = Field(
|
|
64
|
+
default=0,
|
|
65
|
+
description="Total number of dataset evaluations performed"
|
|
66
|
+
)
|
fi/opt/utils/__init__.py
ADDED