agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,647 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Unified Evaluator API.
|
|
3
|
+
|
|
4
|
+
Provides a single interface for running evaluations in any mode:
|
|
5
|
+
- Blocking: Synchronous execution, waits for results
|
|
6
|
+
- Non-blocking: Background execution, zero latency
|
|
7
|
+
- Distributed: Scalable execution via pluggable backends
|
|
8
|
+
|
|
9
|
+
Example:
|
|
10
|
+
from fi.evals.framework import Evaluator, ExecutionMode
|
|
11
|
+
|
|
12
|
+
# Create evaluator
|
|
13
|
+
evaluator = Evaluator(
|
|
14
|
+
evaluations=[ToxicityEval(), BiasEval()],
|
|
15
|
+
mode=ExecutionMode.NON_BLOCKING,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
# Run evaluations (returns immediately in non-blocking mode)
|
|
19
|
+
result = evaluator.run({"response": "..."})
|
|
20
|
+
|
|
21
|
+
# Get results when needed
|
|
22
|
+
if result.is_future:
|
|
23
|
+
batch = result.wait()
|
|
24
|
+
else:
|
|
25
|
+
batch = result.batch
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
import logging
|
|
29
|
+
import time
|
|
30
|
+
from typing import Dict, Any, List, Optional, Callable
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from datetime import datetime, timezone
|
|
33
|
+
|
|
34
|
+
from .types import ExecutionMode, FrameworkEvalResult, BatchEvalResult, EvalStatus
|
|
35
|
+
from .context import EvalContext
|
|
36
|
+
from .protocols import BaseEvaluation, EvalRegistry
|
|
37
|
+
from .evaluators.blocking import BlockingEvaluator
|
|
38
|
+
from .evaluators.non_blocking import (
|
|
39
|
+
NonBlockingEvaluator,
|
|
40
|
+
BatchEvalFuture,
|
|
41
|
+
)
|
|
42
|
+
from .backends import Backend, ThreadPoolBackend
|
|
43
|
+
|
|
44
|
+
logger = logging.getLogger(__name__)
|
|
45
|
+
|
|
46
|
+
# Internal alias for brevity — this is the framework-level EvalResult
|
|
47
|
+
EvalResult = FrameworkEvalResult
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class EvaluatorResult:
|
|
52
|
+
"""
|
|
53
|
+
Result from an evaluator run.
|
|
54
|
+
|
|
55
|
+
Wraps either immediate results (blocking) or futures (non-blocking).
|
|
56
|
+
|
|
57
|
+
Attributes:
|
|
58
|
+
batch: Immediate BatchEvalResult (blocking mode)
|
|
59
|
+
future: BatchEvalFuture for async results (non-blocking mode)
|
|
60
|
+
mode: The execution mode used
|
|
61
|
+
submitted_at: When the evaluation was submitted
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
batch: Optional[BatchEvalResult] = None
|
|
65
|
+
future: Optional[BatchEvalFuture] = None
|
|
66
|
+
mode: ExecutionMode = ExecutionMode.BLOCKING
|
|
67
|
+
submitted_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def is_future(self) -> bool:
|
|
71
|
+
"""Whether this result is a future (non-blocking)."""
|
|
72
|
+
return self.future is not None
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def is_ready(self) -> bool:
|
|
76
|
+
"""Whether results are ready."""
|
|
77
|
+
if self.batch is not None:
|
|
78
|
+
return True
|
|
79
|
+
if self.future is not None:
|
|
80
|
+
return self.future.done()
|
|
81
|
+
return False
|
|
82
|
+
|
|
83
|
+
def wait(self, timeout: Optional[float] = None) -> BatchEvalResult:
|
|
84
|
+
"""
|
|
85
|
+
Wait for and return results.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
timeout: Maximum seconds to wait (non-blocking only)
|
|
89
|
+
|
|
90
|
+
Returns:
|
|
91
|
+
BatchEvalResult with all evaluation results
|
|
92
|
+
"""
|
|
93
|
+
if self.batch is not None:
|
|
94
|
+
return self.batch
|
|
95
|
+
if self.future is not None:
|
|
96
|
+
return self.future.results(timeout=timeout)
|
|
97
|
+
raise ValueError("No results available")
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def results(self) -> List[EvalResult]:
|
|
101
|
+
"""Get individual results (waits if necessary)."""
|
|
102
|
+
return self.wait().results
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def success_rate(self) -> float:
|
|
106
|
+
"""Get success rate (waits if necessary)."""
|
|
107
|
+
return self.wait().success_rate
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class FrameworkEvaluator:
|
|
111
|
+
"""
|
|
112
|
+
Unified evaluator for all execution modes.
|
|
113
|
+
|
|
114
|
+
Provides a consistent interface regardless of how evaluations are executed.
|
|
115
|
+
Supports blocking, non-blocking, and distributed modes with automatic
|
|
116
|
+
span enrichment.
|
|
117
|
+
|
|
118
|
+
Example:
|
|
119
|
+
# Simple blocking usage
|
|
120
|
+
evaluator = FrameworkEvaluator([ToxicityEval()])
|
|
121
|
+
result = evaluator.run({"response": "..."})
|
|
122
|
+
print(f"Score: {result.results[0].value}")
|
|
123
|
+
|
|
124
|
+
# Non-blocking for production
|
|
125
|
+
evaluator = FrameworkEvaluator(
|
|
126
|
+
[ToxicityEval(), BiasEval()],
|
|
127
|
+
mode=ExecutionMode.NON_BLOCKING,
|
|
128
|
+
)
|
|
129
|
+
result = evaluator.run({"response": "..."}) # Returns immediately
|
|
130
|
+
# ... do other work ...
|
|
131
|
+
batch = result.wait() # Get results when needed
|
|
132
|
+
|
|
133
|
+
# With custom backend
|
|
134
|
+
evaluator = FrameworkEvaluator(
|
|
135
|
+
[ToxicityEval()],
|
|
136
|
+
mode=ExecutionMode.NON_BLOCKING,
|
|
137
|
+
backend=MyTemporalBackend(),
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
Thread Safety:
|
|
141
|
+
This class is thread-safe. Multiple threads can call run() concurrently.
|
|
142
|
+
"""
|
|
143
|
+
|
|
144
|
+
def __init__(
|
|
145
|
+
self,
|
|
146
|
+
evaluations: Optional[List[BaseEvaluation]] = None,
|
|
147
|
+
mode: ExecutionMode = ExecutionMode.BLOCKING,
|
|
148
|
+
auto_enrich_span: bool = True,
|
|
149
|
+
fail_fast: bool = False,
|
|
150
|
+
validate_inputs: bool = True,
|
|
151
|
+
max_workers: int = 4,
|
|
152
|
+
backend: Optional[Backend] = None,
|
|
153
|
+
):
|
|
154
|
+
"""
|
|
155
|
+
Initialize the evaluator.
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
evaluations: List of evaluations to run
|
|
159
|
+
mode: Execution mode (BLOCKING, NON_BLOCKING, DISTRIBUTED)
|
|
160
|
+
auto_enrich_span: Whether to automatically enrich OTEL spans
|
|
161
|
+
fail_fast: Stop on first failure
|
|
162
|
+
validate_inputs: Whether to validate inputs before evaluation
|
|
163
|
+
max_workers: Max concurrent workers (non-blocking/distributed)
|
|
164
|
+
backend: Custom backend for execution (uses ThreadPool if None)
|
|
165
|
+
"""
|
|
166
|
+
self.evaluations = list(evaluations) if evaluations else []
|
|
167
|
+
self.mode = mode
|
|
168
|
+
self.auto_enrich_span = auto_enrich_span
|
|
169
|
+
self.fail_fast = fail_fast
|
|
170
|
+
self.validate_inputs = validate_inputs
|
|
171
|
+
self.max_workers = max_workers
|
|
172
|
+
self._backend = backend
|
|
173
|
+
|
|
174
|
+
# Internal evaluators (created lazily)
|
|
175
|
+
self._blocking: Optional[BlockingEvaluator] = None
|
|
176
|
+
self._non_blocking: Optional[NonBlockingEvaluator] = None
|
|
177
|
+
|
|
178
|
+
def add(self, evaluation: BaseEvaluation) -> "FrameworkEvaluator":
|
|
179
|
+
"""
|
|
180
|
+
Add an evaluation to run.
|
|
181
|
+
|
|
182
|
+
Args:
|
|
183
|
+
evaluation: The evaluation to add
|
|
184
|
+
|
|
185
|
+
Returns:
|
|
186
|
+
Self for chaining
|
|
187
|
+
"""
|
|
188
|
+
self.evaluations.append(evaluation)
|
|
189
|
+
return self
|
|
190
|
+
|
|
191
|
+
def add_by_name(self, name: str, version: str = "latest") -> "FrameworkEvaluator":
|
|
192
|
+
"""
|
|
193
|
+
Add an evaluation by name from the registry.
|
|
194
|
+
|
|
195
|
+
Args:
|
|
196
|
+
name: Evaluation name
|
|
197
|
+
version: Version (default: latest)
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
Self for chaining
|
|
201
|
+
|
|
202
|
+
Raises:
|
|
203
|
+
KeyError: If evaluation not found
|
|
204
|
+
"""
|
|
205
|
+
eval_class = EvalRegistry.get(name, version)
|
|
206
|
+
if eval_class is None:
|
|
207
|
+
raise KeyError(f"Evaluation not found: {name}@{version}")
|
|
208
|
+
self.evaluations.append(eval_class())
|
|
209
|
+
return self
|
|
210
|
+
|
|
211
|
+
def run(
|
|
212
|
+
self,
|
|
213
|
+
inputs: Dict[str, Any],
|
|
214
|
+
context: Optional[EvalContext] = None,
|
|
215
|
+
callback: Optional[Callable[[EvalResult], None]] = None,
|
|
216
|
+
) -> EvaluatorResult:
|
|
217
|
+
"""
|
|
218
|
+
Run all evaluations on the given inputs.
|
|
219
|
+
|
|
220
|
+
Args:
|
|
221
|
+
inputs: Input data for evaluations
|
|
222
|
+
context: Optional trace context for span enrichment
|
|
223
|
+
callback: Optional callback for each result (non-blocking only)
|
|
224
|
+
|
|
225
|
+
Returns:
|
|
226
|
+
EvaluatorResult wrapping results or future
|
|
227
|
+
|
|
228
|
+
Raises:
|
|
229
|
+
ValueError: If no evaluations configured
|
|
230
|
+
"""
|
|
231
|
+
if not self.evaluations:
|
|
232
|
+
raise ValueError("No evaluations configured")
|
|
233
|
+
|
|
234
|
+
if self.mode == ExecutionMode.BLOCKING:
|
|
235
|
+
return self._run_blocking(inputs, context)
|
|
236
|
+
elif self.mode == ExecutionMode.NON_BLOCKING:
|
|
237
|
+
return self._run_non_blocking(inputs, context, callback)
|
|
238
|
+
elif self.mode == ExecutionMode.DISTRIBUTED:
|
|
239
|
+
return self._run_distributed(inputs, context, callback)
|
|
240
|
+
else:
|
|
241
|
+
raise ValueError(f"Unknown execution mode: {self.mode}")
|
|
242
|
+
|
|
243
|
+
def run_single(
|
|
244
|
+
self,
|
|
245
|
+
evaluation: BaseEvaluation,
|
|
246
|
+
inputs: Dict[str, Any],
|
|
247
|
+
context: Optional[EvalContext] = None,
|
|
248
|
+
) -> EvalResult:
|
|
249
|
+
"""
|
|
250
|
+
Run a single evaluation.
|
|
251
|
+
|
|
252
|
+
Always runs in blocking mode for simplicity.
|
|
253
|
+
|
|
254
|
+
Args:
|
|
255
|
+
evaluation: The evaluation to run
|
|
256
|
+
inputs: Input data
|
|
257
|
+
context: Optional trace context
|
|
258
|
+
|
|
259
|
+
Returns:
|
|
260
|
+
Single EvalResult
|
|
261
|
+
"""
|
|
262
|
+
evaluator = self._get_blocking_evaluator()
|
|
263
|
+
results = evaluator.evaluate(inputs, evaluations=[evaluation], context=context)
|
|
264
|
+
return results[0]
|
|
265
|
+
|
|
266
|
+
def _run_blocking(
|
|
267
|
+
self,
|
|
268
|
+
inputs: Dict[str, Any],
|
|
269
|
+
context: Optional[EvalContext],
|
|
270
|
+
) -> EvaluatorResult:
|
|
271
|
+
"""Run evaluations in blocking mode."""
|
|
272
|
+
evaluator = self._get_blocking_evaluator()
|
|
273
|
+
results = evaluator.evaluate(inputs, context=context)
|
|
274
|
+
batch = BatchEvalResult.from_results(results)
|
|
275
|
+
|
|
276
|
+
return EvaluatorResult(
|
|
277
|
+
batch=batch,
|
|
278
|
+
mode=ExecutionMode.BLOCKING,
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
def _run_non_blocking(
|
|
282
|
+
self,
|
|
283
|
+
inputs: Dict[str, Any],
|
|
284
|
+
context: Optional[EvalContext],
|
|
285
|
+
callback: Optional[Callable[[EvalResult], None]],
|
|
286
|
+
) -> EvaluatorResult:
|
|
287
|
+
"""Run evaluations in non-blocking mode."""
|
|
288
|
+
evaluator = self._get_non_blocking_evaluator()
|
|
289
|
+
future = evaluator.evaluate(inputs, context=context, callback=callback)
|
|
290
|
+
|
|
291
|
+
return EvaluatorResult(
|
|
292
|
+
future=future,
|
|
293
|
+
mode=ExecutionMode.NON_BLOCKING,
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
def _run_distributed(
|
|
297
|
+
self,
|
|
298
|
+
inputs: Dict[str, Any],
|
|
299
|
+
context: Optional[EvalContext],
|
|
300
|
+
callback: Optional[Callable[[EvalResult], None]],
|
|
301
|
+
) -> EvaluatorResult:
|
|
302
|
+
"""
|
|
303
|
+
Run evaluations in distributed mode.
|
|
304
|
+
|
|
305
|
+
Uses the configured backend for execution. Submits each evaluation
|
|
306
|
+
as a task to the backend, collects results, and returns a BatchEvalResult.
|
|
307
|
+
Falls back to non-blocking if no backend is configured.
|
|
308
|
+
"""
|
|
309
|
+
if self._backend is None:
|
|
310
|
+
return self._run_non_blocking(inputs, context, callback)
|
|
311
|
+
|
|
312
|
+
context_dict = None
|
|
313
|
+
if context and hasattr(context, "to_dict"):
|
|
314
|
+
context_dict = context.to_dict()
|
|
315
|
+
|
|
316
|
+
# Submit each evaluation as a task to the backend, collecting handles
|
|
317
|
+
# or immediate failure results
|
|
318
|
+
handles = [] # List of (index, handle) for successful submissions
|
|
319
|
+
results = [None] * len(self.evaluations) # Pre-allocate result slots
|
|
320
|
+
|
|
321
|
+
for i, evaluation in enumerate(self.evaluations):
|
|
322
|
+
eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
|
|
323
|
+
eval_version = getattr(evaluation, "version", "1.0.0")
|
|
324
|
+
try:
|
|
325
|
+
handle = self._backend.submit(
|
|
326
|
+
_execute_single_evaluation,
|
|
327
|
+
args=(evaluation, inputs, self.validate_inputs),
|
|
328
|
+
context=context_dict,
|
|
329
|
+
)
|
|
330
|
+
handles.append((i, handle))
|
|
331
|
+
except Exception as e:
|
|
332
|
+
error_result = EvalResult(
|
|
333
|
+
value=None,
|
|
334
|
+
eval_name=eval_name,
|
|
335
|
+
eval_version=eval_version,
|
|
336
|
+
latency_ms=0.0,
|
|
337
|
+
status=EvalStatus.FAILED,
|
|
338
|
+
error=str(e),
|
|
339
|
+
)
|
|
340
|
+
results[i] = error_result
|
|
341
|
+
if callback:
|
|
342
|
+
callback(error_result)
|
|
343
|
+
|
|
344
|
+
# Collect results from successful submissions
|
|
345
|
+
timeout = self._backend_timeout
|
|
346
|
+
for i, handle in handles:
|
|
347
|
+
evaluation = self.evaluations[i]
|
|
348
|
+
eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
|
|
349
|
+
eval_version = getattr(evaluation, "version", "1.0.0")
|
|
350
|
+
try:
|
|
351
|
+
result = self._backend.get_result(handle, timeout=timeout)
|
|
352
|
+
results[i] = result
|
|
353
|
+
if callback:
|
|
354
|
+
callback(result)
|
|
355
|
+
except Exception as e:
|
|
356
|
+
error_result = EvalResult(
|
|
357
|
+
value=None,
|
|
358
|
+
eval_name=eval_name,
|
|
359
|
+
eval_version=eval_version,
|
|
360
|
+
latency_ms=0.0,
|
|
361
|
+
status=EvalStatus.FAILED,
|
|
362
|
+
error=str(e),
|
|
363
|
+
)
|
|
364
|
+
results[i] = error_result
|
|
365
|
+
if callback:
|
|
366
|
+
callback(error_result)
|
|
367
|
+
|
|
368
|
+
batch = BatchEvalResult.from_results(results)
|
|
369
|
+
return EvaluatorResult(
|
|
370
|
+
batch=batch,
|
|
371
|
+
mode=ExecutionMode.DISTRIBUTED,
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
@property
|
|
375
|
+
def _backend_timeout(self) -> float:
|
|
376
|
+
"""Get timeout from backend config, defaulting to 300s."""
|
|
377
|
+
if self._backend and hasattr(self._backend, "config"):
|
|
378
|
+
config = self._backend.config
|
|
379
|
+
if hasattr(config, "timeout_seconds"):
|
|
380
|
+
return config.timeout_seconds
|
|
381
|
+
return 300.0
|
|
382
|
+
|
|
383
|
+
def _get_blocking_evaluator(self) -> BlockingEvaluator:
|
|
384
|
+
"""Get or create the blocking evaluator."""
|
|
385
|
+
if self._blocking is None:
|
|
386
|
+
self._blocking = BlockingEvaluator(
|
|
387
|
+
evaluations=self.evaluations,
|
|
388
|
+
auto_enrich_span=self.auto_enrich_span,
|
|
389
|
+
fail_fast=self.fail_fast,
|
|
390
|
+
validate_inputs=self.validate_inputs,
|
|
391
|
+
)
|
|
392
|
+
return self._blocking
|
|
393
|
+
|
|
394
|
+
def _get_non_blocking_evaluator(self) -> NonBlockingEvaluator:
|
|
395
|
+
"""Get or create the non-blocking evaluator."""
|
|
396
|
+
if self._non_blocking is None:
|
|
397
|
+
self._non_blocking = NonBlockingEvaluator(
|
|
398
|
+
evaluations=self.evaluations,
|
|
399
|
+
max_workers=self.max_workers,
|
|
400
|
+
auto_enrich_span=self.auto_enrich_span,
|
|
401
|
+
fail_fast=self.fail_fast,
|
|
402
|
+
validate_inputs=self.validate_inputs,
|
|
403
|
+
)
|
|
404
|
+
return self._non_blocking
|
|
405
|
+
|
|
406
|
+
def shutdown(self, wait: bool = True) -> None:
|
|
407
|
+
"""
|
|
408
|
+
Shutdown the evaluator and release resources.
|
|
409
|
+
|
|
410
|
+
Args:
|
|
411
|
+
wait: Whether to wait for pending evaluations
|
|
412
|
+
"""
|
|
413
|
+
if self._non_blocking:
|
|
414
|
+
self._non_blocking.shutdown(wait=wait)
|
|
415
|
+
self._non_blocking = None
|
|
416
|
+
if self._backend:
|
|
417
|
+
self._backend.shutdown(wait=wait)
|
|
418
|
+
|
|
419
|
+
def __enter__(self) -> "FrameworkEvaluator":
|
|
420
|
+
return self
|
|
421
|
+
|
|
422
|
+
def __exit__(self, exc_type, exc_val, exc_tb) -> None:
|
|
423
|
+
self.shutdown(wait=True)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
# Factory functions for common configurations
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def blocking_evaluator(
|
|
430
|
+
*evaluations: BaseEvaluation,
|
|
431
|
+
auto_enrich_span: bool = True,
|
|
432
|
+
fail_fast: bool = False,
|
|
433
|
+
) -> FrameworkEvaluator:
|
|
434
|
+
"""
|
|
435
|
+
Create a blocking evaluator.
|
|
436
|
+
|
|
437
|
+
Args:
|
|
438
|
+
*evaluations: Evaluations to run
|
|
439
|
+
auto_enrich_span: Whether to enrich OTEL spans
|
|
440
|
+
fail_fast: Stop on first failure
|
|
441
|
+
|
|
442
|
+
Returns:
|
|
443
|
+
Configured Evaluator in blocking mode
|
|
444
|
+
|
|
445
|
+
Example:
|
|
446
|
+
evaluator = blocking_evaluator(ToxicityEval(), BiasEval())
|
|
447
|
+
result = evaluator.run({"response": "..."})
|
|
448
|
+
"""
|
|
449
|
+
return FrameworkEvaluator(
|
|
450
|
+
evaluations=list(evaluations),
|
|
451
|
+
mode=ExecutionMode.BLOCKING,
|
|
452
|
+
auto_enrich_span=auto_enrich_span,
|
|
453
|
+
fail_fast=fail_fast,
|
|
454
|
+
)
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def async_evaluator(
|
|
458
|
+
*evaluations: BaseEvaluation,
|
|
459
|
+
max_workers: int = 4,
|
|
460
|
+
auto_enrich_span: bool = True,
|
|
461
|
+
backend: Optional[Backend] = None,
|
|
462
|
+
) -> FrameworkEvaluator:
|
|
463
|
+
"""
|
|
464
|
+
Create a non-blocking (async) evaluator.
|
|
465
|
+
|
|
466
|
+
Args:
|
|
467
|
+
*evaluations: Evaluations to run
|
|
468
|
+
max_workers: Maximum concurrent evaluations
|
|
469
|
+
auto_enrich_span: Whether to enrich OTEL spans
|
|
470
|
+
backend: Custom execution backend
|
|
471
|
+
|
|
472
|
+
Returns:
|
|
473
|
+
Configured Evaluator in non-blocking mode
|
|
474
|
+
|
|
475
|
+
Example:
|
|
476
|
+
evaluator = async_evaluator(ToxicityEval(), BiasEval())
|
|
477
|
+
result = evaluator.run({"response": "..."}) # Returns immediately
|
|
478
|
+
batch = result.wait() # Get results when needed
|
|
479
|
+
"""
|
|
480
|
+
return FrameworkEvaluator(
|
|
481
|
+
evaluations=list(evaluations),
|
|
482
|
+
mode=ExecutionMode.NON_BLOCKING,
|
|
483
|
+
max_workers=max_workers,
|
|
484
|
+
auto_enrich_span=auto_enrich_span,
|
|
485
|
+
backend=backend,
|
|
486
|
+
)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def distributed_evaluator(
|
|
490
|
+
*evaluations: BaseEvaluation,
|
|
491
|
+
backend: Backend,
|
|
492
|
+
auto_enrich_span: bool = True,
|
|
493
|
+
) -> FrameworkEvaluator:
|
|
494
|
+
"""
|
|
495
|
+
Create a distributed evaluator with custom backend.
|
|
496
|
+
|
|
497
|
+
Args:
|
|
498
|
+
*evaluations: Evaluations to run
|
|
499
|
+
backend: Execution backend (Temporal, Celery, Ray, etc.)
|
|
500
|
+
auto_enrich_span: Whether to enrich OTEL spans
|
|
501
|
+
|
|
502
|
+
Returns:
|
|
503
|
+
Configured Evaluator in distributed mode
|
|
504
|
+
|
|
505
|
+
Example:
|
|
506
|
+
backend = TemporalBackend(config)
|
|
507
|
+
evaluator = distributed_evaluator(
|
|
508
|
+
ToxicityEval(),
|
|
509
|
+
backend=backend,
|
|
510
|
+
)
|
|
511
|
+
result = evaluator.run({"response": "..."})
|
|
512
|
+
"""
|
|
513
|
+
return FrameworkEvaluator(
|
|
514
|
+
evaluations=list(evaluations),
|
|
515
|
+
mode=ExecutionMode.DISTRIBUTED,
|
|
516
|
+
backend=backend,
|
|
517
|
+
auto_enrich_span=auto_enrich_span,
|
|
518
|
+
)
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def _execute_single_evaluation(
|
|
522
|
+
evaluation: BaseEvaluation,
|
|
523
|
+
inputs: Dict[str, Any],
|
|
524
|
+
validate: bool = True,
|
|
525
|
+
) -> EvalResult:
|
|
526
|
+
"""
|
|
527
|
+
Execute a single evaluation, suitable for submission to any backend.
|
|
528
|
+
|
|
529
|
+
Handles validation, timing, and error capture.
|
|
530
|
+
|
|
531
|
+
Args:
|
|
532
|
+
evaluation: The evaluation to run
|
|
533
|
+
inputs: Input data
|
|
534
|
+
validate: Whether to validate inputs
|
|
535
|
+
|
|
536
|
+
Returns:
|
|
537
|
+
EvalResult with value or error
|
|
538
|
+
"""
|
|
539
|
+
eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
|
|
540
|
+
eval_version = getattr(evaluation, "version", "1.0.0")
|
|
541
|
+
|
|
542
|
+
# Validate inputs if enabled
|
|
543
|
+
if validate and hasattr(evaluation, "validate_inputs"):
|
|
544
|
+
try:
|
|
545
|
+
errors = evaluation.validate_inputs(inputs)
|
|
546
|
+
if errors:
|
|
547
|
+
error_msg = "; ".join(str(e) for e in errors) if isinstance(errors, list) else str(errors)
|
|
548
|
+
return EvalResult(
|
|
549
|
+
value=None,
|
|
550
|
+
eval_name=eval_name,
|
|
551
|
+
eval_version=eval_version,
|
|
552
|
+
latency_ms=0.0,
|
|
553
|
+
status=EvalStatus.FAILED,
|
|
554
|
+
error=f"Validation error: {error_msg}",
|
|
555
|
+
)
|
|
556
|
+
except Exception as e:
|
|
557
|
+
return EvalResult(
|
|
558
|
+
value=None,
|
|
559
|
+
eval_name=eval_name,
|
|
560
|
+
eval_version=eval_version,
|
|
561
|
+
latency_ms=0.0,
|
|
562
|
+
status=EvalStatus.FAILED,
|
|
563
|
+
error=f"Validation exception: {e}",
|
|
564
|
+
)
|
|
565
|
+
|
|
566
|
+
start = time.perf_counter()
|
|
567
|
+
try:
|
|
568
|
+
value = evaluation.evaluate(inputs)
|
|
569
|
+
latency_ms = (time.perf_counter() - start) * 1000
|
|
570
|
+
return EvalResult(
|
|
571
|
+
value=value,
|
|
572
|
+
eval_name=eval_name,
|
|
573
|
+
eval_version=eval_version,
|
|
574
|
+
latency_ms=latency_ms,
|
|
575
|
+
status=EvalStatus.COMPLETED,
|
|
576
|
+
)
|
|
577
|
+
except Exception as e:
|
|
578
|
+
latency_ms = (time.perf_counter() - start) * 1000
|
|
579
|
+
return EvalResult(
|
|
580
|
+
value=None,
|
|
581
|
+
eval_name=eval_name,
|
|
582
|
+
eval_version=eval_version,
|
|
583
|
+
latency_ms=latency_ms,
|
|
584
|
+
status=EvalStatus.FAILED,
|
|
585
|
+
error=str(e),
|
|
586
|
+
)
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
def resilient_evaluator(
|
|
590
|
+
*evaluations: BaseEvaluation,
|
|
591
|
+
backend: Optional[Backend] = None,
|
|
592
|
+
resilience: Optional[Any] = None,
|
|
593
|
+
fallback_backend: Optional[Backend] = None,
|
|
594
|
+
event_callback: Optional[Callable] = None,
|
|
595
|
+
auto_enrich_span: bool = True,
|
|
596
|
+
) -> FrameworkEvaluator:
|
|
597
|
+
"""
|
|
598
|
+
Create a distributed evaluator with resilience protections.
|
|
599
|
+
|
|
600
|
+
Wraps a backend with circuit breaker, rate limiting, retry, and fallback
|
|
601
|
+
capabilities, then creates an Evaluator in DISTRIBUTED mode.
|
|
602
|
+
|
|
603
|
+
Args:
|
|
604
|
+
*evaluations: Evaluations to run
|
|
605
|
+
backend: Execution backend (uses ThreadPoolBackend if None)
|
|
606
|
+
resilience: ResilienceConfig for resilience settings (optional)
|
|
607
|
+
fallback_backend: Optional fallback backend for degradation
|
|
608
|
+
event_callback: Callback for resilience events
|
|
609
|
+
auto_enrich_span: Whether to enrich OTEL spans
|
|
610
|
+
|
|
611
|
+
Returns:
|
|
612
|
+
Configured Evaluator in distributed mode with resilience
|
|
613
|
+
|
|
614
|
+
Example:
|
|
615
|
+
from fi.evals.framework import resilient_evaluator, ResilienceConfig
|
|
616
|
+
from fi.evals.framework import CircuitBreakerConfig, RateLimitConfig
|
|
617
|
+
|
|
618
|
+
evaluator = resilient_evaluator(
|
|
619
|
+
ToxicityEval(),
|
|
620
|
+
BiasEval(),
|
|
621
|
+
resilience=ResilienceConfig(
|
|
622
|
+
circuit_breaker=CircuitBreakerConfig(failure_threshold=5),
|
|
623
|
+
rate_limit=RateLimitConfig(requests_per_second=10),
|
|
624
|
+
),
|
|
625
|
+
)
|
|
626
|
+
result = evaluator.run({"response": "..."})
|
|
627
|
+
"""
|
|
628
|
+
from .resilience import ResilientBackend, ResilienceConfig
|
|
629
|
+
|
|
630
|
+
# Create underlying backend if not provided
|
|
631
|
+
underlying = backend or ThreadPoolBackend()
|
|
632
|
+
|
|
633
|
+
# Wrap with resilience
|
|
634
|
+
resilience_config = resilience or ResilienceConfig()
|
|
635
|
+
resilient = ResilientBackend(
|
|
636
|
+
underlying=underlying,
|
|
637
|
+
config=resilience_config,
|
|
638
|
+
fallback_backend=fallback_backend,
|
|
639
|
+
event_callback=event_callback,
|
|
640
|
+
)
|
|
641
|
+
|
|
642
|
+
return FrameworkEvaluator(
|
|
643
|
+
evaluations=list(evaluations),
|
|
644
|
+
mode=ExecutionMode.DISTRIBUTED,
|
|
645
|
+
backend=resilient,
|
|
646
|
+
auto_enrich_span=auto_enrich_span,
|
|
647
|
+
)
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Evaluator implementations for different execution modes."""
|
|
2
|
+
|
|
3
|
+
from .blocking import BlockingEvaluator, blocking_evaluate
|
|
4
|
+
from .non_blocking import (
|
|
5
|
+
NonBlockingEvaluator,
|
|
6
|
+
non_blocking_evaluate,
|
|
7
|
+
EvalFuture,
|
|
8
|
+
BatchEvalFuture,
|
|
9
|
+
EvalResultAggregator,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
# Blocking
|
|
14
|
+
"BlockingEvaluator",
|
|
15
|
+
"blocking_evaluate",
|
|
16
|
+
# Non-blocking
|
|
17
|
+
"NonBlockingEvaluator",
|
|
18
|
+
"non_blocking_evaluate",
|
|
19
|
+
"EvalFuture",
|
|
20
|
+
"BatchEvalFuture",
|
|
21
|
+
"EvalResultAggregator",
|
|
22
|
+
]
|