agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,690 @@
|
|
|
1
|
+
"""Local evaluator for running metrics without API calls.
|
|
2
|
+
|
|
3
|
+
This module provides the LocalEvaluator class which can run heuristic
|
|
4
|
+
metrics locally, enabling offline evaluation and faster feedback loops.
|
|
5
|
+
|
|
6
|
+
It also provides the HybridEvaluator which can intelligently route
|
|
7
|
+
evaluations between local execution, local LLM, and cloud APIs.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Dict, List, Optional, TYPE_CHECKING
|
|
12
|
+
import time
|
|
13
|
+
import logging
|
|
14
|
+
|
|
15
|
+
from ..types import BatchRunResult, EvalResult
|
|
16
|
+
from .execution_mode import RoutingMode, can_run_locally
|
|
17
|
+
from .registry import get_registry, LocalMetricRegistry
|
|
18
|
+
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
from .llm import OllamaLLM
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class LocalEvaluatorConfig:
|
|
27
|
+
"""Configuration for the local evaluator.
|
|
28
|
+
|
|
29
|
+
Attributes:
|
|
30
|
+
execution_mode: The default execution mode.
|
|
31
|
+
fail_on_unsupported: If True, raise an error when a metric can't run locally.
|
|
32
|
+
parallel_workers: Number of parallel workers (for future use).
|
|
33
|
+
timeout: Timeout in seconds for individual evaluations.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
execution_mode: RoutingMode = RoutingMode.HYBRID
|
|
37
|
+
fail_on_unsupported: bool = False
|
|
38
|
+
parallel_workers: int = 4
|
|
39
|
+
timeout: int = 60
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class LocalEvaluationResult:
|
|
44
|
+
"""Result of a local evaluation operation.
|
|
45
|
+
|
|
46
|
+
Attributes:
|
|
47
|
+
results: The batch run results.
|
|
48
|
+
executed_locally: Set of metric names that ran locally.
|
|
49
|
+
skipped: Set of metric names that were skipped (not local-capable).
|
|
50
|
+
errors: Dictionary mapping metric names to error messages.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
results: BatchRunResult
|
|
54
|
+
executed_locally: set = field(default_factory=set)
|
|
55
|
+
skipped: set = field(default_factory=set)
|
|
56
|
+
errors: Dict[str, str] = field(default_factory=dict)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class LocalEvaluator:
|
|
60
|
+
"""Evaluator that runs metrics locally without API calls.
|
|
61
|
+
|
|
62
|
+
This evaluator can run heuristic metrics locally, providing fast feedback
|
|
63
|
+
without requiring network access or API credentials.
|
|
64
|
+
|
|
65
|
+
Example:
|
|
66
|
+
>>> evaluator = LocalEvaluator()
|
|
67
|
+
>>> result = evaluator.evaluate(
|
|
68
|
+
... metric_name="contains",
|
|
69
|
+
... inputs=[{"response": "Hello world", "keyword": "world"}],
|
|
70
|
+
... config={"keyword": "world"}
|
|
71
|
+
... )
|
|
72
|
+
>>> print(result.results.eval_results[0].output)
|
|
73
|
+
1.0
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
def __init__(
|
|
77
|
+
self,
|
|
78
|
+
config: Optional[LocalEvaluatorConfig] = None,
|
|
79
|
+
registry: Optional[LocalMetricRegistry] = None,
|
|
80
|
+
) -> None:
|
|
81
|
+
"""Initialize the local evaluator.
|
|
82
|
+
|
|
83
|
+
Args:
|
|
84
|
+
config: Configuration for the evaluator.
|
|
85
|
+
registry: Optional metric registry (uses global if not provided).
|
|
86
|
+
"""
|
|
87
|
+
self.config = config or LocalEvaluatorConfig()
|
|
88
|
+
self.registry = registry or get_registry()
|
|
89
|
+
|
|
90
|
+
def can_run_locally(self, metric_name: str) -> bool:
|
|
91
|
+
"""Check if a metric can be run locally.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
metric_name: The name of the metric.
|
|
95
|
+
|
|
96
|
+
Returns:
|
|
97
|
+
True if the metric can run locally.
|
|
98
|
+
"""
|
|
99
|
+
return can_run_locally(metric_name) and self.registry.is_registered(metric_name)
|
|
100
|
+
|
|
101
|
+
def evaluate(
|
|
102
|
+
self,
|
|
103
|
+
metric_name: str,
|
|
104
|
+
inputs: List[Dict[str, Any]],
|
|
105
|
+
config: Optional[Dict[str, Any]] = None,
|
|
106
|
+
) -> LocalEvaluationResult:
|
|
107
|
+
"""Evaluate a single metric on a batch of inputs.
|
|
108
|
+
|
|
109
|
+
Args:
|
|
110
|
+
metric_name: The name of the metric to run.
|
|
111
|
+
inputs: List of input dictionaries for the metric.
|
|
112
|
+
config: Optional configuration for the metric.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
LocalEvaluationResult with results and metadata.
|
|
116
|
+
|
|
117
|
+
Raises:
|
|
118
|
+
ValueError: If fail_on_unsupported is True and metric can't run locally.
|
|
119
|
+
"""
|
|
120
|
+
result = LocalEvaluationResult(results=BatchRunResult(eval_results=[]))
|
|
121
|
+
|
|
122
|
+
if not self.can_run_locally(metric_name):
|
|
123
|
+
if self.config.fail_on_unsupported:
|
|
124
|
+
raise ValueError(
|
|
125
|
+
f"Metric '{metric_name}' cannot run locally. "
|
|
126
|
+
f"Available local metrics: {self.registry.list_metrics()}"
|
|
127
|
+
)
|
|
128
|
+
result.skipped.add(metric_name)
|
|
129
|
+
# Return empty results for skipped metrics
|
|
130
|
+
for _ in inputs:
|
|
131
|
+
result.results.eval_results.append(
|
|
132
|
+
EvalResult(
|
|
133
|
+
name=metric_name,
|
|
134
|
+
output=None,
|
|
135
|
+
reason=f"Metric '{metric_name}' cannot run locally",
|
|
136
|
+
runtime=0,
|
|
137
|
+
)
|
|
138
|
+
)
|
|
139
|
+
return result
|
|
140
|
+
|
|
141
|
+
try:
|
|
142
|
+
metric = self.registry.create(metric_name, config)
|
|
143
|
+
if metric is None:
|
|
144
|
+
raise ValueError(f"Failed to create metric '{metric_name}'")
|
|
145
|
+
|
|
146
|
+
batch_result = metric.evaluate(inputs)
|
|
147
|
+
result.results = batch_result
|
|
148
|
+
result.executed_locally.add(metric_name)
|
|
149
|
+
|
|
150
|
+
except Exception as e:
|
|
151
|
+
result.errors[metric_name] = str(e)
|
|
152
|
+
# Fill with error results
|
|
153
|
+
for _ in inputs:
|
|
154
|
+
result.results.eval_results.append(
|
|
155
|
+
EvalResult(
|
|
156
|
+
name=metric_name,
|
|
157
|
+
output=None,
|
|
158
|
+
reason=f"Error: {str(e)}",
|
|
159
|
+
runtime=0,
|
|
160
|
+
)
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
# Automatically enrich current span with evaluation results
|
|
164
|
+
try:
|
|
165
|
+
from fi.evals.otel.enrichment import enrich_span_with_batch_result, is_auto_enrichment_enabled
|
|
166
|
+
if is_auto_enrichment_enabled():
|
|
167
|
+
enriched_count = enrich_span_with_batch_result(result.results)
|
|
168
|
+
if enriched_count > 0:
|
|
169
|
+
import logging
|
|
170
|
+
logging.debug(f"Enriched active span with {enriched_count} evaluation results")
|
|
171
|
+
except ImportError:
|
|
172
|
+
pass # OTEL enrichment not available
|
|
173
|
+
except Exception:
|
|
174
|
+
pass # Silently fail enrichment
|
|
175
|
+
|
|
176
|
+
return result
|
|
177
|
+
|
|
178
|
+
def evaluate_batch(
|
|
179
|
+
self,
|
|
180
|
+
evaluations: List[Dict[str, Any]],
|
|
181
|
+
) -> LocalEvaluationResult:
|
|
182
|
+
"""Evaluate multiple metrics on their respective inputs.
|
|
183
|
+
|
|
184
|
+
Args:
|
|
185
|
+
evaluations: List of evaluation specifications, each containing:
|
|
186
|
+
- metric_name: Name of the metric
|
|
187
|
+
- inputs: List of input dictionaries
|
|
188
|
+
- config: Optional metric configuration
|
|
189
|
+
|
|
190
|
+
Returns:
|
|
191
|
+
LocalEvaluationResult with combined results.
|
|
192
|
+
|
|
193
|
+
Example:
|
|
194
|
+
>>> evaluator = LocalEvaluator()
|
|
195
|
+
>>> result = evaluator.evaluate_batch([
|
|
196
|
+
... {
|
|
197
|
+
... "metric_name": "contains",
|
|
198
|
+
... "inputs": [{"response": "hello world"}],
|
|
199
|
+
... "config": {"keyword": "world"}
|
|
200
|
+
... },
|
|
201
|
+
... {
|
|
202
|
+
... "metric_name": "is_json",
|
|
203
|
+
... "inputs": [{"response": '{"key": "value"}'}]
|
|
204
|
+
... }
|
|
205
|
+
... ])
|
|
206
|
+
"""
|
|
207
|
+
combined_result = LocalEvaluationResult(
|
|
208
|
+
results=BatchRunResult(eval_results=[])
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
for eval_spec in evaluations:
|
|
212
|
+
metric_name = eval_spec.get("metric_name")
|
|
213
|
+
inputs = eval_spec.get("inputs", [])
|
|
214
|
+
config = eval_spec.get("config")
|
|
215
|
+
|
|
216
|
+
if not metric_name:
|
|
217
|
+
continue
|
|
218
|
+
|
|
219
|
+
single_result = self.evaluate(metric_name, inputs, config)
|
|
220
|
+
|
|
221
|
+
# Merge results
|
|
222
|
+
combined_result.results.eval_results.extend(
|
|
223
|
+
single_result.results.eval_results
|
|
224
|
+
)
|
|
225
|
+
combined_result.executed_locally.update(single_result.executed_locally)
|
|
226
|
+
combined_result.skipped.update(single_result.skipped)
|
|
227
|
+
combined_result.errors.update(single_result.errors)
|
|
228
|
+
|
|
229
|
+
return combined_result
|
|
230
|
+
|
|
231
|
+
def list_available_metrics(self) -> List[str]:
|
|
232
|
+
"""List all metrics available for local execution.
|
|
233
|
+
|
|
234
|
+
Returns:
|
|
235
|
+
Sorted list of available metric names.
|
|
236
|
+
"""
|
|
237
|
+
return self.registry.list_metrics()
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class HybridEvaluator:
|
|
241
|
+
"""Evaluator that routes metrics between local and cloud execution.
|
|
242
|
+
|
|
243
|
+
This evaluator analyzes each metric and automatically routes it to
|
|
244
|
+
either local heuristics, local LLM, or cloud execution based on the
|
|
245
|
+
metric type and configuration.
|
|
246
|
+
|
|
247
|
+
Example:
|
|
248
|
+
>>> # Basic usage with automatic routing
|
|
249
|
+
>>> evaluator = HybridEvaluator()
|
|
250
|
+
>>> partitions = evaluator.partition_evaluations([
|
|
251
|
+
... {"metric_name": "contains", "inputs": [{"response": "test"}]},
|
|
252
|
+
... {"metric_name": "groundedness", "inputs": [{"response": "test"}]},
|
|
253
|
+
... ])
|
|
254
|
+
>>> local_results = evaluator.evaluate_local_partition(partitions[RoutingMode.LOCAL])
|
|
255
|
+
|
|
256
|
+
>>> # With local LLM for LLM-based evaluations
|
|
257
|
+
>>> from fi.evals.local.llm import OllamaLLM
|
|
258
|
+
>>> evaluator = HybridEvaluator(local_llm=OllamaLLM())
|
|
259
|
+
>>> result = evaluator.evaluate(
|
|
260
|
+
... template="custom_llm_judge",
|
|
261
|
+
... inputs=[{"query": "What is AI?", "response": "AI is..."}]
|
|
262
|
+
... )
|
|
263
|
+
"""
|
|
264
|
+
|
|
265
|
+
# LLM-based metrics that can use local LLM instead of cloud
|
|
266
|
+
LLM_BASED_METRICS = {
|
|
267
|
+
"groundedness",
|
|
268
|
+
"hallucination",
|
|
269
|
+
"relevance",
|
|
270
|
+
"coherence",
|
|
271
|
+
"context_relevance",
|
|
272
|
+
"answer_relevance",
|
|
273
|
+
"custom_llm_judge",
|
|
274
|
+
"tone",
|
|
275
|
+
"safety",
|
|
276
|
+
"pii",
|
|
277
|
+
"bias",
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
def __init__(
|
|
281
|
+
self,
|
|
282
|
+
config: Optional[LocalEvaluatorConfig] = None,
|
|
283
|
+
local_evaluator: Optional[LocalEvaluator] = None,
|
|
284
|
+
local_llm: Optional["OllamaLLM"] = None,
|
|
285
|
+
cloud_evaluator: Optional[Any] = None,
|
|
286
|
+
prefer_local: bool = True,
|
|
287
|
+
fallback_to_cloud: bool = True,
|
|
288
|
+
offline_mode: bool = False,
|
|
289
|
+
) -> None:
|
|
290
|
+
"""Initialize the hybrid evaluator.
|
|
291
|
+
|
|
292
|
+
Args:
|
|
293
|
+
config: Configuration for the evaluator.
|
|
294
|
+
local_evaluator: Optional local evaluator instance for heuristic metrics.
|
|
295
|
+
local_llm: Optional local LLM instance for LLM-based metrics.
|
|
296
|
+
cloud_evaluator: Optional cloud evaluator instance (Evaluator).
|
|
297
|
+
prefer_local: If True, prefer local execution when possible.
|
|
298
|
+
fallback_to_cloud: If True, fall back to cloud when local fails.
|
|
299
|
+
offline_mode: If True, never use cloud (raise error if metric requires it).
|
|
300
|
+
"""
|
|
301
|
+
self.config = config or LocalEvaluatorConfig(execution_mode=RoutingMode.HYBRID)
|
|
302
|
+
self.local_evaluator = local_evaluator or LocalEvaluator(self.config)
|
|
303
|
+
self.local_llm = local_llm
|
|
304
|
+
self.cloud_evaluator = cloud_evaluator
|
|
305
|
+
self.prefer_local = prefer_local
|
|
306
|
+
self.fallback_to_cloud = fallback_to_cloud
|
|
307
|
+
self.offline_mode = offline_mode
|
|
308
|
+
|
|
309
|
+
def set_local_llm(self, llm: "OllamaLLM") -> None:
|
|
310
|
+
"""Set the local LLM instance.
|
|
311
|
+
|
|
312
|
+
Args:
|
|
313
|
+
llm: OllamaLLM instance to use for LLM-based evaluations.
|
|
314
|
+
"""
|
|
315
|
+
self.local_llm = llm
|
|
316
|
+
|
|
317
|
+
def set_cloud_evaluator(self, evaluator: Any) -> None:
|
|
318
|
+
"""Set the cloud evaluator instance.
|
|
319
|
+
|
|
320
|
+
Args:
|
|
321
|
+
evaluator: Cloud Evaluator instance for API-based evaluations.
|
|
322
|
+
"""
|
|
323
|
+
self.cloud_evaluator = evaluator
|
|
324
|
+
|
|
325
|
+
def can_use_local_llm(self, metric_name: str) -> bool:
|
|
326
|
+
"""Check if a metric can use the local LLM.
|
|
327
|
+
|
|
328
|
+
Args:
|
|
329
|
+
metric_name: The name of the metric.
|
|
330
|
+
|
|
331
|
+
Returns:
|
|
332
|
+
True if the metric can use local LLM.
|
|
333
|
+
"""
|
|
334
|
+
if self.local_llm is None:
|
|
335
|
+
return False
|
|
336
|
+
if not self.local_llm.is_available():
|
|
337
|
+
return False
|
|
338
|
+
return metric_name.lower() in self.LLM_BASED_METRICS
|
|
339
|
+
|
|
340
|
+
def route_evaluation(
|
|
341
|
+
self,
|
|
342
|
+
metric_name: str,
|
|
343
|
+
force_local: bool = False,
|
|
344
|
+
force_cloud: bool = False,
|
|
345
|
+
) -> RoutingMode:
|
|
346
|
+
"""Determine the execution mode for a metric.
|
|
347
|
+
|
|
348
|
+
Args:
|
|
349
|
+
metric_name: The name of the metric.
|
|
350
|
+
force_local: Force local execution.
|
|
351
|
+
force_cloud: Force cloud execution.
|
|
352
|
+
|
|
353
|
+
Returns:
|
|
354
|
+
The recommended execution mode.
|
|
355
|
+
"""
|
|
356
|
+
if force_cloud and not self.offline_mode:
|
|
357
|
+
return RoutingMode.CLOUD
|
|
358
|
+
if force_local:
|
|
359
|
+
return RoutingMode.LOCAL
|
|
360
|
+
|
|
361
|
+
# Check if it's a heuristic metric that can run locally
|
|
362
|
+
if can_run_locally(metric_name):
|
|
363
|
+
return RoutingMode.LOCAL
|
|
364
|
+
|
|
365
|
+
# Check if it's an LLM metric and we have local LLM
|
|
366
|
+
if self.can_use_local_llm(metric_name) and self.prefer_local:
|
|
367
|
+
return RoutingMode.LOCAL
|
|
368
|
+
|
|
369
|
+
# Default to cloud unless in offline mode
|
|
370
|
+
if self.offline_mode:
|
|
371
|
+
raise ValueError(
|
|
372
|
+
f"Metric '{metric_name}' requires cloud execution but offline_mode is enabled"
|
|
373
|
+
)
|
|
374
|
+
return RoutingMode.CLOUD
|
|
375
|
+
|
|
376
|
+
def partition_evaluations(
|
|
377
|
+
self, evaluations: List[Dict[str, Any]]
|
|
378
|
+
) -> Dict[RoutingMode, List[Dict[str, Any]]]:
|
|
379
|
+
"""Partition evaluations by execution mode.
|
|
380
|
+
|
|
381
|
+
Args:
|
|
382
|
+
evaluations: List of evaluation specifications.
|
|
383
|
+
|
|
384
|
+
Returns:
|
|
385
|
+
Dictionary mapping execution modes to their evaluations.
|
|
386
|
+
"""
|
|
387
|
+
partitions: Dict[RoutingMode, List[Dict[str, Any]]] = {
|
|
388
|
+
RoutingMode.LOCAL: [],
|
|
389
|
+
RoutingMode.CLOUD: [],
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
for eval_spec in evaluations:
|
|
393
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
394
|
+
force_local = eval_spec.get("force_local", False)
|
|
395
|
+
force_cloud = eval_spec.get("force_cloud", False)
|
|
396
|
+
|
|
397
|
+
try:
|
|
398
|
+
mode = self.route_evaluation(metric_name, force_local, force_cloud)
|
|
399
|
+
partitions[mode].append(eval_spec)
|
|
400
|
+
except ValueError as e:
|
|
401
|
+
# Offline mode violation - add to errors
|
|
402
|
+
logger.error(f"Routing error for {metric_name}: {e}")
|
|
403
|
+
partitions[RoutingMode.CLOUD].append(eval_spec)
|
|
404
|
+
|
|
405
|
+
return partitions
|
|
406
|
+
|
|
407
|
+
def evaluate_local_partition(
|
|
408
|
+
self, evaluations: List[Dict[str, Any]]
|
|
409
|
+
) -> LocalEvaluationResult:
|
|
410
|
+
"""Evaluate the local partition of evaluations.
|
|
411
|
+
|
|
412
|
+
This handles both heuristic metrics (via LocalEvaluator) and
|
|
413
|
+
LLM-based metrics (via local LLM).
|
|
414
|
+
|
|
415
|
+
Args:
|
|
416
|
+
evaluations: List of evaluation specifications to run locally.
|
|
417
|
+
|
|
418
|
+
Returns:
|
|
419
|
+
LocalEvaluationResult with results.
|
|
420
|
+
"""
|
|
421
|
+
heuristic_evals = []
|
|
422
|
+
llm_evals = []
|
|
423
|
+
|
|
424
|
+
# Separate heuristic and LLM evaluations
|
|
425
|
+
for eval_spec in evaluations:
|
|
426
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
427
|
+
if can_run_locally(metric_name):
|
|
428
|
+
heuristic_evals.append(eval_spec)
|
|
429
|
+
elif self.can_use_local_llm(metric_name):
|
|
430
|
+
llm_evals.append(eval_spec)
|
|
431
|
+
else:
|
|
432
|
+
heuristic_evals.append(eval_spec) # Will be skipped
|
|
433
|
+
|
|
434
|
+
# Run heuristic evaluations
|
|
435
|
+
result = self.local_evaluator.evaluate_batch(heuristic_evals)
|
|
436
|
+
|
|
437
|
+
# Run LLM evaluations if we have a local LLM
|
|
438
|
+
if llm_evals and self.local_llm:
|
|
439
|
+
llm_results = self._evaluate_with_local_llm(llm_evals)
|
|
440
|
+
result.results.eval_results.extend(llm_results.results.eval_results)
|
|
441
|
+
result.executed_locally.update(llm_results.executed_locally)
|
|
442
|
+
result.skipped.update(llm_results.skipped)
|
|
443
|
+
result.errors.update(llm_results.errors)
|
|
444
|
+
|
|
445
|
+
return result
|
|
446
|
+
|
|
447
|
+
def _evaluate_with_local_llm(
|
|
448
|
+
self, evaluations: List[Dict[str, Any]]
|
|
449
|
+
) -> LocalEvaluationResult:
|
|
450
|
+
"""Run evaluations using the local LLM.
|
|
451
|
+
|
|
452
|
+
Args:
|
|
453
|
+
evaluations: List of LLM-based evaluation specifications.
|
|
454
|
+
|
|
455
|
+
Returns:
|
|
456
|
+
LocalEvaluationResult with LLM evaluation results.
|
|
457
|
+
"""
|
|
458
|
+
result = LocalEvaluationResult(results=BatchRunResult(eval_results=[]))
|
|
459
|
+
|
|
460
|
+
if not self.local_llm:
|
|
461
|
+
for eval_spec in evaluations:
|
|
462
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
463
|
+
result.skipped.add(metric_name)
|
|
464
|
+
for _ in eval_spec.get("inputs", []):
|
|
465
|
+
result.results.eval_results.append(
|
|
466
|
+
EvalResult(
|
|
467
|
+
name=metric_name,
|
|
468
|
+
output=None,
|
|
469
|
+
reason="No local LLM configured",
|
|
470
|
+
runtime=0,
|
|
471
|
+
)
|
|
472
|
+
)
|
|
473
|
+
return result
|
|
474
|
+
|
|
475
|
+
for eval_spec in evaluations:
|
|
476
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
477
|
+
inputs = eval_spec.get("inputs", [])
|
|
478
|
+
config = eval_spec.get("config", {})
|
|
479
|
+
|
|
480
|
+
for input_data in inputs:
|
|
481
|
+
start_time = time.time()
|
|
482
|
+
try:
|
|
483
|
+
# Build evaluation from input
|
|
484
|
+
judge_result = self.local_llm.judge(
|
|
485
|
+
query=input_data.get("input", input_data.get("query", "")),
|
|
486
|
+
response=input_data.get("response", input_data.get("output", "")),
|
|
487
|
+
criteria=config.get("criteria", f"Evaluate based on {metric_name}"),
|
|
488
|
+
context=input_data.get("context", input_data.get("contexts", "")),
|
|
489
|
+
)
|
|
490
|
+
|
|
491
|
+
runtime = int((time.time() - start_time) * 1000)
|
|
492
|
+
result.results.eval_results.append(
|
|
493
|
+
EvalResult(
|
|
494
|
+
name=metric_name,
|
|
495
|
+
output=judge_result.get("score", 0.0),
|
|
496
|
+
reason=judge_result.get("reason", ""),
|
|
497
|
+
runtime=runtime,
|
|
498
|
+
metrics=[{
|
|
499
|
+
"name": metric_name,
|
|
500
|
+
"value": judge_result.get("score", 0.0),
|
|
501
|
+
}],
|
|
502
|
+
)
|
|
503
|
+
)
|
|
504
|
+
result.executed_locally.add(metric_name)
|
|
505
|
+
|
|
506
|
+
except Exception as e:
|
|
507
|
+
runtime = int((time.time() - start_time) * 1000)
|
|
508
|
+
result.errors[metric_name] = str(e)
|
|
509
|
+
result.results.eval_results.append(
|
|
510
|
+
EvalResult(
|
|
511
|
+
name=metric_name,
|
|
512
|
+
output=0.0,
|
|
513
|
+
reason=f"Local LLM error: {str(e)}",
|
|
514
|
+
runtime=runtime,
|
|
515
|
+
)
|
|
516
|
+
)
|
|
517
|
+
|
|
518
|
+
return result
|
|
519
|
+
|
|
520
|
+
def evaluate(
|
|
521
|
+
self,
|
|
522
|
+
template: str,
|
|
523
|
+
inputs: List[Dict[str, Any]],
|
|
524
|
+
config: Optional[Dict[str, Any]] = None,
|
|
525
|
+
) -> LocalEvaluationResult:
|
|
526
|
+
"""Evaluate a single template with automatic routing.
|
|
527
|
+
|
|
528
|
+
Args:
|
|
529
|
+
template: The evaluation template/metric name.
|
|
530
|
+
inputs: List of input dictionaries.
|
|
531
|
+
config: Optional configuration for the evaluation.
|
|
532
|
+
|
|
533
|
+
Returns:
|
|
534
|
+
LocalEvaluationResult with evaluation results.
|
|
535
|
+
"""
|
|
536
|
+
eval_spec = {
|
|
537
|
+
"metric_name": template,
|
|
538
|
+
"inputs": inputs,
|
|
539
|
+
"config": config or {},
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
mode = self.route_evaluation(template)
|
|
543
|
+
|
|
544
|
+
if mode == RoutingMode.LOCAL:
|
|
545
|
+
result = self.evaluate_local_partition([eval_spec])
|
|
546
|
+
else:
|
|
547
|
+
result = self._evaluate_cloud([eval_spec])
|
|
548
|
+
|
|
549
|
+
# Automatically enrich current span with evaluation results
|
|
550
|
+
try:
|
|
551
|
+
from fi.evals.otel.enrichment import enrich_span_with_batch_result, is_auto_enrichment_enabled
|
|
552
|
+
if is_auto_enrichment_enabled():
|
|
553
|
+
enrich_span_with_batch_result(result.results)
|
|
554
|
+
except ImportError:
|
|
555
|
+
pass
|
|
556
|
+
except Exception:
|
|
557
|
+
pass
|
|
558
|
+
|
|
559
|
+
return result
|
|
560
|
+
|
|
561
|
+
def evaluate_batch(
|
|
562
|
+
self,
|
|
563
|
+
evaluations: List[Dict[str, Any]],
|
|
564
|
+
) -> LocalEvaluationResult:
|
|
565
|
+
"""Evaluate multiple templates with automatic routing.
|
|
566
|
+
|
|
567
|
+
Args:
|
|
568
|
+
evaluations: List of evaluation specifications.
|
|
569
|
+
|
|
570
|
+
Returns:
|
|
571
|
+
Combined LocalEvaluationResult from all evaluations.
|
|
572
|
+
"""
|
|
573
|
+
partitions = self.partition_evaluations(evaluations)
|
|
574
|
+
|
|
575
|
+
# Run local evaluations
|
|
576
|
+
local_result = self.evaluate_local_partition(partitions[RoutingMode.LOCAL])
|
|
577
|
+
|
|
578
|
+
# Run cloud evaluations
|
|
579
|
+
if partitions[RoutingMode.CLOUD]:
|
|
580
|
+
cloud_result = self._evaluate_cloud(partitions[RoutingMode.CLOUD])
|
|
581
|
+
|
|
582
|
+
# Merge results
|
|
583
|
+
local_result.results.eval_results.extend(cloud_result.results.eval_results)
|
|
584
|
+
local_result.executed_locally.update(cloud_result.executed_locally)
|
|
585
|
+
local_result.skipped.update(cloud_result.skipped)
|
|
586
|
+
local_result.errors.update(cloud_result.errors)
|
|
587
|
+
|
|
588
|
+
return local_result
|
|
589
|
+
|
|
590
|
+
def _evaluate_cloud(
|
|
591
|
+
self, evaluations: List[Dict[str, Any]]
|
|
592
|
+
) -> LocalEvaluationResult:
|
|
593
|
+
"""Run evaluations via cloud API.
|
|
594
|
+
|
|
595
|
+
Args:
|
|
596
|
+
evaluations: List of evaluation specifications for cloud.
|
|
597
|
+
|
|
598
|
+
Returns:
|
|
599
|
+
LocalEvaluationResult with cloud evaluation results.
|
|
600
|
+
"""
|
|
601
|
+
result = LocalEvaluationResult(results=BatchRunResult(eval_results=[]))
|
|
602
|
+
|
|
603
|
+
if self.offline_mode:
|
|
604
|
+
for eval_spec in evaluations:
|
|
605
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
606
|
+
result.skipped.add(metric_name)
|
|
607
|
+
for _ in eval_spec.get("inputs", []):
|
|
608
|
+
result.results.eval_results.append(
|
|
609
|
+
EvalResult(
|
|
610
|
+
name=metric_name,
|
|
611
|
+
output=None,
|
|
612
|
+
reason="Offline mode - cloud execution disabled",
|
|
613
|
+
runtime=0,
|
|
614
|
+
)
|
|
615
|
+
)
|
|
616
|
+
return result
|
|
617
|
+
|
|
618
|
+
if not self.cloud_evaluator:
|
|
619
|
+
for eval_spec in evaluations:
|
|
620
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
621
|
+
result.skipped.add(metric_name)
|
|
622
|
+
for _ in eval_spec.get("inputs", []):
|
|
623
|
+
result.results.eval_results.append(
|
|
624
|
+
EvalResult(
|
|
625
|
+
name=metric_name,
|
|
626
|
+
output=None,
|
|
627
|
+
reason="No cloud evaluator configured",
|
|
628
|
+
runtime=0,
|
|
629
|
+
)
|
|
630
|
+
)
|
|
631
|
+
return result
|
|
632
|
+
|
|
633
|
+
# Route through fi.evals.evaluate() with Turing engine
|
|
634
|
+
try:
|
|
635
|
+
from fi.evals import evaluate as core_evaluate
|
|
636
|
+
|
|
637
|
+
for eval_spec in evaluations:
|
|
638
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
639
|
+
inputs_list = eval_spec.get("inputs", [])
|
|
640
|
+
|
|
641
|
+
for input_data in inputs_list:
|
|
642
|
+
start_time = time.time()
|
|
643
|
+
try:
|
|
644
|
+
eval_result = core_evaluate(
|
|
645
|
+
metric_name,
|
|
646
|
+
engine="turing",
|
|
647
|
+
output=input_data.get("response", input_data.get("output", "")),
|
|
648
|
+
input=input_data.get("input", input_data.get("query", "")),
|
|
649
|
+
context=input_data.get("context", input_data.get("contexts", "")),
|
|
650
|
+
)
|
|
651
|
+
|
|
652
|
+
runtime = int((time.time() - start_time) * 1000)
|
|
653
|
+
result.results.eval_results.append(
|
|
654
|
+
EvalResult(
|
|
655
|
+
name=metric_name,
|
|
656
|
+
output=eval_result.score if hasattr(eval_result, "score") else eval_result.output,
|
|
657
|
+
reason=getattr(eval_result, "reason", ""),
|
|
658
|
+
runtime=runtime,
|
|
659
|
+
metrics=[{
|
|
660
|
+
"name": metric_name,
|
|
661
|
+
"value": eval_result.score if hasattr(eval_result, "score") else 0.0,
|
|
662
|
+
}],
|
|
663
|
+
)
|
|
664
|
+
)
|
|
665
|
+
result.executed_locally.add(metric_name)
|
|
666
|
+
|
|
667
|
+
except Exception as e:
|
|
668
|
+
runtime = int((time.time() - start_time) * 1000)
|
|
669
|
+
result.errors[metric_name] = str(e)
|
|
670
|
+
result.results.eval_results.append(
|
|
671
|
+
EvalResult(
|
|
672
|
+
name=metric_name,
|
|
673
|
+
output=None,
|
|
674
|
+
reason=f"Cloud evaluation error: {e}",
|
|
675
|
+
runtime=runtime,
|
|
676
|
+
)
|
|
677
|
+
)
|
|
678
|
+
|
|
679
|
+
except ImportError:
|
|
680
|
+
logger.error("fi.evals.evaluate not available for cloud routing")
|
|
681
|
+
for eval_spec in evaluations:
|
|
682
|
+
metric_name = eval_spec.get("metric_name", "")
|
|
683
|
+
result.skipped.add(metric_name)
|
|
684
|
+
result.errors[metric_name] = "fi.evals.evaluate not available"
|
|
685
|
+
except Exception as e:
|
|
686
|
+
logger.error(f"Cloud evaluation failed: {e}")
|
|
687
|
+
for eval_spec in evaluations:
|
|
688
|
+
result.errors[eval_spec.get("metric_name", "")] = str(e)
|
|
689
|
+
|
|
690
|
+
return result
|