agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,625 @@
|
|
|
1
|
+
"""AutoEval Pipeline - Main API for automatic evaluation pipelines.
|
|
2
|
+
|
|
3
|
+
Provides the core AutoEvalPipeline class for creating and running
|
|
4
|
+
evaluation pipelines from natural language descriptions or templates.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
import time
|
|
9
|
+
from typing import Dict, Any, List, Optional, Type, Union
|
|
10
|
+
|
|
11
|
+
from .types import AutoEvalResult, AppAnalysis
|
|
12
|
+
from .config import AutoEvalConfig, EvalConfig, ScannerConfig
|
|
13
|
+
from .analyzer import AppAnalyzer
|
|
14
|
+
from .recommender import EvalRecommender
|
|
15
|
+
from .templates import get_template, list_templates
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# Registry for eval/scanner class lookups
|
|
21
|
+
_EVAL_CLASS_REGISTRY: Dict[str, Type] = {}
|
|
22
|
+
_SCANNER_CLASS_REGISTRY: Dict[str, Type] = {}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def register_eval_class(name: str, cls: Type) -> None:
|
|
26
|
+
"""Register an evaluation class for AutoEval lookup."""
|
|
27
|
+
_EVAL_CLASS_REGISTRY[name] = cls
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def register_scanner_class(name: str, cls: Type) -> None:
|
|
31
|
+
"""Register a scanner class for AutoEval lookup."""
|
|
32
|
+
_SCANNER_CLASS_REGISTRY[name] = cls
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _get_eval_class(name: str) -> Optional[Type]:
|
|
36
|
+
"""Get evaluation class by name."""
|
|
37
|
+
# Try direct registry lookup first
|
|
38
|
+
if name in _EVAL_CLASS_REGISTRY:
|
|
39
|
+
return _EVAL_CLASS_REGISTRY[name]
|
|
40
|
+
|
|
41
|
+
# Try to import from framework
|
|
42
|
+
try:
|
|
43
|
+
from fi.evals.framework import EvalRegistry
|
|
44
|
+
|
|
45
|
+
if EvalRegistry.is_registered(name):
|
|
46
|
+
return EvalRegistry.get(name)
|
|
47
|
+
except (ImportError, ValueError):
|
|
48
|
+
pass
|
|
49
|
+
|
|
50
|
+
# Try to import from evals module
|
|
51
|
+
try:
|
|
52
|
+
from fi.evals import templates as eval_templates
|
|
53
|
+
|
|
54
|
+
if hasattr(eval_templates, name):
|
|
55
|
+
return getattr(eval_templates, name)
|
|
56
|
+
except ImportError:
|
|
57
|
+
pass
|
|
58
|
+
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _get_scanner_class(name: str) -> Optional[Type]:
|
|
63
|
+
"""Get scanner class by name."""
|
|
64
|
+
# Try direct registry lookup first
|
|
65
|
+
if name in _SCANNER_CLASS_REGISTRY:
|
|
66
|
+
return _SCANNER_CLASS_REGISTRY[name]
|
|
67
|
+
|
|
68
|
+
# Try to import from guardrails.scanners
|
|
69
|
+
try:
|
|
70
|
+
from fi.evals.guardrails import scanners as scanner_module
|
|
71
|
+
|
|
72
|
+
if hasattr(scanner_module, name):
|
|
73
|
+
return getattr(scanner_module, name)
|
|
74
|
+
except ImportError:
|
|
75
|
+
pass
|
|
76
|
+
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class AutoEvalPipeline:
|
|
81
|
+
"""
|
|
82
|
+
Automatic evaluation pipeline builder.
|
|
83
|
+
|
|
84
|
+
Creates evaluation pipelines from natural language descriptions,
|
|
85
|
+
pre-built templates, or manual configuration.
|
|
86
|
+
|
|
87
|
+
Example:
|
|
88
|
+
# From natural language description
|
|
89
|
+
pipeline = AutoEvalPipeline.from_description(
|
|
90
|
+
"A RAG-based customer support chatbot for healthcare. "
|
|
91
|
+
"Retrieves patient records and answers questions about appointments."
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
# From pre-built template
|
|
95
|
+
pipeline = AutoEvalPipeline.from_template("rag_system")
|
|
96
|
+
|
|
97
|
+
# Run evaluation
|
|
98
|
+
result = pipeline.evaluate({
|
|
99
|
+
"query": "When is my appointment?",
|
|
100
|
+
"response": "Your appointment is Monday at 2pm.",
|
|
101
|
+
"context": "Patient has appointment on 2024-01-15 14:00",
|
|
102
|
+
})
|
|
103
|
+
|
|
104
|
+
print(result.passed) # True/False
|
|
105
|
+
print(result.explain()) # Detailed breakdown
|
|
106
|
+
|
|
107
|
+
# Export configuration
|
|
108
|
+
pipeline.export_yaml("eval_config.yaml")
|
|
109
|
+
"""
|
|
110
|
+
|
|
111
|
+
def __init__(
|
|
112
|
+
self,
|
|
113
|
+
config: AutoEvalConfig,
|
|
114
|
+
analysis: Optional[AppAnalysis] = None,
|
|
115
|
+
):
|
|
116
|
+
"""
|
|
117
|
+
Initialize the pipeline with configuration.
|
|
118
|
+
|
|
119
|
+
Args:
|
|
120
|
+
config: AutoEvalConfig with evaluations and scanners
|
|
121
|
+
analysis: Optional AppAnalysis for explanation context
|
|
122
|
+
"""
|
|
123
|
+
self.config = config
|
|
124
|
+
self.analysis = analysis
|
|
125
|
+
self._evaluator = None
|
|
126
|
+
self._scanner_pipeline = None
|
|
127
|
+
self._eval_instances: List[Any] = []
|
|
128
|
+
self._scanner_instances: List[Any] = []
|
|
129
|
+
|
|
130
|
+
@classmethod
|
|
131
|
+
def from_description(
|
|
132
|
+
cls,
|
|
133
|
+
description: str,
|
|
134
|
+
llm_provider: Optional[Any] = None,
|
|
135
|
+
name: Optional[str] = None,
|
|
136
|
+
) -> "AutoEvalPipeline":
|
|
137
|
+
"""
|
|
138
|
+
Create pipeline from natural language description.
|
|
139
|
+
|
|
140
|
+
Uses LLM-powered analysis when available, falls back to
|
|
141
|
+
rule-based analysis otherwise.
|
|
142
|
+
|
|
143
|
+
Args:
|
|
144
|
+
description: Natural language application description
|
|
145
|
+
llm_provider: Optional LLM provider for intelligent analysis
|
|
146
|
+
name: Optional name for the pipeline
|
|
147
|
+
|
|
148
|
+
Returns:
|
|
149
|
+
Configured AutoEvalPipeline
|
|
150
|
+
|
|
151
|
+
Example:
|
|
152
|
+
pipeline = AutoEvalPipeline.from_description(
|
|
153
|
+
"A customer support chatbot for a healthcare company. "
|
|
154
|
+
"It retrieves patient information and answers questions."
|
|
155
|
+
)
|
|
156
|
+
"""
|
|
157
|
+
# Analyze the description
|
|
158
|
+
analyzer = AppAnalyzer(llm_provider=llm_provider)
|
|
159
|
+
analysis = analyzer.analyze(description)
|
|
160
|
+
|
|
161
|
+
# Generate recommendations
|
|
162
|
+
recommender = EvalRecommender()
|
|
163
|
+
evals, scanners = recommender.recommend(analysis)
|
|
164
|
+
|
|
165
|
+
# Build config
|
|
166
|
+
config = AutoEvalConfig(
|
|
167
|
+
name=name or f"autoeval_{analysis.category.value}",
|
|
168
|
+
description=description[:200] if len(description) > 200 else description,
|
|
169
|
+
app_category=analysis.category.value,
|
|
170
|
+
risk_level=analysis.risk_level.value,
|
|
171
|
+
domain_sensitivity=analysis.domain_sensitivity.value,
|
|
172
|
+
evaluations=evals,
|
|
173
|
+
scanners=scanners,
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
return cls(config, analysis)
|
|
177
|
+
|
|
178
|
+
@classmethod
|
|
179
|
+
def from_template(cls, template_name: str) -> "AutoEvalPipeline":
|
|
180
|
+
"""
|
|
181
|
+
Create pipeline from pre-built template.
|
|
182
|
+
|
|
183
|
+
Available templates:
|
|
184
|
+
- customer_support: Customer service chatbots
|
|
185
|
+
- rag_system: RAG-based document Q&A
|
|
186
|
+
- code_assistant: Code generation and review
|
|
187
|
+
- content_moderation: Content filtering and safety
|
|
188
|
+
- agent_workflow: Autonomous agents with tool use
|
|
189
|
+
- healthcare: Healthcare applications (HIPAA)
|
|
190
|
+
- financial: Financial services
|
|
191
|
+
|
|
192
|
+
Args:
|
|
193
|
+
template_name: Name of the template to use
|
|
194
|
+
|
|
195
|
+
Returns:
|
|
196
|
+
Configured AutoEvalPipeline
|
|
197
|
+
|
|
198
|
+
Raises:
|
|
199
|
+
ValueError: If template not found
|
|
200
|
+
|
|
201
|
+
Example:
|
|
202
|
+
pipeline = AutoEvalPipeline.from_template("rag_system")
|
|
203
|
+
"""
|
|
204
|
+
config = get_template(template_name)
|
|
205
|
+
if config is None:
|
|
206
|
+
available = list(list_templates().keys())
|
|
207
|
+
raise ValueError(
|
|
208
|
+
f"Template '{template_name}' not found. "
|
|
209
|
+
f"Available templates: {available}"
|
|
210
|
+
)
|
|
211
|
+
return cls(config)
|
|
212
|
+
|
|
213
|
+
@classmethod
|
|
214
|
+
def from_config(cls, config: AutoEvalConfig) -> "AutoEvalPipeline":
|
|
215
|
+
"""
|
|
216
|
+
Create pipeline from existing configuration.
|
|
217
|
+
|
|
218
|
+
Args:
|
|
219
|
+
config: AutoEvalConfig instance
|
|
220
|
+
|
|
221
|
+
Returns:
|
|
222
|
+
Configured AutoEvalPipeline
|
|
223
|
+
"""
|
|
224
|
+
return cls(config)
|
|
225
|
+
|
|
226
|
+
@classmethod
|
|
227
|
+
def from_yaml(cls, path: str) -> "AutoEvalPipeline":
|
|
228
|
+
"""
|
|
229
|
+
Load pipeline from YAML file.
|
|
230
|
+
|
|
231
|
+
Args:
|
|
232
|
+
path: Path to YAML configuration file
|
|
233
|
+
|
|
234
|
+
Returns:
|
|
235
|
+
Configured AutoEvalPipeline
|
|
236
|
+
"""
|
|
237
|
+
from .export import load_config
|
|
238
|
+
|
|
239
|
+
config = load_config(path)
|
|
240
|
+
return cls(config)
|
|
241
|
+
|
|
242
|
+
def _is_core_metric(self, name: str) -> bool:
|
|
243
|
+
"""Check if a name corresponds to a core metric in the local registry."""
|
|
244
|
+
try:
|
|
245
|
+
from fi.evals.local.registry import get_registry
|
|
246
|
+
return get_registry().is_registered(name)
|
|
247
|
+
except ImportError:
|
|
248
|
+
return False
|
|
249
|
+
|
|
250
|
+
def _get_metric_configs(self) -> List[EvalConfig]:
|
|
251
|
+
"""Get eval configs that are core metrics (routed through evaluate())."""
|
|
252
|
+
return [
|
|
253
|
+
e for e in self.config.evaluations
|
|
254
|
+
if e.enabled and self._is_core_metric(e.name)
|
|
255
|
+
]
|
|
256
|
+
|
|
257
|
+
def _get_class_configs(self) -> List[EvalConfig]:
|
|
258
|
+
"""Get eval configs that are framework classes (routed through Evaluator)."""
|
|
259
|
+
return [
|
|
260
|
+
e for e in self.config.evaluations
|
|
261
|
+
if e.enabled and not self._is_core_metric(e.name)
|
|
262
|
+
]
|
|
263
|
+
|
|
264
|
+
def _run_core_metrics(
|
|
265
|
+
self,
|
|
266
|
+
inputs: Dict[str, Any],
|
|
267
|
+
feedback_store: Optional[Any] = None,
|
|
268
|
+
) -> List[Any]:
|
|
269
|
+
"""Run core metrics via evaluate() API and return EvalResults."""
|
|
270
|
+
from fi.evals import evaluate as core_evaluate
|
|
271
|
+
|
|
272
|
+
metric_configs = self._get_metric_configs()
|
|
273
|
+
if not metric_configs:
|
|
274
|
+
return []
|
|
275
|
+
|
|
276
|
+
results = []
|
|
277
|
+
for eval_config in metric_configs:
|
|
278
|
+
try:
|
|
279
|
+
result = core_evaluate(
|
|
280
|
+
eval_config.name,
|
|
281
|
+
model=eval_config.model,
|
|
282
|
+
augment=eval_config.augment,
|
|
283
|
+
feedback_store=feedback_store,
|
|
284
|
+
**inputs,
|
|
285
|
+
)
|
|
286
|
+
results.append(result)
|
|
287
|
+
except Exception as e:
|
|
288
|
+
logger.warning(f"Core metric '{eval_config.name}' failed: {e}")
|
|
289
|
+
return results
|
|
290
|
+
|
|
291
|
+
def _build_evaluator(self) -> None:
|
|
292
|
+
"""Build the evaluator with framework-class evaluations only."""
|
|
293
|
+
if self._evaluator is not None:
|
|
294
|
+
return
|
|
295
|
+
|
|
296
|
+
from fi.evals.framework import Evaluator, ExecutionMode
|
|
297
|
+
|
|
298
|
+
# Only build for non-metric (framework class) evals
|
|
299
|
+
self._eval_instances = []
|
|
300
|
+
for eval_config in self._get_class_configs():
|
|
301
|
+
eval_class = _get_eval_class(eval_config.name)
|
|
302
|
+
if eval_class is None:
|
|
303
|
+
logger.warning(f"Evaluation class not found: {eval_config.name}")
|
|
304
|
+
continue
|
|
305
|
+
|
|
306
|
+
try:
|
|
307
|
+
instance = eval_class(**eval_config.params)
|
|
308
|
+
self._eval_instances.append(instance)
|
|
309
|
+
except Exception as e:
|
|
310
|
+
logger.warning(f"Failed to instantiate {eval_config.name}: {e}")
|
|
311
|
+
|
|
312
|
+
if self._eval_instances:
|
|
313
|
+
mode = (
|
|
314
|
+
ExecutionMode.NON_BLOCKING
|
|
315
|
+
if self.config.execution_mode == "non_blocking"
|
|
316
|
+
else ExecutionMode.BLOCKING
|
|
317
|
+
)
|
|
318
|
+
self._evaluator = Evaluator(
|
|
319
|
+
evaluations=self._eval_instances,
|
|
320
|
+
mode=mode,
|
|
321
|
+
max_workers=self.config.parallel_workers,
|
|
322
|
+
fail_fast=self.config.fail_fast,
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
def _build_scanner_pipeline(self) -> None:
|
|
326
|
+
"""Build the scanner pipeline with configured scanners."""
|
|
327
|
+
if self._scanner_pipeline is not None:
|
|
328
|
+
return
|
|
329
|
+
|
|
330
|
+
from fi.evals.guardrails.scanners import ScannerPipeline
|
|
331
|
+
|
|
332
|
+
# Instantiate scanner classes
|
|
333
|
+
self._scanner_instances = []
|
|
334
|
+
for scanner_config in self.config.scanners:
|
|
335
|
+
if not scanner_config.enabled:
|
|
336
|
+
continue
|
|
337
|
+
|
|
338
|
+
scanner_class = _get_scanner_class(scanner_config.name)
|
|
339
|
+
if scanner_class is None:
|
|
340
|
+
logger.warning(f"Scanner class not found: {scanner_config.name}")
|
|
341
|
+
continue
|
|
342
|
+
|
|
343
|
+
try:
|
|
344
|
+
instance = scanner_class(**scanner_config.params)
|
|
345
|
+
# Set threshold and action if the scanner supports them
|
|
346
|
+
if hasattr(instance, "threshold"):
|
|
347
|
+
instance.threshold = scanner_config.threshold
|
|
348
|
+
if hasattr(instance, "action"):
|
|
349
|
+
from fi.evals.guardrails.scanners.base import ScannerAction
|
|
350
|
+
|
|
351
|
+
instance.action = ScannerAction(scanner_config.action)
|
|
352
|
+
self._scanner_instances.append(instance)
|
|
353
|
+
except Exception as e:
|
|
354
|
+
logger.warning(f"Failed to instantiate {scanner_config.name}: {e}")
|
|
355
|
+
|
|
356
|
+
if self._scanner_instances:
|
|
357
|
+
self._scanner_pipeline = ScannerPipeline(
|
|
358
|
+
scanners=self._scanner_instances,
|
|
359
|
+
parallel=True,
|
|
360
|
+
fail_fast=self.config.fail_fast,
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
def evaluate(
|
|
364
|
+
self,
|
|
365
|
+
inputs: Dict[str, Any],
|
|
366
|
+
scan_content: Optional[str] = None,
|
|
367
|
+
feedback_store: Optional[Any] = None,
|
|
368
|
+
) -> AutoEvalResult:
|
|
369
|
+
"""
|
|
370
|
+
Run the full evaluation pipeline.
|
|
371
|
+
|
|
372
|
+
Executes scanners first (fast), then evaluations.
|
|
373
|
+
If scanners block, evaluations may be skipped.
|
|
374
|
+
|
|
375
|
+
Args:
|
|
376
|
+
inputs: Input data for evaluations (e.g., query, response, context)
|
|
377
|
+
scan_content: Content to scan (defaults to response from inputs)
|
|
378
|
+
|
|
379
|
+
Returns:
|
|
380
|
+
AutoEvalResult with combined results
|
|
381
|
+
|
|
382
|
+
Example:
|
|
383
|
+
result = pipeline.evaluate({
|
|
384
|
+
"query": "What is the patient's blood type?",
|
|
385
|
+
"response": "The patient's blood type is O+.",
|
|
386
|
+
"context": "Medical record: Blood type O+",
|
|
387
|
+
})
|
|
388
|
+
|
|
389
|
+
if result.passed:
|
|
390
|
+
print("Evaluation passed!")
|
|
391
|
+
else:
|
|
392
|
+
print(f"Failed: {result.explain()}")
|
|
393
|
+
"""
|
|
394
|
+
start_time = time.perf_counter()
|
|
395
|
+
|
|
396
|
+
# Build components lazily
|
|
397
|
+
self._build_evaluator()
|
|
398
|
+
self._build_scanner_pipeline()
|
|
399
|
+
|
|
400
|
+
scan_result = None
|
|
401
|
+
eval_result = None
|
|
402
|
+
metric_results = []
|
|
403
|
+
blocked_by_scanner = False
|
|
404
|
+
|
|
405
|
+
# Run scanners first (fast)
|
|
406
|
+
if self._scanner_pipeline:
|
|
407
|
+
content = scan_content or inputs.get("response", "")
|
|
408
|
+
context = inputs.get("context")
|
|
409
|
+
scan_result = self._scanner_pipeline.scan(content, context)
|
|
410
|
+
blocked_by_scanner = not scan_result.passed
|
|
411
|
+
|
|
412
|
+
# Run evaluations if not blocked
|
|
413
|
+
if not blocked_by_scanner:
|
|
414
|
+
# Core metrics via evaluate() API
|
|
415
|
+
metric_results = self._run_core_metrics(inputs, feedback_store=feedback_store)
|
|
416
|
+
|
|
417
|
+
# Framework class evals via Evaluator
|
|
418
|
+
if self._evaluator:
|
|
419
|
+
eval_result = self._evaluator.run(inputs)
|
|
420
|
+
|
|
421
|
+
# Determine overall pass/fail
|
|
422
|
+
passed = True
|
|
423
|
+
if scan_result and not scan_result.passed:
|
|
424
|
+
passed = False
|
|
425
|
+
|
|
426
|
+
# Check core metric results against thresholds
|
|
427
|
+
if metric_results:
|
|
428
|
+
metric_configs = {e.name: e for e in self._get_metric_configs()}
|
|
429
|
+
failed_count = 0
|
|
430
|
+
total_count = len(metric_results)
|
|
431
|
+
for r in metric_results:
|
|
432
|
+
config = metric_configs.get(getattr(r, "eval_name", ""))
|
|
433
|
+
threshold = config.threshold if config else 0.5
|
|
434
|
+
score = getattr(r, "score", None)
|
|
435
|
+
if score is not None and score < threshold:
|
|
436
|
+
failed_count += 1
|
|
437
|
+
if total_count > 0:
|
|
438
|
+
success_rate = (total_count - failed_count) / total_count
|
|
439
|
+
if success_rate < self.config.global_pass_rate:
|
|
440
|
+
passed = False
|
|
441
|
+
|
|
442
|
+
if eval_result:
|
|
443
|
+
# Check if framework evaluations meet threshold
|
|
444
|
+
batch = eval_result.wait() if eval_result.is_future else eval_result.batch
|
|
445
|
+
if batch and batch.success_rate < self.config.global_pass_rate:
|
|
446
|
+
passed = False
|
|
447
|
+
|
|
448
|
+
total_latency = (time.perf_counter() - start_time) * 1000
|
|
449
|
+
|
|
450
|
+
return AutoEvalResult(
|
|
451
|
+
passed=passed,
|
|
452
|
+
scan_result=scan_result,
|
|
453
|
+
eval_result=eval_result,
|
|
454
|
+
metric_results=metric_results,
|
|
455
|
+
blocked_by_scanner=blocked_by_scanner,
|
|
456
|
+
total_latency_ms=total_latency,
|
|
457
|
+
)
|
|
458
|
+
|
|
459
|
+
def add(self, item: Union[EvalConfig, ScannerConfig]) -> "AutoEvalPipeline":
|
|
460
|
+
"""
|
|
461
|
+
Add an evaluation or scanner to the pipeline.
|
|
462
|
+
|
|
463
|
+
Args:
|
|
464
|
+
item: EvalConfig or ScannerConfig to add
|
|
465
|
+
|
|
466
|
+
Returns:
|
|
467
|
+
Self for chaining
|
|
468
|
+
|
|
469
|
+
Example:
|
|
470
|
+
pipeline.add(EvalConfig("CustomEval", threshold=0.8))
|
|
471
|
+
pipeline.add(ScannerConfig("CustomScanner", action="flag"))
|
|
472
|
+
"""
|
|
473
|
+
if isinstance(item, EvalConfig):
|
|
474
|
+
self.config.evaluations.append(item)
|
|
475
|
+
self._evaluator = None # Reset to rebuild
|
|
476
|
+
elif isinstance(item, ScannerConfig):
|
|
477
|
+
self.config.scanners.append(item)
|
|
478
|
+
self._scanner_pipeline = None # Reset to rebuild
|
|
479
|
+
return self
|
|
480
|
+
|
|
481
|
+
def remove(self, name: str) -> "AutoEvalPipeline":
|
|
482
|
+
"""
|
|
483
|
+
Remove an evaluation or scanner by name.
|
|
484
|
+
|
|
485
|
+
Args:
|
|
486
|
+
name: Name of the evaluation or scanner to remove
|
|
487
|
+
|
|
488
|
+
Returns:
|
|
489
|
+
Self for chaining
|
|
490
|
+
|
|
491
|
+
Example:
|
|
492
|
+
pipeline.remove("CoherenceEval")
|
|
493
|
+
"""
|
|
494
|
+
# Try to remove from evaluations
|
|
495
|
+
original_count = len(self.config.evaluations)
|
|
496
|
+
self.config.evaluations = [
|
|
497
|
+
e for e in self.config.evaluations if e.name != name
|
|
498
|
+
]
|
|
499
|
+
if len(self.config.evaluations) < original_count:
|
|
500
|
+
self._evaluator = None
|
|
501
|
+
|
|
502
|
+
# Try to remove from scanners
|
|
503
|
+
original_count = len(self.config.scanners)
|
|
504
|
+
self.config.scanners = [s for s in self.config.scanners if s.name != name]
|
|
505
|
+
if len(self.config.scanners) < original_count:
|
|
506
|
+
self._scanner_pipeline = None
|
|
507
|
+
|
|
508
|
+
return self
|
|
509
|
+
|
|
510
|
+
def set_threshold(self, name: str, threshold: float) -> "AutoEvalPipeline":
|
|
511
|
+
"""
|
|
512
|
+
Set threshold for an evaluation or scanner.
|
|
513
|
+
|
|
514
|
+
Args:
|
|
515
|
+
name: Name of the evaluation or scanner
|
|
516
|
+
threshold: New threshold value (0.0-1.0)
|
|
517
|
+
|
|
518
|
+
Returns:
|
|
519
|
+
Self for chaining
|
|
520
|
+
|
|
521
|
+
Example:
|
|
522
|
+
pipeline.set_threshold("CoherenceEval", 0.9)
|
|
523
|
+
"""
|
|
524
|
+
for e in self.config.evaluations:
|
|
525
|
+
if e.name == name:
|
|
526
|
+
e.threshold = threshold
|
|
527
|
+
self._evaluator = None
|
|
528
|
+
|
|
529
|
+
for s in self.config.scanners:
|
|
530
|
+
if s.name == name:
|
|
531
|
+
s.threshold = threshold
|
|
532
|
+
self._scanner_pipeline = None
|
|
533
|
+
|
|
534
|
+
return self
|
|
535
|
+
|
|
536
|
+
def enable(self, name: str) -> "AutoEvalPipeline":
|
|
537
|
+
"""Enable an evaluation or scanner."""
|
|
538
|
+
for e in self.config.evaluations:
|
|
539
|
+
if e.name == name:
|
|
540
|
+
e.enabled = True
|
|
541
|
+
self._evaluator = None
|
|
542
|
+
|
|
543
|
+
for s in self.config.scanners:
|
|
544
|
+
if s.name == name:
|
|
545
|
+
s.enabled = True
|
|
546
|
+
self._scanner_pipeline = None
|
|
547
|
+
|
|
548
|
+
return self
|
|
549
|
+
|
|
550
|
+
def disable(self, name: str) -> "AutoEvalPipeline":
|
|
551
|
+
"""Disable an evaluation or scanner."""
|
|
552
|
+
for e in self.config.evaluations:
|
|
553
|
+
if e.name == name:
|
|
554
|
+
e.enabled = False
|
|
555
|
+
self._evaluator = None
|
|
556
|
+
|
|
557
|
+
for s in self.config.scanners:
|
|
558
|
+
if s.name == name:
|
|
559
|
+
s.enabled = False
|
|
560
|
+
self._scanner_pipeline = None
|
|
561
|
+
|
|
562
|
+
return self
|
|
563
|
+
|
|
564
|
+
def export_yaml(self, path: str) -> None:
|
|
565
|
+
"""
|
|
566
|
+
Export pipeline configuration to YAML file.
|
|
567
|
+
|
|
568
|
+
Args:
|
|
569
|
+
path: Path to write YAML file
|
|
570
|
+
|
|
571
|
+
Example:
|
|
572
|
+
pipeline.export_yaml("eval_config.yaml")
|
|
573
|
+
"""
|
|
574
|
+
from .export import export_yaml
|
|
575
|
+
|
|
576
|
+
export_yaml(self.config, path)
|
|
577
|
+
|
|
578
|
+
def export_json(self, path: str) -> None:
|
|
579
|
+
"""
|
|
580
|
+
Export pipeline configuration to JSON file.
|
|
581
|
+
|
|
582
|
+
Args:
|
|
583
|
+
path: Path to write JSON file
|
|
584
|
+
"""
|
|
585
|
+
from .export import export_json
|
|
586
|
+
|
|
587
|
+
export_json(self.config, path)
|
|
588
|
+
|
|
589
|
+
def explain(self) -> str:
|
|
590
|
+
"""
|
|
591
|
+
Get human-readable explanation of the pipeline.
|
|
592
|
+
|
|
593
|
+
Returns:
|
|
594
|
+
Detailed explanation string
|
|
595
|
+
|
|
596
|
+
Example:
|
|
597
|
+
print(pipeline.explain())
|
|
598
|
+
"""
|
|
599
|
+
lines = [self.config.summary()]
|
|
600
|
+
|
|
601
|
+
if self.analysis:
|
|
602
|
+
lines.append("")
|
|
603
|
+
lines.append("Analysis Details:")
|
|
604
|
+
lines.append(f" Confidence: {self.analysis.confidence:.0%}")
|
|
605
|
+
lines.append(f" Explanation: {self.analysis.explanation}")
|
|
606
|
+
if self.analysis.detected_features:
|
|
607
|
+
lines.append(f" Detected Features: {', '.join(self.analysis.detected_features)}")
|
|
608
|
+
|
|
609
|
+
return "\n".join(lines)
|
|
610
|
+
|
|
611
|
+
def summary(self) -> str:
|
|
612
|
+
"""Get brief summary of the pipeline."""
|
|
613
|
+
return (
|
|
614
|
+
f"AutoEvalPipeline: {self.config.name}\n"
|
|
615
|
+
f" Evaluations: {len(self.config.evaluations)}\n"
|
|
616
|
+
f" Scanners: {len(self.config.scanners)}\n"
|
|
617
|
+
f" Risk Level: {self.config.risk_level}"
|
|
618
|
+
)
|
|
619
|
+
|
|
620
|
+
def __repr__(self) -> str:
|
|
621
|
+
return (
|
|
622
|
+
f"AutoEvalPipeline(name={self.config.name!r}, "
|
|
623
|
+
f"evaluations={len(self.config.evaluations)}, "
|
|
624
|
+
f"scanners={len(self.config.scanners)})"
|
|
625
|
+
)
|