agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/cli/commands/run.py
ADDED
|
@@ -0,0 +1,486 @@
|
|
|
1
|
+
"""Run command for executing evaluations."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Optional, Dict, Any
|
|
7
|
+
from enum import Enum
|
|
8
|
+
|
|
9
|
+
import typer
|
|
10
|
+
from rich.progress import Progress, SpinnerColumn, TextColumn
|
|
11
|
+
|
|
12
|
+
from fi.cli.config.loader import load_config, load_test_data
|
|
13
|
+
from fi.cli.output.formatters import format_results
|
|
14
|
+
from fi.cli.output.reporters import ResultReporter
|
|
15
|
+
from fi.cli.utils.console import console, print_error, print_success, print_warning
|
|
16
|
+
from fi.cli.assertions import (
|
|
17
|
+
AssertionEvaluator,
|
|
18
|
+
AssertionReporter,
|
|
19
|
+
ExitCode,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ExecutionModeOption(str, Enum):
|
|
24
|
+
"""Execution mode options for CLI."""
|
|
25
|
+
local = "local"
|
|
26
|
+
cloud = "cloud"
|
|
27
|
+
hybrid = "hybrid"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def run(
|
|
31
|
+
config: Optional[Path] = typer.Option(
|
|
32
|
+
None,
|
|
33
|
+
"--config", "-c",
|
|
34
|
+
help="Path to configuration file",
|
|
35
|
+
),
|
|
36
|
+
eval_template: Optional[str] = typer.Option(
|
|
37
|
+
None,
|
|
38
|
+
"--eval", "-e",
|
|
39
|
+
help="Evaluation template to run (overrides config)",
|
|
40
|
+
),
|
|
41
|
+
data: Optional[Path] = typer.Option(
|
|
42
|
+
None,
|
|
43
|
+
"--data", "-d",
|
|
44
|
+
help="Path to test data file (overrides config)",
|
|
45
|
+
),
|
|
46
|
+
output: str = typer.Option(
|
|
47
|
+
"table",
|
|
48
|
+
"--output", "-o",
|
|
49
|
+
help="Output format: table, json, csv, html",
|
|
50
|
+
),
|
|
51
|
+
parallel: int = typer.Option(
|
|
52
|
+
8,
|
|
53
|
+
"--parallel", "-p",
|
|
54
|
+
help="Number of parallel workers",
|
|
55
|
+
),
|
|
56
|
+
timeout: int = typer.Option(
|
|
57
|
+
200,
|
|
58
|
+
"--timeout", "-T",
|
|
59
|
+
help="Timeout per evaluation in seconds",
|
|
60
|
+
),
|
|
61
|
+
model: Optional[str] = typer.Option(
|
|
62
|
+
None,
|
|
63
|
+
"--model", "-m",
|
|
64
|
+
help="Model for LLM-as-judge evaluations",
|
|
65
|
+
),
|
|
66
|
+
dry_run: bool = typer.Option(
|
|
67
|
+
False,
|
|
68
|
+
"--dry-run",
|
|
69
|
+
help="Validate config without running evaluations",
|
|
70
|
+
),
|
|
71
|
+
output_file: Optional[Path] = typer.Option(
|
|
72
|
+
None,
|
|
73
|
+
"--output-file", "-O",
|
|
74
|
+
help="Path to save output file",
|
|
75
|
+
),
|
|
76
|
+
quiet: bool = typer.Option(
|
|
77
|
+
False,
|
|
78
|
+
"--quiet", "-q",
|
|
79
|
+
help="Suppress progress output",
|
|
80
|
+
),
|
|
81
|
+
no_save: bool = typer.Option(
|
|
82
|
+
False,
|
|
83
|
+
"--no-save",
|
|
84
|
+
help="Don't save run to history (for 'fi view' command)",
|
|
85
|
+
),
|
|
86
|
+
check_assertions: bool = typer.Option(
|
|
87
|
+
True,
|
|
88
|
+
"--check/--no-check",
|
|
89
|
+
help="Check assertions after evaluation (if configured)",
|
|
90
|
+
),
|
|
91
|
+
fail_fast: bool = typer.Option(
|
|
92
|
+
False,
|
|
93
|
+
"--fail-fast",
|
|
94
|
+
help="Stop on first assertion failure",
|
|
95
|
+
),
|
|
96
|
+
strict: bool = typer.Option(
|
|
97
|
+
False,
|
|
98
|
+
"--strict",
|
|
99
|
+
help="Exit with error code on assertion warnings",
|
|
100
|
+
),
|
|
101
|
+
mode: ExecutionModeOption = typer.Option(
|
|
102
|
+
ExecutionModeOption.cloud,
|
|
103
|
+
"--mode",
|
|
104
|
+
help="Execution mode: local (no API), cloud (API only), or hybrid (auto-route)",
|
|
105
|
+
),
|
|
106
|
+
local_llm: Optional[str] = typer.Option(
|
|
107
|
+
None,
|
|
108
|
+
"--local-llm",
|
|
109
|
+
help="Local LLM for LLM-based evals (e.g., 'ollama/llama3.2')",
|
|
110
|
+
),
|
|
111
|
+
offline: bool = typer.Option(
|
|
112
|
+
False,
|
|
113
|
+
"--offline",
|
|
114
|
+
help="Run in offline mode (no cloud API calls)",
|
|
115
|
+
),
|
|
116
|
+
) -> None:
|
|
117
|
+
"""
|
|
118
|
+
Run evaluations from config file or CLI arguments.
|
|
119
|
+
|
|
120
|
+
Examples:
|
|
121
|
+
fi run # Use default config (cloud mode)
|
|
122
|
+
fi run -c custom.yaml # Use custom config
|
|
123
|
+
fi run -e groundedness -d data.json # Run single evaluation
|
|
124
|
+
fi run -o json > results.json # Output as JSON
|
|
125
|
+
fi run --mode local # Run only local heuristic metrics
|
|
126
|
+
fi run --mode hybrid # Auto-route between local and cloud
|
|
127
|
+
fi run --local-llm ollama/llama3.2 # Use local LLM for LLM-based evals
|
|
128
|
+
fi run --offline # No cloud API calls (implies local mode)
|
|
129
|
+
"""
|
|
130
|
+
from fi.evals.evaluator import Evaluator
|
|
131
|
+
from fi.evals.local import HybridEvaluator
|
|
132
|
+
|
|
133
|
+
# Handle offline mode implications
|
|
134
|
+
effective_mode = mode
|
|
135
|
+
if offline:
|
|
136
|
+
if mode == ExecutionModeOption.cloud:
|
|
137
|
+
effective_mode = ExecutionModeOption.local
|
|
138
|
+
print_warning("Offline mode enabled - switching from cloud to local mode")
|
|
139
|
+
|
|
140
|
+
# Check for API keys (only required for cloud/hybrid modes)
|
|
141
|
+
api_key = os.environ.get("FI_API_KEY")
|
|
142
|
+
secret_key = os.environ.get("FI_SECRET_KEY")
|
|
143
|
+
|
|
144
|
+
if effective_mode != ExecutionModeOption.local and (not api_key or not secret_key):
|
|
145
|
+
if effective_mode == ExecutionModeOption.cloud:
|
|
146
|
+
print_warning(
|
|
147
|
+
"API keys not found in environment.\n"
|
|
148
|
+
"Set FI_API_KEY and FI_SECRET_KEY environment variables."
|
|
149
|
+
)
|
|
150
|
+
if not dry_run:
|
|
151
|
+
raise typer.Exit(1)
|
|
152
|
+
elif effective_mode == ExecutionModeOption.hybrid:
|
|
153
|
+
print_warning(
|
|
154
|
+
"API keys not found - hybrid mode will only run local metrics."
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
# Load configuration or use CLI arguments
|
|
158
|
+
if eval_template and data:
|
|
159
|
+
# CLI-only mode
|
|
160
|
+
test_data = load_test_data(data)
|
|
161
|
+
evaluations = [{"template": eval_template, "data": test_data}]
|
|
162
|
+
defaults = {"timeout": timeout, "parallel_workers": parallel}
|
|
163
|
+
else:
|
|
164
|
+
# Config file mode
|
|
165
|
+
try:
|
|
166
|
+
eval_config = load_config(config)
|
|
167
|
+
except FileNotFoundError as e:
|
|
168
|
+
print_error(str(e))
|
|
169
|
+
raise typer.Exit(1)
|
|
170
|
+
except ValueError as e:
|
|
171
|
+
print_error(f"Configuration error: {e}")
|
|
172
|
+
raise typer.Exit(1)
|
|
173
|
+
|
|
174
|
+
defaults = {
|
|
175
|
+
"timeout": eval_config.get_defaults().timeout,
|
|
176
|
+
"parallel_workers": eval_config.get_defaults().parallel_workers,
|
|
177
|
+
"model": eval_config.get_defaults().model,
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
# Override with CLI arguments
|
|
181
|
+
if timeout != 200:
|
|
182
|
+
defaults["timeout"] = timeout
|
|
183
|
+
if parallel != 8:
|
|
184
|
+
defaults["parallel_workers"] = parallel
|
|
185
|
+
if model:
|
|
186
|
+
defaults["model"] = model
|
|
187
|
+
|
|
188
|
+
# Build evaluations list
|
|
189
|
+
evaluations = []
|
|
190
|
+
for eval_def in eval_config.evaluations:
|
|
191
|
+
try:
|
|
192
|
+
test_data = load_test_data(eval_def.data)
|
|
193
|
+
except FileNotFoundError:
|
|
194
|
+
print_error(f"Data file not found: {eval_def.data}")
|
|
195
|
+
raise typer.Exit(1)
|
|
196
|
+
|
|
197
|
+
templates = eval_def.templates or [eval_def.template]
|
|
198
|
+
for template in templates:
|
|
199
|
+
if template:
|
|
200
|
+
evaluations.append({
|
|
201
|
+
"name": eval_def.name,
|
|
202
|
+
"template": template,
|
|
203
|
+
"data": test_data,
|
|
204
|
+
"config": eval_def.config,
|
|
205
|
+
})
|
|
206
|
+
|
|
207
|
+
if dry_run:
|
|
208
|
+
print_success(
|
|
209
|
+
f"Configuration valid!\n"
|
|
210
|
+
f"Would run {len(evaluations)} evaluation(s) in {effective_mode.value} mode."
|
|
211
|
+
)
|
|
212
|
+
return
|
|
213
|
+
|
|
214
|
+
# Initialize local LLM if specified
|
|
215
|
+
llm_instance = None
|
|
216
|
+
if local_llm:
|
|
217
|
+
try:
|
|
218
|
+
from fi.evals.local import LocalLLMFactory
|
|
219
|
+
llm_instance = LocalLLMFactory.from_string(local_llm)
|
|
220
|
+
if not llm_instance.is_available():
|
|
221
|
+
print_warning(
|
|
222
|
+
f"Local LLM '{local_llm}' is not available. "
|
|
223
|
+
"Make sure Ollama is running: `ollama serve`"
|
|
224
|
+
)
|
|
225
|
+
llm_instance = None
|
|
226
|
+
elif not quiet:
|
|
227
|
+
print_success(f"Local LLM initialized: {local_llm}")
|
|
228
|
+
except Exception as e:
|
|
229
|
+
print_warning(f"Failed to initialize local LLM: {e}")
|
|
230
|
+
llm_instance = None
|
|
231
|
+
|
|
232
|
+
# Initialize evaluator(s) based on mode
|
|
233
|
+
cloud_evaluator = None
|
|
234
|
+
hybrid_evaluator = None
|
|
235
|
+
|
|
236
|
+
if effective_mode == ExecutionModeOption.cloud:
|
|
237
|
+
# Pure cloud mode - use standard evaluator
|
|
238
|
+
cloud_evaluator = Evaluator(
|
|
239
|
+
fi_api_key=api_key,
|
|
240
|
+
fi_secret_key=secret_key,
|
|
241
|
+
max_workers=defaults["parallel_workers"],
|
|
242
|
+
)
|
|
243
|
+
elif effective_mode == ExecutionModeOption.local:
|
|
244
|
+
# Pure local mode - use hybrid evaluator in offline mode
|
|
245
|
+
hybrid_evaluator = HybridEvaluator(
|
|
246
|
+
local_llm=llm_instance,
|
|
247
|
+
prefer_local=True,
|
|
248
|
+
fallback_to_cloud=False,
|
|
249
|
+
offline_mode=True,
|
|
250
|
+
)
|
|
251
|
+
if not quiet:
|
|
252
|
+
print_success("Running in local mode (no cloud API calls)")
|
|
253
|
+
else:
|
|
254
|
+
# Hybrid mode - use hybrid evaluator with cloud fallback
|
|
255
|
+
if api_key and secret_key:
|
|
256
|
+
cloud_evaluator = Evaluator(
|
|
257
|
+
fi_api_key=api_key,
|
|
258
|
+
fi_secret_key=secret_key,
|
|
259
|
+
max_workers=defaults["parallel_workers"],
|
|
260
|
+
)
|
|
261
|
+
hybrid_evaluator = HybridEvaluator(
|
|
262
|
+
local_llm=llm_instance,
|
|
263
|
+
cloud_evaluator=cloud_evaluator,
|
|
264
|
+
prefer_local=True,
|
|
265
|
+
fallback_to_cloud=not offline,
|
|
266
|
+
offline_mode=offline,
|
|
267
|
+
)
|
|
268
|
+
if not quiet:
|
|
269
|
+
print_success("Running in hybrid mode (auto-routing local/cloud)")
|
|
270
|
+
|
|
271
|
+
all_results = []
|
|
272
|
+
local_metrics_run = 0
|
|
273
|
+
cloud_metrics_run = 0
|
|
274
|
+
|
|
275
|
+
# Helper function to run single evaluation
|
|
276
|
+
def run_evaluation(eval_def: Dict[str, Any]) -> None:
|
|
277
|
+
nonlocal local_metrics_run, cloud_metrics_run
|
|
278
|
+
|
|
279
|
+
template = eval_def["template"]
|
|
280
|
+
data = eval_def["data"]
|
|
281
|
+
|
|
282
|
+
if effective_mode == ExecutionModeOption.cloud:
|
|
283
|
+
# Pure cloud mode
|
|
284
|
+
results = cloud_evaluator.evaluate(
|
|
285
|
+
eval_templates=template,
|
|
286
|
+
inputs=data,
|
|
287
|
+
timeout=defaults["timeout"],
|
|
288
|
+
model_name=defaults.get("model"),
|
|
289
|
+
)
|
|
290
|
+
all_results.extend(results.eval_results)
|
|
291
|
+
cloud_metrics_run += 1
|
|
292
|
+
|
|
293
|
+
elif effective_mode == ExecutionModeOption.local:
|
|
294
|
+
# Pure local mode
|
|
295
|
+
result = hybrid_evaluator.evaluate(
|
|
296
|
+
template=template,
|
|
297
|
+
inputs=data,
|
|
298
|
+
config=eval_def.get("config", {}),
|
|
299
|
+
)
|
|
300
|
+
all_results.extend(result.results.eval_results)
|
|
301
|
+
local_metrics_run += len(result.executed_locally)
|
|
302
|
+
|
|
303
|
+
else:
|
|
304
|
+
# Hybrid mode - route based on metric type
|
|
305
|
+
from fi.evals.local import can_run_locally
|
|
306
|
+
|
|
307
|
+
if can_run_locally(template) or hybrid_evaluator.can_use_local_llm(template):
|
|
308
|
+
# Run locally
|
|
309
|
+
result = hybrid_evaluator.evaluate(
|
|
310
|
+
template=template,
|
|
311
|
+
inputs=data,
|
|
312
|
+
config=eval_def.get("config", {}),
|
|
313
|
+
)
|
|
314
|
+
all_results.extend(result.results.eval_results)
|
|
315
|
+
local_metrics_run += len(result.executed_locally)
|
|
316
|
+
elif cloud_evaluator:
|
|
317
|
+
# Run in cloud
|
|
318
|
+
results = cloud_evaluator.evaluate(
|
|
319
|
+
eval_templates=template,
|
|
320
|
+
inputs=data,
|
|
321
|
+
timeout=defaults["timeout"],
|
|
322
|
+
model_name=defaults.get("model"),
|
|
323
|
+
)
|
|
324
|
+
all_results.extend(results.eval_results)
|
|
325
|
+
cloud_metrics_run += 1
|
|
326
|
+
else:
|
|
327
|
+
# No cloud available
|
|
328
|
+
from fi.evals.types import EvalResult
|
|
329
|
+
for _ in data:
|
|
330
|
+
all_results.append(
|
|
331
|
+
EvalResult(
|
|
332
|
+
name=template,
|
|
333
|
+
output=None,
|
|
334
|
+
reason="Cloud unavailable - cannot run this metric locally",
|
|
335
|
+
runtime=0,
|
|
336
|
+
)
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
# Run evaluations
|
|
340
|
+
if not quiet:
|
|
341
|
+
with Progress(
|
|
342
|
+
SpinnerColumn(),
|
|
343
|
+
TextColumn("[progress.description]{task.description}"),
|
|
344
|
+
console=console,
|
|
345
|
+
) as progress:
|
|
346
|
+
for eval_def in evaluations:
|
|
347
|
+
task = progress.add_task(
|
|
348
|
+
f"Running {eval_def['template']}...",
|
|
349
|
+
total=None
|
|
350
|
+
)
|
|
351
|
+
|
|
352
|
+
try:
|
|
353
|
+
run_evaluation(eval_def)
|
|
354
|
+
progress.update(task, description=f"[green]✓[/green] {eval_def['template']}")
|
|
355
|
+
except Exception as e:
|
|
356
|
+
progress.update(task, description=f"[red]✗[/red] {eval_def['template']}: {e}")
|
|
357
|
+
if not quiet:
|
|
358
|
+
console.print(f"[red]Error running {eval_def['template']}: {e}[/red]")
|
|
359
|
+
|
|
360
|
+
progress.remove_task(task)
|
|
361
|
+
else:
|
|
362
|
+
for eval_def in evaluations:
|
|
363
|
+
try:
|
|
364
|
+
run_evaluation(eval_def)
|
|
365
|
+
except Exception as e:
|
|
366
|
+
console.print(f"[red]Error: {e}[/red]", file=sys.stderr)
|
|
367
|
+
|
|
368
|
+
# Show execution summary in hybrid mode
|
|
369
|
+
if not quiet and effective_mode == ExecutionModeOption.hybrid:
|
|
370
|
+
console.print(
|
|
371
|
+
f"\n[dim]Executed: {local_metrics_run} local, {cloud_metrics_run} cloud[/dim]"
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
# Create combined results
|
|
375
|
+
from fi.evals.types import BatchRunResult
|
|
376
|
+
combined_results = BatchRunResult(eval_results=all_results)
|
|
377
|
+
|
|
378
|
+
# Format and display results
|
|
379
|
+
output_path = str(output_file) if output_file else None
|
|
380
|
+
result_str = format_results(combined_results, output, console, output_path)
|
|
381
|
+
|
|
382
|
+
# For non-table formats, print the result
|
|
383
|
+
if output != "table" and result_str:
|
|
384
|
+
if output_file:
|
|
385
|
+
print_success(f"Results saved to: {output_file}")
|
|
386
|
+
else:
|
|
387
|
+
console.print(result_str)
|
|
388
|
+
|
|
389
|
+
# Print summary for table output
|
|
390
|
+
if output == "table" and not quiet:
|
|
391
|
+
reporter = ResultReporter(console)
|
|
392
|
+
reporter.report_summary(combined_results)
|
|
393
|
+
|
|
394
|
+
# Save run to history
|
|
395
|
+
if not no_save and all_results:
|
|
396
|
+
from fi.cli.storage import RunHistory
|
|
397
|
+
|
|
398
|
+
history = RunHistory()
|
|
399
|
+
templates_used = list(set(e["template"] for e in evaluations))
|
|
400
|
+
config_path_str = str(config) if config else None
|
|
401
|
+
|
|
402
|
+
record = history.save_run(
|
|
403
|
+
results=combined_results,
|
|
404
|
+
config_file=config_path_str,
|
|
405
|
+
templates=templates_used,
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
if not quiet:
|
|
409
|
+
console.print(f"\n[dim]Run saved: {record.run_id}[/dim]")
|
|
410
|
+
console.print("[dim]View with: fi view --last[/dim]")
|
|
411
|
+
|
|
412
|
+
# Check assertions if configured
|
|
413
|
+
if check_assertions and 'eval_config' in dir() and eval_config is not None:
|
|
414
|
+
assertion_config = _build_assertion_config(eval_config, fail_fast)
|
|
415
|
+
|
|
416
|
+
if assertion_config.get('assertions') or assertion_config.get('thresholds', {}).get('default_pass_rate'):
|
|
417
|
+
# Convert results to dict format for evaluator
|
|
418
|
+
results_dict = {
|
|
419
|
+
"eval_results": [
|
|
420
|
+
{
|
|
421
|
+
"name": r.name,
|
|
422
|
+
"output": r.output,
|
|
423
|
+
"reason": r.reason,
|
|
424
|
+
"runtime": r.runtime,
|
|
425
|
+
"output_type": r.output_type,
|
|
426
|
+
"eval_id": r.eval_id,
|
|
427
|
+
}
|
|
428
|
+
for r in all_results
|
|
429
|
+
]
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
evaluator = AssertionEvaluator(results_dict, assertion_config)
|
|
433
|
+
report = evaluator.evaluate_all()
|
|
434
|
+
|
|
435
|
+
# Display assertion report
|
|
436
|
+
if not quiet and report.total_assertions > 0:
|
|
437
|
+
reporter = AssertionReporter(console)
|
|
438
|
+
console.print() # Blank line before assertions
|
|
439
|
+
reporter.display(report)
|
|
440
|
+
reporter.display_summary_line(report)
|
|
441
|
+
|
|
442
|
+
# Determine exit code based on assertion results
|
|
443
|
+
if report.failed > 0:
|
|
444
|
+
raise typer.Exit(ExitCode.ASSERTION_FAILED)
|
|
445
|
+
elif report.warnings > 0 and strict:
|
|
446
|
+
raise typer.Exit(ExitCode.ASSERTION_WARNING)
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _build_assertion_config(
|
|
450
|
+
eval_config,
|
|
451
|
+
fail_fast: bool = False
|
|
452
|
+
) -> Dict[str, Any]:
|
|
453
|
+
"""Build assertion config dictionary from FIEvaluationConfig.
|
|
454
|
+
|
|
455
|
+
Args:
|
|
456
|
+
eval_config: The loaded FIEvaluationConfig object.
|
|
457
|
+
fail_fast: Whether to enable fail-fast mode.
|
|
458
|
+
|
|
459
|
+
Returns:
|
|
460
|
+
Dictionary with 'assertions' and 'thresholds' keys.
|
|
461
|
+
"""
|
|
462
|
+
assertion_config: Dict[str, Any] = {
|
|
463
|
+
"assertions": [],
|
|
464
|
+
"thresholds": {}
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
# Convert assertion configs
|
|
468
|
+
if eval_config.assertions:
|
|
469
|
+
for assertion in eval_config.assertions:
|
|
470
|
+
assertion_dict = {
|
|
471
|
+
"template": assertion.template,
|
|
472
|
+
"global": assertion.is_global,
|
|
473
|
+
"conditions": assertion.conditions,
|
|
474
|
+
"on_fail": assertion.on_fail,
|
|
475
|
+
}
|
|
476
|
+
assertion_config["assertions"].append(assertion_dict)
|
|
477
|
+
|
|
478
|
+
# Convert thresholds config
|
|
479
|
+
if eval_config.thresholds:
|
|
480
|
+
assertion_config["thresholds"] = {
|
|
481
|
+
"default_pass_rate": eval_config.thresholds.default_pass_rate,
|
|
482
|
+
"fail_fast": fail_fast or eval_config.thresholds.fail_fast,
|
|
483
|
+
"overrides": eval_config.thresholds.overrides or {},
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
return assertion_config
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Validate command for checking configuration files."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
|
|
9
|
+
from fi.cli.config.loader import load_config, load_test_data, find_config_file
|
|
10
|
+
from fi.cli.utils.console import console, print_error, print_success, print_warning
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def validate(
|
|
14
|
+
config: Optional[Path] = typer.Option(
|
|
15
|
+
None,
|
|
16
|
+
"--config", "-c",
|
|
17
|
+
help="Path to configuration file",
|
|
18
|
+
),
|
|
19
|
+
strict: bool = typer.Option(
|
|
20
|
+
False,
|
|
21
|
+
"--strict", "-s",
|
|
22
|
+
help="Enable strict validation mode",
|
|
23
|
+
),
|
|
24
|
+
) -> None:
|
|
25
|
+
"""
|
|
26
|
+
Validate configuration file and test data.
|
|
27
|
+
|
|
28
|
+
Checks:
|
|
29
|
+
- YAML syntax validity
|
|
30
|
+
- Template name existence
|
|
31
|
+
- Required input fields for templates
|
|
32
|
+
- Data file accessibility
|
|
33
|
+
- API key presence (warning if missing)
|
|
34
|
+
"""
|
|
35
|
+
errors = []
|
|
36
|
+
warnings = []
|
|
37
|
+
|
|
38
|
+
# Find config file
|
|
39
|
+
if config:
|
|
40
|
+
config_path = Path(config)
|
|
41
|
+
else:
|
|
42
|
+
config_path = find_config_file()
|
|
43
|
+
|
|
44
|
+
if not config_path:
|
|
45
|
+
print_error("No configuration file found.")
|
|
46
|
+
raise typer.Exit(1)
|
|
47
|
+
|
|
48
|
+
console.print(f"[dim]Validating: {config_path}[/dim]\n")
|
|
49
|
+
|
|
50
|
+
# 1. Load and validate config
|
|
51
|
+
try:
|
|
52
|
+
eval_config = load_config(config_path)
|
|
53
|
+
console.print("[green]✓[/green] Configuration file syntax is valid")
|
|
54
|
+
except FileNotFoundError as e:
|
|
55
|
+
errors.append(f"Configuration file not found: {e}")
|
|
56
|
+
except ValueError as e:
|
|
57
|
+
errors.append(f"Configuration validation failed: {e}")
|
|
58
|
+
|
|
59
|
+
if errors:
|
|
60
|
+
_print_validation_results(errors, warnings, strict)
|
|
61
|
+
raise typer.Exit(1)
|
|
62
|
+
|
|
63
|
+
# 2. Validate templates exist
|
|
64
|
+
from fi.evals import templates as templates_module
|
|
65
|
+
from fi.evals.templates import EvalTemplate
|
|
66
|
+
|
|
67
|
+
available_templates = set()
|
|
68
|
+
for name in dir(templates_module):
|
|
69
|
+
obj = getattr(templates_module, name)
|
|
70
|
+
if (
|
|
71
|
+
isinstance(obj, type)
|
|
72
|
+
and issubclass(obj, EvalTemplate)
|
|
73
|
+
and obj is not EvalTemplate
|
|
74
|
+
and hasattr(obj, "eval_name")
|
|
75
|
+
):
|
|
76
|
+
available_templates.add(obj.eval_name)
|
|
77
|
+
|
|
78
|
+
for eval_def in eval_config.evaluations:
|
|
79
|
+
templates = eval_def.templates or ([eval_def.template] if eval_def.template else [])
|
|
80
|
+
for template in templates:
|
|
81
|
+
if template and template not in available_templates:
|
|
82
|
+
errors.append(
|
|
83
|
+
f"Unknown template '{template}' in evaluation '{eval_def.name}'. "
|
|
84
|
+
f"Run 'fi list templates' to see available templates."
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
if not errors:
|
|
88
|
+
console.print("[green]✓[/green] All template names are valid")
|
|
89
|
+
|
|
90
|
+
# 3. Validate data files
|
|
91
|
+
data_files_valid = True
|
|
92
|
+
for eval_def in eval_config.evaluations:
|
|
93
|
+
data_path = Path(eval_def.data)
|
|
94
|
+
|
|
95
|
+
# Check if path is relative to config file
|
|
96
|
+
if not data_path.is_absolute():
|
|
97
|
+
data_path = config_path.parent / data_path
|
|
98
|
+
|
|
99
|
+
if not data_path.exists():
|
|
100
|
+
errors.append(f"Data file not found: {eval_def.data}")
|
|
101
|
+
data_files_valid = False
|
|
102
|
+
else:
|
|
103
|
+
try:
|
|
104
|
+
test_data = load_test_data(data_path)
|
|
105
|
+
if not test_data:
|
|
106
|
+
warnings.append(f"Data file is empty: {eval_def.data}")
|
|
107
|
+
elif len(test_data) == 0:
|
|
108
|
+
warnings.append(f"No test cases in: {eval_def.data}")
|
|
109
|
+
except Exception as e:
|
|
110
|
+
errors.append(f"Error loading data file {eval_def.data}: {e}")
|
|
111
|
+
data_files_valid = False
|
|
112
|
+
|
|
113
|
+
if data_files_valid and not errors:
|
|
114
|
+
console.print("[green]✓[/green] All data files are accessible")
|
|
115
|
+
|
|
116
|
+
# 4. Check API keys
|
|
117
|
+
api_key = os.environ.get("FI_API_KEY")
|
|
118
|
+
secret_key = os.environ.get("FI_SECRET_KEY")
|
|
119
|
+
|
|
120
|
+
if not api_key:
|
|
121
|
+
warnings.append("FI_API_KEY environment variable not set")
|
|
122
|
+
if not secret_key:
|
|
123
|
+
warnings.append("FI_SECRET_KEY environment variable not set")
|
|
124
|
+
|
|
125
|
+
if api_key and secret_key:
|
|
126
|
+
console.print("[green]✓[/green] API keys are configured")
|
|
127
|
+
else:
|
|
128
|
+
console.print("[yellow]![/yellow] API keys not configured (evaluations will fail)")
|
|
129
|
+
|
|
130
|
+
# 5. Validate output configuration
|
|
131
|
+
if eval_config.output:
|
|
132
|
+
output_path = Path(eval_config.output.path)
|
|
133
|
+
if not output_path.is_absolute():
|
|
134
|
+
output_path = config_path.parent / output_path
|
|
135
|
+
|
|
136
|
+
if output_path.exists() and not output_path.is_dir():
|
|
137
|
+
warnings.append(f"Output path exists but is not a directory: {eval_config.output.path}")
|
|
138
|
+
elif not output_path.exists():
|
|
139
|
+
warnings.append(f"Output directory does not exist (will be created): {eval_config.output.path}")
|
|
140
|
+
|
|
141
|
+
# Print results
|
|
142
|
+
_print_validation_results(errors, warnings, strict)
|
|
143
|
+
|
|
144
|
+
if errors:
|
|
145
|
+
raise typer.Exit(1)
|
|
146
|
+
elif strict and warnings:
|
|
147
|
+
raise typer.Exit(1)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _print_validation_results(errors: list, warnings: list, strict: bool) -> None:
|
|
151
|
+
"""Print validation errors and warnings."""
|
|
152
|
+
console.print()
|
|
153
|
+
|
|
154
|
+
if errors:
|
|
155
|
+
console.print("[bold red]Errors:[/bold red]")
|
|
156
|
+
for error in errors:
|
|
157
|
+
console.print(f" [red]✗[/red] {error}")
|
|
158
|
+
|
|
159
|
+
if warnings:
|
|
160
|
+
console.print("[bold yellow]Warnings:[/bold yellow]")
|
|
161
|
+
for warning in warnings:
|
|
162
|
+
console.print(f" [yellow]![/yellow] {warning}")
|
|
163
|
+
|
|
164
|
+
console.print()
|
|
165
|
+
|
|
166
|
+
if errors:
|
|
167
|
+
print_error(f"Validation failed with {len(errors)} error(s)")
|
|
168
|
+
elif warnings and strict:
|
|
169
|
+
print_warning(f"Validation failed with {len(warnings)} warning(s) (strict mode)")
|
|
170
|
+
elif warnings:
|
|
171
|
+
print_warning(f"Validation passed with {len(warnings)} warning(s)")
|
|
172
|
+
else:
|
|
173
|
+
print_success("Validation passed!")
|