agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/cli/config/loader.py
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""Configuration file loading and discovery."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Optional, Union
|
|
5
|
+
|
|
6
|
+
import yaml
|
|
7
|
+
from pydantic import ValidationError
|
|
8
|
+
|
|
9
|
+
from fi.cli.config.schema import FIEvaluationConfig
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
# Default config file names to look for
|
|
13
|
+
CONFIG_FILE_NAMES = [
|
|
14
|
+
"fi-evaluation.yaml",
|
|
15
|
+
"fi-evaluation.yml",
|
|
16
|
+
".fi-evaluation.yaml",
|
|
17
|
+
".fi-evaluation.yml",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def find_config_file(start_path: Optional[Path] = None) -> Optional[Path]:
|
|
22
|
+
"""
|
|
23
|
+
Find a configuration file by searching current directory and parents.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
start_path: Starting directory for search (default: current working directory)
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
Path to configuration file if found, None otherwise
|
|
30
|
+
"""
|
|
31
|
+
if start_path is None:
|
|
32
|
+
start_path = Path.cwd()
|
|
33
|
+
|
|
34
|
+
current = start_path.resolve()
|
|
35
|
+
|
|
36
|
+
# Search up the directory tree
|
|
37
|
+
while current != current.parent:
|
|
38
|
+
for config_name in CONFIG_FILE_NAMES:
|
|
39
|
+
config_path = current / config_name
|
|
40
|
+
if config_path.exists():
|
|
41
|
+
return config_path
|
|
42
|
+
current = current.parent
|
|
43
|
+
|
|
44
|
+
# Check root as well
|
|
45
|
+
for config_name in CONFIG_FILE_NAMES:
|
|
46
|
+
config_path = current / config_name
|
|
47
|
+
if config_path.exists():
|
|
48
|
+
return config_path
|
|
49
|
+
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def load_config(
|
|
54
|
+
config_path: Optional[Union[str, Path]] = None
|
|
55
|
+
) -> FIEvaluationConfig:
|
|
56
|
+
"""
|
|
57
|
+
Load and validate configuration from a YAML file.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
config_path: Path to configuration file. If not provided,
|
|
61
|
+
will search for config file automatically.
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
Validated FIEvaluationConfig object
|
|
65
|
+
|
|
66
|
+
Raises:
|
|
67
|
+
FileNotFoundError: If config file not found
|
|
68
|
+
ValueError: If config file is invalid YAML
|
|
69
|
+
ValidationError: If config doesn't match schema
|
|
70
|
+
"""
|
|
71
|
+
if config_path is None:
|
|
72
|
+
config_path = find_config_file()
|
|
73
|
+
if config_path is None:
|
|
74
|
+
raise FileNotFoundError(
|
|
75
|
+
"No configuration file found. "
|
|
76
|
+
"Create a fi-evaluation.yaml file or specify --config path."
|
|
77
|
+
)
|
|
78
|
+
else:
|
|
79
|
+
config_path = Path(config_path)
|
|
80
|
+
|
|
81
|
+
if not config_path.exists():
|
|
82
|
+
raise FileNotFoundError(f"Configuration file not found: {config_path}")
|
|
83
|
+
|
|
84
|
+
# Load YAML
|
|
85
|
+
try:
|
|
86
|
+
with open(config_path, "r") as f:
|
|
87
|
+
raw_config = yaml.safe_load(f)
|
|
88
|
+
except yaml.YAMLError as e:
|
|
89
|
+
raise ValueError(f"Invalid YAML in configuration file: {e}")
|
|
90
|
+
|
|
91
|
+
if raw_config is None:
|
|
92
|
+
raise ValueError("Configuration file is empty")
|
|
93
|
+
|
|
94
|
+
# Validate against schema
|
|
95
|
+
try:
|
|
96
|
+
config = FIEvaluationConfig(**raw_config)
|
|
97
|
+
except ValidationError as e:
|
|
98
|
+
raise ValueError(f"Configuration validation failed: {e}")
|
|
99
|
+
|
|
100
|
+
return config
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def load_test_data(data_path: Union[str, Path]) -> list:
|
|
104
|
+
"""
|
|
105
|
+
Load test data from a JSON, JSONL, or CSV file.
|
|
106
|
+
|
|
107
|
+
Args:
|
|
108
|
+
data_path: Path to test data file
|
|
109
|
+
|
|
110
|
+
Returns:
|
|
111
|
+
List of test case dictionaries
|
|
112
|
+
|
|
113
|
+
Raises:
|
|
114
|
+
FileNotFoundError: If data file not found
|
|
115
|
+
ValueError: If data format is unsupported or invalid
|
|
116
|
+
"""
|
|
117
|
+
import json
|
|
118
|
+
import csv
|
|
119
|
+
|
|
120
|
+
data_path = Path(data_path)
|
|
121
|
+
|
|
122
|
+
if not data_path.exists():
|
|
123
|
+
raise FileNotFoundError(f"Test data file not found: {data_path}")
|
|
124
|
+
|
|
125
|
+
suffix = data_path.suffix.lower()
|
|
126
|
+
|
|
127
|
+
if suffix == ".json":
|
|
128
|
+
with open(data_path, "r") as f:
|
|
129
|
+
data = json.load(f)
|
|
130
|
+
if isinstance(data, dict):
|
|
131
|
+
return [data]
|
|
132
|
+
return data
|
|
133
|
+
|
|
134
|
+
elif suffix == ".jsonl":
|
|
135
|
+
data = []
|
|
136
|
+
with open(data_path, "r") as f:
|
|
137
|
+
for line in f:
|
|
138
|
+
line = line.strip()
|
|
139
|
+
if line:
|
|
140
|
+
data.append(json.loads(line))
|
|
141
|
+
return data
|
|
142
|
+
|
|
143
|
+
elif suffix == ".csv":
|
|
144
|
+
data = []
|
|
145
|
+
with open(data_path, "r", newline="") as f:
|
|
146
|
+
reader = csv.DictReader(f)
|
|
147
|
+
for row in reader:
|
|
148
|
+
data.append(dict(row))
|
|
149
|
+
return data
|
|
150
|
+
|
|
151
|
+
else:
|
|
152
|
+
raise ValueError(
|
|
153
|
+
f"Unsupported data format: {suffix}. "
|
|
154
|
+
"Supported formats: .json, .jsonl, .csv"
|
|
155
|
+
)
|
fi/cli/config/schema.py
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""Configuration schema definitions for fi-evaluation.yaml."""
|
|
2
|
+
|
|
3
|
+
from typing import List, Optional, Dict, Any
|
|
4
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class APIConfig(BaseModel):
|
|
8
|
+
"""API configuration settings."""
|
|
9
|
+
base_url: str = Field(
|
|
10
|
+
default="https://api.futureagi.com",
|
|
11
|
+
description="Base URL for the Future AGI API"
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class DefaultsConfig(BaseModel):
|
|
16
|
+
"""Default settings for evaluations."""
|
|
17
|
+
model: str = Field(
|
|
18
|
+
default="gpt-4o",
|
|
19
|
+
description="Default model for LLM-as-judge evaluations"
|
|
20
|
+
)
|
|
21
|
+
timeout: int = Field(
|
|
22
|
+
default=200,
|
|
23
|
+
description="Default timeout in seconds"
|
|
24
|
+
)
|
|
25
|
+
parallel_workers: int = Field(
|
|
26
|
+
default=8,
|
|
27
|
+
description="Number of parallel workers for evaluation"
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class EvaluationConfig(BaseModel):
|
|
32
|
+
"""Configuration for a single evaluation."""
|
|
33
|
+
name: str = Field(..., description="Name of this evaluation")
|
|
34
|
+
template: Optional[str] = Field(
|
|
35
|
+
default=None,
|
|
36
|
+
description="Single evaluation template to use"
|
|
37
|
+
)
|
|
38
|
+
templates: Optional[List[str]] = Field(
|
|
39
|
+
default=None,
|
|
40
|
+
description="List of evaluation templates to use"
|
|
41
|
+
)
|
|
42
|
+
data: str = Field(..., description="Path to test data file")
|
|
43
|
+
config: Optional[Dict[str, Any]] = Field(
|
|
44
|
+
default=None,
|
|
45
|
+
description="Additional configuration for the evaluation"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
@field_validator("templates", "template")
|
|
49
|
+
@classmethod
|
|
50
|
+
def validate_template_presence(cls, v, info):
|
|
51
|
+
"""Ensure at least one template specification method is used."""
|
|
52
|
+
return v
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class AssertionConfig(BaseModel):
|
|
56
|
+
"""Configuration for evaluation assertions."""
|
|
57
|
+
template: Optional[str] = Field(
|
|
58
|
+
default=None,
|
|
59
|
+
description="Template to assert on (mutually exclusive with 'global')"
|
|
60
|
+
)
|
|
61
|
+
conditions: List[str] = Field(
|
|
62
|
+
default_factory=list,
|
|
63
|
+
description="List of assertion conditions (e.g., 'pass_rate >= 0.85')"
|
|
64
|
+
)
|
|
65
|
+
on_fail: str = Field(
|
|
66
|
+
default="error",
|
|
67
|
+
description="Action on failure: error, warn, or skip"
|
|
68
|
+
)
|
|
69
|
+
is_global: bool = Field(
|
|
70
|
+
default=False,
|
|
71
|
+
alias="global",
|
|
72
|
+
description="If true, assertion applies globally across all templates"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
model_config = ConfigDict(populate_by_name=True)
|
|
76
|
+
|
|
77
|
+
@field_validator("on_fail")
|
|
78
|
+
@classmethod
|
|
79
|
+
def validate_on_fail(cls, v):
|
|
80
|
+
"""Validate on_fail value."""
|
|
81
|
+
valid_values = ["warn", "error", "skip"]
|
|
82
|
+
if v not in valid_values:
|
|
83
|
+
raise ValueError(f"on_fail must be one of: {valid_values}")
|
|
84
|
+
return v
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class ThresholdOverrides(BaseModel):
|
|
88
|
+
"""Per-template threshold overrides."""
|
|
89
|
+
model_config = ConfigDict(extra="allow")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class ThresholdsConfig(BaseModel):
|
|
93
|
+
"""Threshold shortcuts for assertions."""
|
|
94
|
+
default_pass_rate: Optional[float] = Field(
|
|
95
|
+
default=None,
|
|
96
|
+
description="Default pass rate threshold for all templates"
|
|
97
|
+
)
|
|
98
|
+
fail_fast: bool = Field(
|
|
99
|
+
default=False,
|
|
100
|
+
description="Stop on first assertion failure"
|
|
101
|
+
)
|
|
102
|
+
overrides: Optional[Dict[str, float]] = Field(
|
|
103
|
+
default=None,
|
|
104
|
+
description="Per-template threshold overrides"
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class OutputConfig(BaseModel):
|
|
109
|
+
"""Output configuration settings."""
|
|
110
|
+
format: str = Field(
|
|
111
|
+
default="json",
|
|
112
|
+
description="Output format: json, table, csv, html"
|
|
113
|
+
)
|
|
114
|
+
path: str = Field(
|
|
115
|
+
default="./results/",
|
|
116
|
+
description="Path to save results"
|
|
117
|
+
)
|
|
118
|
+
include_metadata: bool = Field(
|
|
119
|
+
default=True,
|
|
120
|
+
description="Include metadata in output"
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
@field_validator("format")
|
|
124
|
+
@classmethod
|
|
125
|
+
def validate_format(cls, v):
|
|
126
|
+
"""Validate output format."""
|
|
127
|
+
valid_formats = ["json", "table", "csv", "html"]
|
|
128
|
+
if v not in valid_formats:
|
|
129
|
+
raise ValueError(f"format must be one of: {valid_formats}")
|
|
130
|
+
return v
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
class FIEvaluationConfig(BaseModel):
|
|
134
|
+
"""Root configuration schema for fi-evaluation.yaml."""
|
|
135
|
+
version: str = Field(
|
|
136
|
+
default="1.0",
|
|
137
|
+
description="Configuration file version"
|
|
138
|
+
)
|
|
139
|
+
api: Optional[APIConfig] = Field(
|
|
140
|
+
default=None,
|
|
141
|
+
description="API configuration"
|
|
142
|
+
)
|
|
143
|
+
defaults: Optional[DefaultsConfig] = Field(
|
|
144
|
+
default=None,
|
|
145
|
+
description="Default evaluation settings"
|
|
146
|
+
)
|
|
147
|
+
evaluations: List[EvaluationConfig] = Field(
|
|
148
|
+
...,
|
|
149
|
+
description="List of evaluation configurations"
|
|
150
|
+
)
|
|
151
|
+
output: Optional[OutputConfig] = Field(
|
|
152
|
+
default=None,
|
|
153
|
+
description="Output configuration"
|
|
154
|
+
)
|
|
155
|
+
assertions: Optional[List[AssertionConfig]] = Field(
|
|
156
|
+
default=None,
|
|
157
|
+
description="Assertions to run on evaluation results"
|
|
158
|
+
)
|
|
159
|
+
thresholds: Optional[ThresholdsConfig] = Field(
|
|
160
|
+
default=None,
|
|
161
|
+
description="Threshold shortcuts for assertions"
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
def get_defaults(self) -> DefaultsConfig:
|
|
165
|
+
"""Get defaults config, creating one if not present."""
|
|
166
|
+
return self.defaults or DefaultsConfig()
|
|
167
|
+
|
|
168
|
+
def get_output_config(self) -> OutputConfig:
|
|
169
|
+
"""Get output config, creating one if not present."""
|
|
170
|
+
return self.output or OutputConfig()
|
|
171
|
+
|
|
172
|
+
def get_api_config(self) -> APIConfig:
|
|
173
|
+
"""Get API config, creating one if not present."""
|
|
174
|
+
return self.api or APIConfig()
|
fi/cli/main.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""
|
|
2
|
+
AI Evaluation CLI
|
|
3
|
+
|
|
4
|
+
Command-line interface for running LLM evaluations with 60+ templates.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
|
|
10
|
+
from fi.cli.commands.init import init_project
|
|
11
|
+
from fi.cli.commands.run import run
|
|
12
|
+
from fi.cli.commands.list_cmd import list_resources
|
|
13
|
+
from fi.cli.commands.validate import validate
|
|
14
|
+
from fi.cli.commands.config import config_app
|
|
15
|
+
from fi.cli.commands.view import view
|
|
16
|
+
from fi.cli.commands.export import export
|
|
17
|
+
|
|
18
|
+
# Create main app
|
|
19
|
+
app = typer.Typer(
|
|
20
|
+
name="fi",
|
|
21
|
+
help="AI Evaluation CLI - Evaluate LLM outputs with 60+ templates",
|
|
22
|
+
no_args_is_help=True,
|
|
23
|
+
rich_markup_mode="rich",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
console = Console()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Register commands
|
|
30
|
+
app.command("init", help="Initialize a new evaluation project")(init_project)
|
|
31
|
+
app.command("run", help="Run evaluations from config or CLI")(run)
|
|
32
|
+
app.command("list", help="List available templates and resources")(list_resources)
|
|
33
|
+
app.command("validate", help="Validate configuration file")(validate)
|
|
34
|
+
app.command("view", help="View evaluation results from previous runs")(view)
|
|
35
|
+
app.command("export", help="Export evaluation results to file")(export)
|
|
36
|
+
app.add_typer(config_app, name="config", help="Manage CLI configuration")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@app.callback(invoke_without_command=True)
|
|
40
|
+
def main_callback(
|
|
41
|
+
ctx: typer.Context,
|
|
42
|
+
version: bool = typer.Option(
|
|
43
|
+
False,
|
|
44
|
+
"--version", "-v",
|
|
45
|
+
help="Show version and exit",
|
|
46
|
+
),
|
|
47
|
+
) -> None:
|
|
48
|
+
"""
|
|
49
|
+
AI Evaluation CLI by Future AGI.
|
|
50
|
+
|
|
51
|
+
Evaluate LLM outputs with 60+ pre-built evaluation templates.
|
|
52
|
+
|
|
53
|
+
Quick Start:
|
|
54
|
+
fi init my-project # Initialize project
|
|
55
|
+
fi run # Run evaluations
|
|
56
|
+
fi list templates # List available templates
|
|
57
|
+
"""
|
|
58
|
+
if version:
|
|
59
|
+
from importlib.metadata import version as get_version
|
|
60
|
+
try:
|
|
61
|
+
v = get_version("ai-evaluation")
|
|
62
|
+
except Exception:
|
|
63
|
+
v = "1.0.0"
|
|
64
|
+
console.print(f"ai-evaluation version {v}")
|
|
65
|
+
raise typer.Exit()
|
|
66
|
+
|
|
67
|
+
# If no command provided and not asking for version, show help
|
|
68
|
+
if ctx.invoked_subcommand is None and not version:
|
|
69
|
+
console.print(ctx.get_help())
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def main() -> None:
|
|
73
|
+
"""Main entry point for the CLI."""
|
|
74
|
+
app()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
if __name__ == "__main__":
|
|
78
|
+
main()
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Format evaluation results for CLI output."""
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import io
|
|
5
|
+
import json
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
from rich.table import Table
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def format_results(
|
|
13
|
+
batch_result,
|
|
14
|
+
fmt: str,
|
|
15
|
+
console: Console,
|
|
16
|
+
output_path: Optional[str] = None,
|
|
17
|
+
) -> Optional[str]:
|
|
18
|
+
"""Format BatchRunResult for display or file output.
|
|
19
|
+
|
|
20
|
+
Args:
|
|
21
|
+
batch_result: BatchRunResult with eval_results list.
|
|
22
|
+
fmt: Output format — "table", "json", "csv", or "html".
|
|
23
|
+
console: Rich Console for table rendering.
|
|
24
|
+
output_path: Optional file path to write output to.
|
|
25
|
+
|
|
26
|
+
Returns:
|
|
27
|
+
Formatted string for non-table formats, None for table format.
|
|
28
|
+
"""
|
|
29
|
+
results = [r for r in batch_result.eval_results if r is not None]
|
|
30
|
+
|
|
31
|
+
if fmt == "table":
|
|
32
|
+
_print_table(results, console)
|
|
33
|
+
return None
|
|
34
|
+
elif fmt == "json":
|
|
35
|
+
text = _to_json(results)
|
|
36
|
+
elif fmt == "csv":
|
|
37
|
+
text = _to_csv(results)
|
|
38
|
+
elif fmt == "html":
|
|
39
|
+
text = _to_html(results)
|
|
40
|
+
else:
|
|
41
|
+
text = _to_json(results)
|
|
42
|
+
|
|
43
|
+
if output_path:
|
|
44
|
+
with open(output_path, "w") as f:
|
|
45
|
+
f.write(text)
|
|
46
|
+
|
|
47
|
+
return text
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _print_table(results, console: Console) -> None:
|
|
51
|
+
table = Table(title="Evaluation Results", show_lines=True)
|
|
52
|
+
table.add_column("Metric", style="cyan", no_wrap=True)
|
|
53
|
+
table.add_column("Output", style="green")
|
|
54
|
+
table.add_column("Reason", style="dim")
|
|
55
|
+
table.add_column("Runtime (ms)", justify="right")
|
|
56
|
+
|
|
57
|
+
for r in results:
|
|
58
|
+
table.add_row(
|
|
59
|
+
r.name,
|
|
60
|
+
str(r.output) if r.output is not None else "-",
|
|
61
|
+
(r.reason or "-")[:80],
|
|
62
|
+
str(r.runtime),
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
console.print(table)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _to_json(results) -> str:
|
|
69
|
+
rows = []
|
|
70
|
+
for r in results:
|
|
71
|
+
rows.append({
|
|
72
|
+
"name": r.name,
|
|
73
|
+
"output": r.output,
|
|
74
|
+
"reason": r.reason,
|
|
75
|
+
"runtime": r.runtime,
|
|
76
|
+
})
|
|
77
|
+
return json.dumps(rows, indent=2, default=str)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _to_csv(results) -> str:
|
|
81
|
+
buf = io.StringIO()
|
|
82
|
+
writer = csv.writer(buf)
|
|
83
|
+
writer.writerow(["name", "output", "reason", "runtime"])
|
|
84
|
+
for r in results:
|
|
85
|
+
writer.writerow([r.name, r.output, r.reason, r.runtime])
|
|
86
|
+
return buf.getvalue()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _to_html(results) -> str:
|
|
90
|
+
rows_html = ""
|
|
91
|
+
for r in results:
|
|
92
|
+
rows_html += (
|
|
93
|
+
f"<tr><td>{r.name}</td><td>{r.output}</td>"
|
|
94
|
+
f"<td>{r.reason or ''}</td><td>{r.runtime}</td></tr>\n"
|
|
95
|
+
)
|
|
96
|
+
return f"""<!DOCTYPE html>
|
|
97
|
+
<html><head><title>Evaluation Results</title>
|
|
98
|
+
<style>
|
|
99
|
+
body {{ font-family: sans-serif; margin: 2rem; }}
|
|
100
|
+
table {{ border-collapse: collapse; width: 100%; }}
|
|
101
|
+
th, td {{ border: 1px solid #ddd; padding: 8px; text-align: left; }}
|
|
102
|
+
th {{ background: #f5f5f5; }}
|
|
103
|
+
</style></head><body>
|
|
104
|
+
<h1>Evaluation Results</h1>
|
|
105
|
+
<table><tr><th>Metric</th><th>Output</th><th>Reason</th><th>Runtime (ms)</th></tr>
|
|
106
|
+
{rows_html}</table></body></html>"""
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Summary reporters for CLI evaluation results."""
|
|
2
|
+
|
|
3
|
+
from rich.console import Console
|
|
4
|
+
from rich.panel import Panel
|
|
5
|
+
from rich.table import Table
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ResultReporter:
|
|
9
|
+
"""Print a summary panel after evaluation results are displayed."""
|
|
10
|
+
|
|
11
|
+
def __init__(self, console: Console):
|
|
12
|
+
self.console = console
|
|
13
|
+
|
|
14
|
+
def report_summary(self, batch_result) -> None:
|
|
15
|
+
"""Print a summary of the batch evaluation results."""
|
|
16
|
+
results = [r for r in batch_result.eval_results if r is not None]
|
|
17
|
+
total = len(results)
|
|
18
|
+
|
|
19
|
+
if total == 0:
|
|
20
|
+
self.console.print("[yellow]No results to summarise.[/yellow]")
|
|
21
|
+
return
|
|
22
|
+
|
|
23
|
+
# Count numeric scores (output that can be interpreted as a number)
|
|
24
|
+
scores = []
|
|
25
|
+
for r in results:
|
|
26
|
+
try:
|
|
27
|
+
scores.append(float(r.output))
|
|
28
|
+
except (TypeError, ValueError):
|
|
29
|
+
pass
|
|
30
|
+
|
|
31
|
+
table = Table(show_header=False, box=None, padding=(0, 2))
|
|
32
|
+
table.add_column("label", style="bold")
|
|
33
|
+
table.add_column("value")
|
|
34
|
+
|
|
35
|
+
table.add_row("Total metrics", str(total))
|
|
36
|
+
|
|
37
|
+
if scores:
|
|
38
|
+
avg = sum(scores) / len(scores)
|
|
39
|
+
table.add_row("Avg score", f"{avg:.3f}")
|
|
40
|
+
table.add_row("Min score", f"{min(scores):.3f}")
|
|
41
|
+
table.add_row("Max score", f"{max(scores):.3f}")
|
|
42
|
+
|
|
43
|
+
total_runtime = sum(r.runtime for r in results)
|
|
44
|
+
table.add_row("Total runtime", f"{total_runtime} ms")
|
|
45
|
+
|
|
46
|
+
self.console.print(Panel(table, title="Summary", border_style="blue"))
|