agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""Assertion evaluator for checking conditions against evaluation results."""
|
|
2
|
+
|
|
3
|
+
from typing import Dict, List, Optional, Any
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from enum import Enum
|
|
6
|
+
import statistics
|
|
7
|
+
|
|
8
|
+
from .conditions import MetricType
|
|
9
|
+
from .parser import ConditionParser, ConditionParseError
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class AssertionResult(Enum):
|
|
13
|
+
"""Result status of an assertion evaluation."""
|
|
14
|
+
PASSED = "passed"
|
|
15
|
+
FAILED = "failed"
|
|
16
|
+
SKIPPED = "skipped"
|
|
17
|
+
WARNING = "warning"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class AssertionOutcome:
|
|
22
|
+
"""Result of evaluating a single assertion."""
|
|
23
|
+
template: Optional[str]
|
|
24
|
+
condition: str
|
|
25
|
+
expected: str
|
|
26
|
+
actual: float
|
|
27
|
+
result: AssertionResult
|
|
28
|
+
message: str
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class AssertionReport:
|
|
33
|
+
"""Complete assertion evaluation report."""
|
|
34
|
+
outcomes: List[AssertionOutcome] = field(default_factory=list)
|
|
35
|
+
total_assertions: int = 0
|
|
36
|
+
passed: int = 0
|
|
37
|
+
failed: int = 0
|
|
38
|
+
warnings: int = 0
|
|
39
|
+
skipped: int = 0
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def all_passed(self) -> bool:
|
|
43
|
+
"""Check if all assertions passed (no failures)."""
|
|
44
|
+
return self.failed == 0
|
|
45
|
+
|
|
46
|
+
@property
|
|
47
|
+
def has_warnings(self) -> bool:
|
|
48
|
+
"""Check if there are any warnings."""
|
|
49
|
+
return self.warnings > 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class AssertionEvaluator:
|
|
53
|
+
"""Evaluate assertions against evaluation results."""
|
|
54
|
+
|
|
55
|
+
def __init__(self, results: Dict[str, Any], config: Dict[str, Any]):
|
|
56
|
+
"""Initialize the evaluator.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
results: Dictionary containing evaluation results with 'eval_results' key.
|
|
60
|
+
config: Configuration dictionary that may contain 'assertions' and 'thresholds'.
|
|
61
|
+
"""
|
|
62
|
+
self.results = results
|
|
63
|
+
self.config = config
|
|
64
|
+
self.outcomes: List[AssertionOutcome] = []
|
|
65
|
+
|
|
66
|
+
def compute_metrics(self, template: Optional[str] = None) -> Dict[MetricType, float]:
|
|
67
|
+
"""Compute all metrics for a template or globally.
|
|
68
|
+
|
|
69
|
+
Args:
|
|
70
|
+
template: If provided, compute metrics only for this template.
|
|
71
|
+
If None, compute global metrics across all results.
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
Dictionary mapping MetricType to computed values.
|
|
75
|
+
"""
|
|
76
|
+
if template:
|
|
77
|
+
eval_results = [
|
|
78
|
+
r for r in self.results.get('eval_results', [])
|
|
79
|
+
if r.get('name') == template
|
|
80
|
+
]
|
|
81
|
+
else:
|
|
82
|
+
eval_results = self.results.get('eval_results', [])
|
|
83
|
+
|
|
84
|
+
if not eval_results:
|
|
85
|
+
return {}
|
|
86
|
+
|
|
87
|
+
# Extract scores/outputs
|
|
88
|
+
outputs = [r.get('output') for r in eval_results]
|
|
89
|
+
runtimes = [r.get('runtime', 0) for r in eval_results if r.get('runtime')]
|
|
90
|
+
|
|
91
|
+
# Calculate boolean pass/fail
|
|
92
|
+
bool_outputs = [o for o in outputs if isinstance(o, bool)]
|
|
93
|
+
numeric_outputs = [
|
|
94
|
+
float(o) for o in outputs
|
|
95
|
+
if isinstance(o, (int, float)) and not isinstance(o, bool)
|
|
96
|
+
]
|
|
97
|
+
|
|
98
|
+
# Pass rate (boolean outputs: True = pass, numeric: >= 0.5 = pass)
|
|
99
|
+
passes = sum(1 for o in bool_outputs if o is True)
|
|
100
|
+
passes += sum(1 for o in numeric_outputs if o >= 0.5)
|
|
101
|
+
total = len(outputs)
|
|
102
|
+
|
|
103
|
+
metrics: Dict[MetricType, float] = {
|
|
104
|
+
MetricType.PASS_RATE: passes / total if total > 0 else 0,
|
|
105
|
+
MetricType.TOTAL_PASS_RATE: passes / total if total > 0 else 0, # Alias for global
|
|
106
|
+
MetricType.PASSED_COUNT: float(passes),
|
|
107
|
+
MetricType.FAILED_COUNT: float(total - passes),
|
|
108
|
+
MetricType.TOTAL_COUNT: float(total),
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
# Score metrics (only for numeric outputs)
|
|
112
|
+
if numeric_outputs:
|
|
113
|
+
metrics[MetricType.AVG_SCORE] = statistics.mean(numeric_outputs)
|
|
114
|
+
metrics[MetricType.MIN_SCORE] = min(numeric_outputs)
|
|
115
|
+
metrics[MetricType.MAX_SCORE] = max(numeric_outputs)
|
|
116
|
+
|
|
117
|
+
sorted_outputs = sorted(numeric_outputs)
|
|
118
|
+
|
|
119
|
+
# Percentiles
|
|
120
|
+
metrics[MetricType.P50_SCORE] = self._percentile(sorted_outputs, 50)
|
|
121
|
+
metrics[MetricType.P90_SCORE] = self._percentile(sorted_outputs, 90)
|
|
122
|
+
metrics[MetricType.P95_SCORE] = self._percentile(sorted_outputs, 95)
|
|
123
|
+
|
|
124
|
+
# Runtime metrics
|
|
125
|
+
if runtimes:
|
|
126
|
+
metrics[MetricType.RUNTIME_AVG] = statistics.mean(runtimes)
|
|
127
|
+
metrics[MetricType.RUNTIME_P95] = self._percentile(sorted(runtimes), 95)
|
|
128
|
+
|
|
129
|
+
return metrics
|
|
130
|
+
|
|
131
|
+
def _percentile(self, sorted_data: List[float], p: float) -> float:
|
|
132
|
+
"""Calculate percentile from sorted data.
|
|
133
|
+
|
|
134
|
+
Args:
|
|
135
|
+
sorted_data: Pre-sorted list of values.
|
|
136
|
+
p: Percentile to calculate (0-100).
|
|
137
|
+
|
|
138
|
+
Returns:
|
|
139
|
+
The calculated percentile value.
|
|
140
|
+
"""
|
|
141
|
+
if not sorted_data:
|
|
142
|
+
return 0.0
|
|
143
|
+
|
|
144
|
+
n = len(sorted_data)
|
|
145
|
+
if n == 1:
|
|
146
|
+
return sorted_data[0]
|
|
147
|
+
|
|
148
|
+
# Linear interpolation
|
|
149
|
+
k = (n - 1) * (p / 100)
|
|
150
|
+
f = int(k)
|
|
151
|
+
c = f + 1 if f + 1 < n else f
|
|
152
|
+
|
|
153
|
+
if f == c:
|
|
154
|
+
return sorted_data[f]
|
|
155
|
+
|
|
156
|
+
return sorted_data[f] + (k - f) * (sorted_data[c] - sorted_data[f])
|
|
157
|
+
|
|
158
|
+
def evaluate_assertion(
|
|
159
|
+
self,
|
|
160
|
+
template: Optional[str],
|
|
161
|
+
condition_str: str,
|
|
162
|
+
on_fail: str = "error"
|
|
163
|
+
) -> AssertionOutcome:
|
|
164
|
+
"""Evaluate a single assertion condition.
|
|
165
|
+
|
|
166
|
+
Args:
|
|
167
|
+
template: Template name to evaluate against, or None for global.
|
|
168
|
+
condition_str: The condition string to evaluate.
|
|
169
|
+
on_fail: Action on failure - "error", "warn", or "skip".
|
|
170
|
+
|
|
171
|
+
Returns:
|
|
172
|
+
AssertionOutcome with the result.
|
|
173
|
+
"""
|
|
174
|
+
try:
|
|
175
|
+
condition = ConditionParser.parse(condition_str)
|
|
176
|
+
except ConditionParseError as e:
|
|
177
|
+
return AssertionOutcome(
|
|
178
|
+
template=template,
|
|
179
|
+
condition=condition_str,
|
|
180
|
+
expected="valid condition",
|
|
181
|
+
actual=0,
|
|
182
|
+
result=AssertionResult.FAILED,
|
|
183
|
+
message=f"Parse error: {e}"
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
metrics = self.compute_metrics(template)
|
|
187
|
+
|
|
188
|
+
if condition.metric not in metrics:
|
|
189
|
+
return AssertionOutcome(
|
|
190
|
+
template=template,
|
|
191
|
+
condition=condition_str,
|
|
192
|
+
expected=f"{condition.metric.value} {condition.operator.value} {condition.value}",
|
|
193
|
+
actual=0,
|
|
194
|
+
result=AssertionResult.SKIPPED,
|
|
195
|
+
message=f"Metric '{condition.metric.value}' not available for template '{template or 'global'}'"
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
actual_value = metrics[condition.metric]
|
|
199
|
+
passed = condition.evaluate(actual_value)
|
|
200
|
+
|
|
201
|
+
if passed:
|
|
202
|
+
result = AssertionResult.PASSED
|
|
203
|
+
message = "Assertion passed"
|
|
204
|
+
elif on_fail == "warn":
|
|
205
|
+
result = AssertionResult.WARNING
|
|
206
|
+
message = (
|
|
207
|
+
f"Warning: {condition.metric.value} is {actual_value:.4f}, "
|
|
208
|
+
f"expected {condition.operator.value} {condition.value}"
|
|
209
|
+
)
|
|
210
|
+
elif on_fail == "skip":
|
|
211
|
+
result = AssertionResult.SKIPPED
|
|
212
|
+
message = "Assertion skipped"
|
|
213
|
+
else:
|
|
214
|
+
result = AssertionResult.FAILED
|
|
215
|
+
message = (
|
|
216
|
+
f"Failed: {condition.metric.value} is {actual_value:.4f}, "
|
|
217
|
+
f"expected {condition.operator.value} {condition.value}"
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
return AssertionOutcome(
|
|
221
|
+
template=template,
|
|
222
|
+
condition=condition_str,
|
|
223
|
+
expected=f"{condition.operator.value} {condition.value}",
|
|
224
|
+
actual=actual_value,
|
|
225
|
+
result=result,
|
|
226
|
+
message=message
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
def evaluate_all(self) -> AssertionReport:
|
|
230
|
+
"""Evaluate all assertions from config.
|
|
231
|
+
|
|
232
|
+
Returns:
|
|
233
|
+
AssertionReport containing all outcomes and summary statistics.
|
|
234
|
+
"""
|
|
235
|
+
assertions = self.config.get('assertions', [])
|
|
236
|
+
thresholds = self.config.get('thresholds', {})
|
|
237
|
+
|
|
238
|
+
outcomes: List[AssertionOutcome] = []
|
|
239
|
+
|
|
240
|
+
# Evaluate explicit assertions
|
|
241
|
+
for assertion in assertions:
|
|
242
|
+
template = assertion.get('template')
|
|
243
|
+
is_global = assertion.get('global', False)
|
|
244
|
+
conditions = assertion.get('conditions', [])
|
|
245
|
+
on_fail = assertion.get('on_fail', 'error')
|
|
246
|
+
|
|
247
|
+
for condition_str in conditions:
|
|
248
|
+
outcome = self.evaluate_assertion(
|
|
249
|
+
template=None if is_global else template,
|
|
250
|
+
condition_str=condition_str,
|
|
251
|
+
on_fail=on_fail
|
|
252
|
+
)
|
|
253
|
+
outcomes.append(outcome)
|
|
254
|
+
|
|
255
|
+
# Evaluate threshold shortcuts
|
|
256
|
+
default_threshold = thresholds.get('default_pass_rate')
|
|
257
|
+
overrides = thresholds.get('overrides', {})
|
|
258
|
+
|
|
259
|
+
if default_threshold is not None:
|
|
260
|
+
templates = set(
|
|
261
|
+
r.get('name') for r in self.results.get('eval_results', [])
|
|
262
|
+
if r.get('name')
|
|
263
|
+
)
|
|
264
|
+
for template in templates:
|
|
265
|
+
threshold = overrides.get(template, default_threshold)
|
|
266
|
+
outcome = self.evaluate_assertion(
|
|
267
|
+
template=template,
|
|
268
|
+
condition_str=f"pass_rate >= {threshold}",
|
|
269
|
+
on_fail="error"
|
|
270
|
+
)
|
|
271
|
+
outcomes.append(outcome)
|
|
272
|
+
|
|
273
|
+
# Build report
|
|
274
|
+
passed = sum(1 for o in outcomes if o.result == AssertionResult.PASSED)
|
|
275
|
+
failed = sum(1 for o in outcomes if o.result == AssertionResult.FAILED)
|
|
276
|
+
warnings = sum(1 for o in outcomes if o.result == AssertionResult.WARNING)
|
|
277
|
+
skipped = sum(1 for o in outcomes if o.result == AssertionResult.SKIPPED)
|
|
278
|
+
|
|
279
|
+
return AssertionReport(
|
|
280
|
+
outcomes=outcomes,
|
|
281
|
+
total_assertions=len(outcomes),
|
|
282
|
+
passed=passed,
|
|
283
|
+
failed=failed,
|
|
284
|
+
warnings=warnings,
|
|
285
|
+
skipped=skipped
|
|
286
|
+
)
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Exit codes for CLI operations."""
|
|
2
|
+
|
|
3
|
+
from enum import IntEnum
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class ExitCode(IntEnum):
|
|
7
|
+
"""Exit codes for CI/CD integration.
|
|
8
|
+
|
|
9
|
+
These codes allow CI systems to distinguish between different
|
|
10
|
+
types of failures and take appropriate actions.
|
|
11
|
+
"""
|
|
12
|
+
SUCCESS = 0 # All evaluations and assertions passed
|
|
13
|
+
EVALUATION_ERROR = 1 # Error during evaluation execution
|
|
14
|
+
ASSERTION_FAILED = 2 # One or more assertions failed
|
|
15
|
+
ASSERTION_WARNING = 3 # Assertions passed but with warnings (--strict mode)
|
|
16
|
+
CONFIG_ERROR = 4 # Configuration file error
|
|
17
|
+
DATA_ERROR = 5 # Test data file error
|
|
18
|
+
API_ERROR = 6 # API connection/authentication error
|
|
19
|
+
TIMEOUT_ERROR = 7 # Evaluation timeout
|
|
20
|
+
UNKNOWN_ERROR = 99 # Unknown error
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Parser for assertion condition strings."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from typing import List
|
|
5
|
+
|
|
6
|
+
from .conditions import Condition, Operator, MetricType
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ConditionParseError(ValueError):
|
|
10
|
+
"""Error raised when a condition string cannot be parsed."""
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ConditionParser:
|
|
15
|
+
"""Parse string conditions into Condition objects."""
|
|
16
|
+
|
|
17
|
+
# Pattern: metric_name operator value
|
|
18
|
+
PATTERN = re.compile(
|
|
19
|
+
r'^(\w+)\s*(>=|<=|==|!=|>|<)\s*([\d.]+)$'
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
# Pattern for between: value <= metric <= value
|
|
23
|
+
BETWEEN_PATTERN = re.compile(
|
|
24
|
+
r'^([\d.]+)\s*<=\s*(\w+)\s*<=\s*([\d.]+)$'
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
METRIC_MAP = {
|
|
28
|
+
'pass_rate': MetricType.PASS_RATE,
|
|
29
|
+
'avg_score': MetricType.AVG_SCORE,
|
|
30
|
+
'min_score': MetricType.MIN_SCORE,
|
|
31
|
+
'max_score': MetricType.MAX_SCORE,
|
|
32
|
+
'failed_count': MetricType.FAILED_COUNT,
|
|
33
|
+
'passed_count': MetricType.PASSED_COUNT,
|
|
34
|
+
'total_count': MetricType.TOTAL_COUNT,
|
|
35
|
+
'p50_score': MetricType.P50_SCORE,
|
|
36
|
+
'p90_score': MetricType.P90_SCORE,
|
|
37
|
+
'p95_score': MetricType.P95_SCORE,
|
|
38
|
+
'runtime_avg': MetricType.RUNTIME_AVG,
|
|
39
|
+
'runtime_p95': MetricType.RUNTIME_P95,
|
|
40
|
+
'total_pass_rate': MetricType.TOTAL_PASS_RATE,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
OPERATOR_MAP = {
|
|
44
|
+
'>=': Operator.GTE,
|
|
45
|
+
'<=': Operator.LTE,
|
|
46
|
+
'>': Operator.GT,
|
|
47
|
+
'<': Operator.LT,
|
|
48
|
+
'==': Operator.EQ,
|
|
49
|
+
'!=': Operator.NEQ,
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
@classmethod
|
|
53
|
+
def parse(cls, condition_str: str) -> Condition:
|
|
54
|
+
"""Parse a condition string into a Condition object.
|
|
55
|
+
|
|
56
|
+
Supports two formats:
|
|
57
|
+
1. Standard: "metric_name operator value" (e.g., "pass_rate >= 0.85")
|
|
58
|
+
2. Between: "value1 <= metric_name <= value2" (e.g., "0.5 <= avg_score <= 1.0")
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
condition_str: The condition string to parse.
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
A Condition object.
|
|
65
|
+
|
|
66
|
+
Raises:
|
|
67
|
+
ConditionParseError: If the condition string is invalid.
|
|
68
|
+
"""
|
|
69
|
+
condition_str = condition_str.strip()
|
|
70
|
+
|
|
71
|
+
# Try between pattern first
|
|
72
|
+
between_match = cls.BETWEEN_PATTERN.match(condition_str)
|
|
73
|
+
if between_match:
|
|
74
|
+
value1, metric, value2 = between_match.groups()
|
|
75
|
+
|
|
76
|
+
if metric not in cls.METRIC_MAP:
|
|
77
|
+
raise ConditionParseError(
|
|
78
|
+
f"Unknown metric: {metric}. "
|
|
79
|
+
f"Available metrics: {', '.join(sorted(cls.METRIC_MAP.keys()))}"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
return Condition(
|
|
83
|
+
metric=cls.METRIC_MAP[metric],
|
|
84
|
+
operator=Operator.BETWEEN,
|
|
85
|
+
value=float(value1),
|
|
86
|
+
value2=float(value2)
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# Try standard pattern
|
|
90
|
+
match = cls.PATTERN.match(condition_str)
|
|
91
|
+
if not match:
|
|
92
|
+
raise ConditionParseError(
|
|
93
|
+
f"Invalid condition format: '{condition_str}'. "
|
|
94
|
+
f"Expected format: 'metric operator value' (e.g., 'pass_rate >= 0.85')"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
metric_str, operator_str, value_str = match.groups()
|
|
98
|
+
|
|
99
|
+
if metric_str not in cls.METRIC_MAP:
|
|
100
|
+
raise ConditionParseError(
|
|
101
|
+
f"Unknown metric: {metric_str}. "
|
|
102
|
+
f"Available metrics: {', '.join(sorted(cls.METRIC_MAP.keys()))}"
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
return Condition(
|
|
106
|
+
metric=cls.METRIC_MAP[metric_str],
|
|
107
|
+
operator=cls.OPERATOR_MAP[operator_str],
|
|
108
|
+
value=float(value_str)
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
@classmethod
|
|
112
|
+
def parse_many(cls, conditions: List[str]) -> List[Condition]:
|
|
113
|
+
"""Parse multiple condition strings.
|
|
114
|
+
|
|
115
|
+
Args:
|
|
116
|
+
conditions: List of condition strings to parse.
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
List of Condition objects.
|
|
120
|
+
"""
|
|
121
|
+
return [cls.parse(c) for c in conditions]
|
|
122
|
+
|
|
123
|
+
@classmethod
|
|
124
|
+
def get_available_metrics(cls) -> List[str]:
|
|
125
|
+
"""Get list of available metric names."""
|
|
126
|
+
return sorted(cls.METRIC_MAP.keys())
|
|
127
|
+
|
|
128
|
+
@classmethod
|
|
129
|
+
def get_available_operators(cls) -> List[str]:
|
|
130
|
+
"""Get list of available operators."""
|
|
131
|
+
return list(cls.OPERATOR_MAP.keys()) + ['between']
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""Reporter for displaying assertion results."""
|
|
2
|
+
|
|
3
|
+
from typing import Dict, Any
|
|
4
|
+
import xml.etree.ElementTree as ET
|
|
5
|
+
from xml.dom import minidom
|
|
6
|
+
|
|
7
|
+
from rich.console import Console
|
|
8
|
+
from rich.table import Table
|
|
9
|
+
from rich.panel import Panel
|
|
10
|
+
from rich.text import Text
|
|
11
|
+
|
|
12
|
+
from .evaluator import AssertionReport, AssertionResult
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class AssertionReporter:
|
|
16
|
+
"""Display assertion results in various formats."""
|
|
17
|
+
|
|
18
|
+
def __init__(self, console: Console):
|
|
19
|
+
"""Initialize the reporter.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
console: Rich Console instance for output.
|
|
23
|
+
"""
|
|
24
|
+
self.console = console
|
|
25
|
+
|
|
26
|
+
def display(self, report: AssertionReport, detailed: bool = True) -> None:
|
|
27
|
+
"""Display assertion report to console.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
report: The assertion report to display.
|
|
31
|
+
detailed: If True, show detailed table of all assertions.
|
|
32
|
+
"""
|
|
33
|
+
if not report.outcomes:
|
|
34
|
+
self.console.print("[dim]No assertions defined[/dim]")
|
|
35
|
+
return
|
|
36
|
+
|
|
37
|
+
# Summary panel
|
|
38
|
+
summary = self._build_summary(report)
|
|
39
|
+
self.console.print(Panel(summary, title="Assertion Results", border_style="blue"))
|
|
40
|
+
|
|
41
|
+
# Detailed table
|
|
42
|
+
if detailed:
|
|
43
|
+
table = self._build_table(report)
|
|
44
|
+
self.console.print(table)
|
|
45
|
+
|
|
46
|
+
def _build_summary(self, report: AssertionReport) -> Text:
|
|
47
|
+
"""Build summary text.
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
report: The assertion report.
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
Rich Text object with formatted summary.
|
|
54
|
+
"""
|
|
55
|
+
text = Text()
|
|
56
|
+
text.append(f"Total: {report.total_assertions} ")
|
|
57
|
+
text.append(f"Passed: {report.passed}", style="green")
|
|
58
|
+
text.append(" ")
|
|
59
|
+
if report.failed > 0:
|
|
60
|
+
text.append(f"Failed: {report.failed}", style="red bold")
|
|
61
|
+
else:
|
|
62
|
+
text.append(f"Failed: {report.failed}", style="dim")
|
|
63
|
+
text.append(" ")
|
|
64
|
+
if report.warnings > 0:
|
|
65
|
+
text.append(f"Warnings: {report.warnings}", style="yellow")
|
|
66
|
+
else:
|
|
67
|
+
text.append(f"Warnings: {report.warnings}", style="dim")
|
|
68
|
+
if report.skipped > 0:
|
|
69
|
+
text.append(" ")
|
|
70
|
+
text.append(f"Skipped: {report.skipped}", style="dim")
|
|
71
|
+
return text
|
|
72
|
+
|
|
73
|
+
def _build_table(self, report: AssertionReport) -> Table:
|
|
74
|
+
"""Build detailed results table.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
report: The assertion report.
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
Rich Table with assertion details.
|
|
81
|
+
"""
|
|
82
|
+
table = Table(show_header=True, header_style="bold")
|
|
83
|
+
table.add_column("Template", style="cyan")
|
|
84
|
+
table.add_column("Condition")
|
|
85
|
+
table.add_column("Expected")
|
|
86
|
+
table.add_column("Actual", justify="right")
|
|
87
|
+
table.add_column("Result")
|
|
88
|
+
|
|
89
|
+
result_style = {
|
|
90
|
+
AssertionResult.PASSED: "[green]✓ PASS[/green]",
|
|
91
|
+
AssertionResult.FAILED: "[red]✗ FAIL[/red]",
|
|
92
|
+
AssertionResult.WARNING: "[yellow]⚠ WARN[/yellow]",
|
|
93
|
+
AssertionResult.SKIPPED: "[dim]○ SKIP[/dim]",
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
for outcome in report.outcomes:
|
|
97
|
+
table.add_row(
|
|
98
|
+
outcome.template or "[global]",
|
|
99
|
+
outcome.condition,
|
|
100
|
+
outcome.expected,
|
|
101
|
+
f"{outcome.actual:.4f}",
|
|
102
|
+
result_style[outcome.result]
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
return table
|
|
106
|
+
|
|
107
|
+
def to_json(self, report: AssertionReport) -> Dict[str, Any]:
|
|
108
|
+
"""Convert report to JSON-serializable dict.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
report: The assertion report.
|
|
112
|
+
|
|
113
|
+
Returns:
|
|
114
|
+
Dictionary suitable for JSON serialization.
|
|
115
|
+
"""
|
|
116
|
+
return {
|
|
117
|
+
"summary": {
|
|
118
|
+
"total": report.total_assertions,
|
|
119
|
+
"passed": report.passed,
|
|
120
|
+
"failed": report.failed,
|
|
121
|
+
"warnings": report.warnings,
|
|
122
|
+
"skipped": report.skipped,
|
|
123
|
+
"all_passed": report.all_passed,
|
|
124
|
+
},
|
|
125
|
+
"assertions": [
|
|
126
|
+
{
|
|
127
|
+
"template": o.template,
|
|
128
|
+
"condition": o.condition,
|
|
129
|
+
"expected": o.expected,
|
|
130
|
+
"actual": o.actual,
|
|
131
|
+
"result": o.result.value,
|
|
132
|
+
"message": o.message,
|
|
133
|
+
}
|
|
134
|
+
for o in report.outcomes
|
|
135
|
+
]
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
def to_junit(self, report: AssertionReport) -> str:
|
|
139
|
+
"""Convert report to JUnit XML format for CI/CD integration.
|
|
140
|
+
|
|
141
|
+
Args:
|
|
142
|
+
report: The assertion report.
|
|
143
|
+
|
|
144
|
+
Returns:
|
|
145
|
+
JUnit XML string.
|
|
146
|
+
"""
|
|
147
|
+
testsuites = ET.Element("testsuites")
|
|
148
|
+
testsuites.set("name", "Assertions")
|
|
149
|
+
testsuites.set("tests", str(report.total_assertions))
|
|
150
|
+
testsuites.set("failures", str(report.failed))
|
|
151
|
+
|
|
152
|
+
testsuite = ET.SubElement(testsuites, "testsuite")
|
|
153
|
+
testsuite.set("name", "Evaluation Assertions")
|
|
154
|
+
testsuite.set("tests", str(report.total_assertions))
|
|
155
|
+
testsuite.set("failures", str(report.failed))
|
|
156
|
+
|
|
157
|
+
for outcome in report.outcomes:
|
|
158
|
+
testcase = ET.SubElement(testsuite, "testcase")
|
|
159
|
+
testcase.set("name", f"{outcome.template or 'global'}: {outcome.condition}")
|
|
160
|
+
testcase.set("classname", "fi.assertions")
|
|
161
|
+
|
|
162
|
+
if outcome.result == AssertionResult.FAILED:
|
|
163
|
+
failure = ET.SubElement(testcase, "failure")
|
|
164
|
+
failure.set("message", outcome.message)
|
|
165
|
+
failure.text = f"Expected: {outcome.expected}, Actual: {outcome.actual}"
|
|
166
|
+
elif outcome.result == AssertionResult.SKIPPED:
|
|
167
|
+
ET.SubElement(testcase, "skipped")
|
|
168
|
+
|
|
169
|
+
xml_str = ET.tostring(testsuites, encoding="unicode")
|
|
170
|
+
dom = minidom.parseString(xml_str)
|
|
171
|
+
pretty_xml = dom.toprettyxml(indent=" ")
|
|
172
|
+
|
|
173
|
+
# Remove extra blank lines
|
|
174
|
+
lines = [line for line in pretty_xml.split("\n") if line.strip()]
|
|
175
|
+
return "\n".join(lines)
|
|
176
|
+
|
|
177
|
+
def display_summary_line(self, report: AssertionReport) -> None:
|
|
178
|
+
"""Display a single-line summary suitable for end of run output.
|
|
179
|
+
|
|
180
|
+
Args:
|
|
181
|
+
report: The assertion report.
|
|
182
|
+
"""
|
|
183
|
+
if report.failed > 0:
|
|
184
|
+
self.console.print(
|
|
185
|
+
f"\n[red]✗ {report.failed} assertion(s) failed[/red]"
|
|
186
|
+
)
|
|
187
|
+
elif report.warnings > 0:
|
|
188
|
+
self.console.print(
|
|
189
|
+
f"\n[yellow]⚠ {report.passed} assertion(s) passed with {report.warnings} warning(s)[/yellow]"
|
|
190
|
+
)
|
|
191
|
+
else:
|
|
192
|
+
self.console.print(
|
|
193
|
+
f"\n[green]✓ All {report.passed} assertion(s) passed[/green]"
|
|
194
|
+
)
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""CLI Commands Package"""
|
|
2
|
+
|
|
3
|
+
from fi.cli.commands.init import init_project
|
|
4
|
+
from fi.cli.commands.run import run
|
|
5
|
+
from fi.cli.commands.list_cmd import list_resources
|
|
6
|
+
from fi.cli.commands.validate import validate
|
|
7
|
+
from fi.cli.commands.config import config_app
|
|
8
|
+
|
|
9
|
+
__all__ = ["init_project", "run", "list_resources", "validate", "config_app"]
|