agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
"""Type definitions for Streaming Evaluation.
|
|
2
|
+
|
|
3
|
+
Provides types for evaluating LLM outputs in real-time as tokens are generated.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from enum import Enum
|
|
8
|
+
from typing import Any, Dict, List, Optional, Callable
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class EarlyStopReason(Enum):
|
|
13
|
+
"""Reasons for early stopping during streaming evaluation."""
|
|
14
|
+
|
|
15
|
+
NONE = "none" # No early stop
|
|
16
|
+
TOXICITY = "toxicity" # Toxic content detected
|
|
17
|
+
SAFETY = "safety" # Safety violation detected
|
|
18
|
+
PII = "pii" # PII detected
|
|
19
|
+
JAILBREAK = "jailbreak" # Jailbreak attempt detected
|
|
20
|
+
MAX_TOKENS = "max_tokens" # Maximum token limit reached
|
|
21
|
+
MAX_CHARS = "max_chars" # Maximum character limit reached
|
|
22
|
+
THRESHOLD = "threshold" # Score dropped below threshold
|
|
23
|
+
CUSTOM = "custom" # Custom stop condition triggered
|
|
24
|
+
TIMEOUT = "timeout" # Evaluation timeout
|
|
25
|
+
ERROR = "error" # Evaluation error
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class StreamingState(Enum):
|
|
29
|
+
"""State of the streaming evaluation."""
|
|
30
|
+
|
|
31
|
+
IDLE = "idle" # Not started
|
|
32
|
+
STREAMING = "streaming" # Actively processing chunks
|
|
33
|
+
PAUSED = "paused" # Temporarily paused
|
|
34
|
+
STOPPED = "stopped" # Early stopped
|
|
35
|
+
COMPLETED = "completed" # Finished normally
|
|
36
|
+
ERROR = "error" # Error occurred
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class ChunkResult:
|
|
41
|
+
"""Result of evaluating a single chunk."""
|
|
42
|
+
|
|
43
|
+
chunk_index: int
|
|
44
|
+
chunk_text: str
|
|
45
|
+
cumulative_text: str
|
|
46
|
+
scores: Dict[str, float] # eval_name -> score
|
|
47
|
+
flags: Dict[str, bool] # eval_name -> passed
|
|
48
|
+
should_stop: bool = False
|
|
49
|
+
stop_reason: EarlyStopReason = EarlyStopReason.NONE
|
|
50
|
+
latency_ms: float = 0.0
|
|
51
|
+
timestamp: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
|
52
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def all_passed(self) -> bool:
|
|
56
|
+
"""Check if all evaluations passed for this chunk."""
|
|
57
|
+
return all(self.flags.values()) if self.flags else True
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def min_score(self) -> float:
|
|
61
|
+
"""Get the minimum score across all evaluations."""
|
|
62
|
+
return min(self.scores.values()) if self.scores else 1.0
|
|
63
|
+
|
|
64
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
65
|
+
"""Convert to dictionary."""
|
|
66
|
+
return {
|
|
67
|
+
"chunk_index": self.chunk_index,
|
|
68
|
+
"chunk_text": self.chunk_text,
|
|
69
|
+
"scores": self.scores,
|
|
70
|
+
"flags": self.flags,
|
|
71
|
+
"should_stop": self.should_stop,
|
|
72
|
+
"stop_reason": self.stop_reason.value,
|
|
73
|
+
"latency_ms": self.latency_ms,
|
|
74
|
+
"timestamp": self.timestamp.isoformat(),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
@dataclass
|
|
79
|
+
class StreamingEvalResult:
|
|
80
|
+
"""Final result of streaming evaluation."""
|
|
81
|
+
|
|
82
|
+
passed: bool
|
|
83
|
+
final_text: str
|
|
84
|
+
total_chunks: int
|
|
85
|
+
chunk_results: List[ChunkResult]
|
|
86
|
+
final_scores: Dict[str, float] # eval_name -> final score
|
|
87
|
+
early_stopped: bool = False
|
|
88
|
+
stop_reason: EarlyStopReason = EarlyStopReason.NONE
|
|
89
|
+
stopped_at_chunk: Optional[int] = None
|
|
90
|
+
total_latency_ms: float = 0.0
|
|
91
|
+
state: StreamingState = StreamingState.COMPLETED
|
|
92
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
93
|
+
|
|
94
|
+
@property
|
|
95
|
+
def average_chunk_latency_ms(self) -> float:
|
|
96
|
+
"""Average latency per chunk."""
|
|
97
|
+
if not self.chunk_results:
|
|
98
|
+
return 0.0
|
|
99
|
+
return sum(c.latency_ms for c in self.chunk_results) / len(self.chunk_results)
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def min_score_history(self) -> List[float]:
|
|
103
|
+
"""Get minimum score at each chunk for plotting."""
|
|
104
|
+
return [c.min_score for c in self.chunk_results]
|
|
105
|
+
|
|
106
|
+
@property
|
|
107
|
+
def score_by_eval(self) -> Dict[str, List[float]]:
|
|
108
|
+
"""Get score history by evaluation name."""
|
|
109
|
+
result: Dict[str, List[float]] = {}
|
|
110
|
+
for chunk in self.chunk_results:
|
|
111
|
+
for eval_name, score in chunk.scores.items():
|
|
112
|
+
if eval_name not in result:
|
|
113
|
+
result[eval_name] = []
|
|
114
|
+
result[eval_name].append(score)
|
|
115
|
+
return result
|
|
116
|
+
|
|
117
|
+
def summary(self) -> str:
|
|
118
|
+
"""Generate human-readable summary."""
|
|
119
|
+
lines = [
|
|
120
|
+
f"Streaming Evaluation: {'PASSED' if self.passed else 'FAILED'}",
|
|
121
|
+
f" Total Chunks: {self.total_chunks}",
|
|
122
|
+
f" Final Text Length: {len(self.final_text)} chars",
|
|
123
|
+
f" Total Latency: {self.total_latency_ms:.2f}ms",
|
|
124
|
+
f" Avg Chunk Latency: {self.average_chunk_latency_ms:.2f}ms",
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
if self.early_stopped:
|
|
128
|
+
lines.append(f" Early Stopped: Yes (at chunk {self.stopped_at_chunk})")
|
|
129
|
+
lines.append(f" Stop Reason: {self.stop_reason.value}")
|
|
130
|
+
|
|
131
|
+
lines.append(" Final Scores:")
|
|
132
|
+
for eval_name, score in self.final_scores.items():
|
|
133
|
+
lines.append(f" {eval_name}: {score:.3f}")
|
|
134
|
+
|
|
135
|
+
return "\n".join(lines)
|
|
136
|
+
|
|
137
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
138
|
+
"""Convert to dictionary."""
|
|
139
|
+
return {
|
|
140
|
+
"passed": self.passed,
|
|
141
|
+
"final_text": self.final_text,
|
|
142
|
+
"total_chunks": self.total_chunks,
|
|
143
|
+
"chunk_results": [c.to_dict() for c in self.chunk_results],
|
|
144
|
+
"final_scores": self.final_scores,
|
|
145
|
+
"early_stopped": self.early_stopped,
|
|
146
|
+
"stop_reason": self.stop_reason.value,
|
|
147
|
+
"stopped_at_chunk": self.stopped_at_chunk,
|
|
148
|
+
"total_latency_ms": self.total_latency_ms,
|
|
149
|
+
"state": self.state.value,
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@dataclass
|
|
154
|
+
class StreamingConfig:
|
|
155
|
+
"""Configuration for streaming evaluation."""
|
|
156
|
+
|
|
157
|
+
# Chunk processing
|
|
158
|
+
min_chunk_size: int = 1 # Minimum characters to trigger evaluation
|
|
159
|
+
max_chunk_size: int = 100 # Maximum characters per chunk before forced eval
|
|
160
|
+
eval_interval_ms: int = 100 # Minimum ms between evaluations
|
|
161
|
+
|
|
162
|
+
# Token limits
|
|
163
|
+
max_tokens: Optional[int] = None # Maximum tokens before stopping
|
|
164
|
+
max_chars: Optional[int] = None # Maximum characters before stopping
|
|
165
|
+
|
|
166
|
+
# Timeouts
|
|
167
|
+
chunk_timeout_ms: int = 5000 # Timeout for single chunk evaluation
|
|
168
|
+
total_timeout_ms: int = 60000 # Total timeout for streaming
|
|
169
|
+
|
|
170
|
+
# Early stopping
|
|
171
|
+
enable_early_stop: bool = True
|
|
172
|
+
stop_on_first_failure: bool = False # Stop immediately on any failure
|
|
173
|
+
|
|
174
|
+
# Thresholds for early stopping
|
|
175
|
+
toxicity_threshold: float = 0.7 # Stop if toxicity score exceeds this
|
|
176
|
+
safety_threshold: float = 0.3 # Stop if safety score drops below this
|
|
177
|
+
|
|
178
|
+
# Evaluation frequency
|
|
179
|
+
eval_every_n_chunks: int = 1 # Run evaluation every N chunks
|
|
180
|
+
eval_on_sentence_end: bool = True # Also evaluate at sentence boundaries
|
|
181
|
+
|
|
182
|
+
# Callbacks
|
|
183
|
+
on_chunk_callback: Optional[Callable[[ChunkResult], None]] = None
|
|
184
|
+
on_stop_callback: Optional[Callable[[EarlyStopReason, str], None]] = None
|
|
185
|
+
|
|
186
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
187
|
+
"""Convert to dictionary (excluding callbacks)."""
|
|
188
|
+
return {
|
|
189
|
+
"min_chunk_size": self.min_chunk_size,
|
|
190
|
+
"max_chunk_size": self.max_chunk_size,
|
|
191
|
+
"eval_interval_ms": self.eval_interval_ms,
|
|
192
|
+
"max_tokens": self.max_tokens,
|
|
193
|
+
"max_chars": self.max_chars,
|
|
194
|
+
"chunk_timeout_ms": self.chunk_timeout_ms,
|
|
195
|
+
"total_timeout_ms": self.total_timeout_ms,
|
|
196
|
+
"enable_early_stop": self.enable_early_stop,
|
|
197
|
+
"stop_on_first_failure": self.stop_on_first_failure,
|
|
198
|
+
"toxicity_threshold": self.toxicity_threshold,
|
|
199
|
+
"safety_threshold": self.safety_threshold,
|
|
200
|
+
"eval_every_n_chunks": self.eval_every_n_chunks,
|
|
201
|
+
"eval_on_sentence_end": self.eval_on_sentence_end,
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@dataclass
|
|
206
|
+
class EarlyStopCondition:
|
|
207
|
+
"""Defines a condition for early stopping."""
|
|
208
|
+
|
|
209
|
+
name: str
|
|
210
|
+
eval_name: str # Which evaluation to check
|
|
211
|
+
threshold: float # Threshold value
|
|
212
|
+
comparison: str = "below" # "below" or "above"
|
|
213
|
+
consecutive_chunks: int = 1 # How many consecutive chunks must fail
|
|
214
|
+
enabled: bool = True
|
|
215
|
+
|
|
216
|
+
def check(self, score: float, consecutive_count: int) -> bool:
|
|
217
|
+
"""Check if this condition triggers early stop."""
|
|
218
|
+
if not self.enabled:
|
|
219
|
+
return False
|
|
220
|
+
|
|
221
|
+
if consecutive_count < self.consecutive_chunks:
|
|
222
|
+
return False
|
|
223
|
+
|
|
224
|
+
if self.comparison == "below":
|
|
225
|
+
return score < self.threshold
|
|
226
|
+
else: # above
|
|
227
|
+
return score > self.threshold
|
|
228
|
+
|
|
229
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
230
|
+
"""Convert to dictionary."""
|
|
231
|
+
return {
|
|
232
|
+
"name": self.name,
|
|
233
|
+
"eval_name": self.eval_name,
|
|
234
|
+
"threshold": self.threshold,
|
|
235
|
+
"comparison": self.comparison,
|
|
236
|
+
"consecutive_chunks": self.consecutive_chunks,
|
|
237
|
+
"enabled": self.enabled,
|
|
238
|
+
}
|
fi/evals/templates.py
ADDED
|
@@ -0,0 +1,472 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Template class shells for cloud evals.
|
|
3
|
+
|
|
4
|
+
These classes exist purely for IDE autocomplete and backward compatibility
|
|
5
|
+
with code that does ``from fi.evals import Toxicity``. Input schemas are
|
|
6
|
+
no longer hardcoded here — the cloud registry (``fi.evals.core.cloud_registry``)
|
|
7
|
+
fetches ``required_keys`` from the api and maps user inputs dynamically.
|
|
8
|
+
|
|
9
|
+
To call an eval by name without importing a class, just pass the string::
|
|
10
|
+
|
|
11
|
+
evaluate("customer_agent_query_handling", conversation=[...], model="turing_flash")
|
|
12
|
+
|
|
13
|
+
Renamed templates have a module-level alias from the old name to the new
|
|
14
|
+
class (e.g. ``NoOpenAIReference = NoLLMReference``) so existing user code
|
|
15
|
+
keeps working seamlessly. Templates that were removed outright are gone —
|
|
16
|
+
an ``ImportError`` is a clearer signal than a silent runtime 400.
|
|
17
|
+
"""
|
|
18
|
+
from typing import Any, Dict, List, Optional
|
|
19
|
+
|
|
20
|
+
from fi.utils.errors import MissingRequiredConfigForEvalTemplate
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
# EvalTemplate base class
|
|
25
|
+
# ---------------------------------------------------------------------------
|
|
26
|
+
|
|
27
|
+
class EvalTemplate:
|
|
28
|
+
"""Lightweight marker for cloud eval templates.
|
|
29
|
+
|
|
30
|
+
Input filtering and validation happen dynamically via the cloud
|
|
31
|
+
registry — no hardcoded Input schema.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
eval_id: str = ""
|
|
35
|
+
eval_name: str = ""
|
|
36
|
+
description: str = ""
|
|
37
|
+
eval_tags: List[str] = []
|
|
38
|
+
required_keys: List[str] = []
|
|
39
|
+
output: str = ""
|
|
40
|
+
eval_type_id: str = ""
|
|
41
|
+
config_schema: Dict[str, Any] = {}
|
|
42
|
+
criteria: str = ""
|
|
43
|
+
choices: List[str] = []
|
|
44
|
+
multi_choice: bool = False
|
|
45
|
+
|
|
46
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None) -> None:
|
|
47
|
+
self.config = config or {}
|
|
48
|
+
|
|
49
|
+
def __repr__(self) -> str:
|
|
50
|
+
return f"EvalTemplate(name={self.eval_name})"
|
|
51
|
+
|
|
52
|
+
def validate_config(self, config: Dict[str, Any]) -> None:
|
|
53
|
+
for key in self.config_schema:
|
|
54
|
+
if key not in config:
|
|
55
|
+
raise MissingRequiredConfigForEvalTemplate(key, self.eval_name)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# ---------------------------------------------------------------------------
|
|
59
|
+
# Conversation
|
|
60
|
+
# ---------------------------------------------------------------------------
|
|
61
|
+
|
|
62
|
+
class ConversationCoherence(EvalTemplate):
|
|
63
|
+
eval_name = "conversation_coherence"
|
|
64
|
+
eval_id = "1"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class ConversationResolution(EvalTemplate):
|
|
68
|
+
eval_name = "conversation_resolution"
|
|
69
|
+
eval_id = "2"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
# Content moderation & safety
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
class ContentModeration(EvalTemplate):
|
|
77
|
+
eval_name = "content_moderation"
|
|
78
|
+
eval_id = "4"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class PII(EvalTemplate):
|
|
82
|
+
eval_name = "pii"
|
|
83
|
+
eval_id = "14"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class Toxicity(EvalTemplate):
|
|
87
|
+
eval_name = "toxicity"
|
|
88
|
+
eval_id = "15"
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class Sexist(EvalTemplate):
|
|
92
|
+
eval_name = "sexist"
|
|
93
|
+
eval_id = "17"
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class PromptInjection(EvalTemplate):
|
|
97
|
+
eval_name = "prompt_injection"
|
|
98
|
+
eval_id = "18"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class DataPrivacyCompliance(EvalTemplate):
|
|
102
|
+
eval_name = "data_privacy_compliance"
|
|
103
|
+
eval_id = "22"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class NoRacialBias(EvalTemplate):
|
|
107
|
+
eval_name = "no_racial_bias"
|
|
108
|
+
eval_id = "77"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class NoGenderBias(EvalTemplate):
|
|
112
|
+
eval_name = "no_gender_bias"
|
|
113
|
+
eval_id = "78"
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class NoAgeBias(EvalTemplate):
|
|
117
|
+
eval_name = "no_age_bias"
|
|
118
|
+
eval_id = "79"
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class NoLLMReference(EvalTemplate):
|
|
122
|
+
"""Checks that the output doesn't reference the underlying LLM (e.g. 'I am GPT…').
|
|
123
|
+
|
|
124
|
+
Replaced the old ``NoOpenAIReference`` template in the api revamp.
|
|
125
|
+
"""
|
|
126
|
+
|
|
127
|
+
eval_name = "no_llm_reference"
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# Alias: old class name → current eval. Import-level backward compatibility.
|
|
131
|
+
NoOpenAIReference = NoLLMReference
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class NoApologies(EvalTemplate):
|
|
135
|
+
eval_name = "no_apologies"
|
|
136
|
+
eval_id = "81"
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class ContentSafety(EvalTemplate):
|
|
140
|
+
eval_name = "content_safety_violation"
|
|
141
|
+
eval_id = "93"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
class NoHarmfulTherapeuticGuidance(EvalTemplate):
|
|
145
|
+
eval_name = "no_harmful_therapeutic_guidance"
|
|
146
|
+
eval_id = "90"
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class ClinicallyInappropriateTone(EvalTemplate):
|
|
150
|
+
eval_name = "clinically_inappropriate_tone"
|
|
151
|
+
eval_id = "91"
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class IsHarmfulAdvice(EvalTemplate):
|
|
155
|
+
eval_name = "is_harmful_advice"
|
|
156
|
+
eval_id = "92"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# ---------------------------------------------------------------------------
|
|
160
|
+
# RAG / Grounding
|
|
161
|
+
# ---------------------------------------------------------------------------
|
|
162
|
+
|
|
163
|
+
class ContextAdherence(EvalTemplate):
|
|
164
|
+
eval_name = "context_adherence"
|
|
165
|
+
eval_id = "5"
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
class ContextRelevance(EvalTemplate):
|
|
169
|
+
eval_name = "context_relevance"
|
|
170
|
+
eval_id = "9"
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
class Completeness(EvalTemplate):
|
|
174
|
+
eval_name = "completeness"
|
|
175
|
+
eval_id = "10"
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
class ChunkAttribution(EvalTemplate):
|
|
179
|
+
eval_name = "chunk_attribution"
|
|
180
|
+
eval_id = "11"
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
class ChunkUtilization(EvalTemplate):
|
|
184
|
+
eval_name = "chunk_utilization"
|
|
185
|
+
eval_id = "12"
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
class Groundedness(EvalTemplate):
|
|
189
|
+
eval_name = "groundedness"
|
|
190
|
+
eval_id = "47"
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
class FactualAccuracy(EvalTemplate):
|
|
194
|
+
eval_name = "factual_accuracy"
|
|
195
|
+
eval_id = "66"
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
class DetectHallucination(EvalTemplate):
|
|
199
|
+
"""Detects hallucinated or unsupported claims in the output.
|
|
200
|
+
|
|
201
|
+
Replaced the old ``DetectHallucinationMissingInfo`` template in the revamp.
|
|
202
|
+
"""
|
|
203
|
+
|
|
204
|
+
eval_name = "detect_hallucination"
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
# Alias: old class name → current eval.
|
|
208
|
+
DetectHallucinationMissingInfo = DetectHallucination
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
class IsFactuallyConsistent(EvalTemplate):
|
|
212
|
+
eval_name = "is_factually_consistent"
|
|
213
|
+
eval_id = "95"
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
# ---------------------------------------------------------------------------
|
|
217
|
+
# Tone & quality
|
|
218
|
+
# ---------------------------------------------------------------------------
|
|
219
|
+
|
|
220
|
+
class Tone(EvalTemplate):
|
|
221
|
+
eval_name = "tone"
|
|
222
|
+
eval_id = "16"
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class PromptAdherence(EvalTemplate):
|
|
226
|
+
eval_name = "prompt_adherence"
|
|
227
|
+
eval_id = "65"
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
class IsPolite(EvalTemplate):
|
|
231
|
+
eval_name = "is_polite"
|
|
232
|
+
eval_id = "82"
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
class IsConcise(EvalTemplate):
|
|
236
|
+
eval_name = "is_concise"
|
|
237
|
+
eval_id = "83"
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class IsHelpful(EvalTemplate):
|
|
241
|
+
eval_name = "is_helpful"
|
|
242
|
+
eval_id = "84"
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
class IsInformalTone(EvalTemplate):
|
|
246
|
+
eval_name = "is_informal_tone"
|
|
247
|
+
eval_id = "97"
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
class AnswerRefusal(EvalTemplate):
|
|
251
|
+
eval_name = "answer_refusal"
|
|
252
|
+
eval_id = "88"
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
class TaskCompletion(EvalTemplate):
|
|
256
|
+
eval_name = "task_completion"
|
|
257
|
+
eval_id = "99"
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
# ---------------------------------------------------------------------------
|
|
261
|
+
# Format / structure
|
|
262
|
+
# ---------------------------------------------------------------------------
|
|
263
|
+
|
|
264
|
+
class IsJson(EvalTemplate):
|
|
265
|
+
eval_name = "is_json"
|
|
266
|
+
eval_id = "23"
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
class OneLine(EvalTemplate):
|
|
270
|
+
eval_name = "one_line"
|
|
271
|
+
eval_id = "38"
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
class ContainsValidLink(EvalTemplate):
|
|
275
|
+
eval_name = "contains_valid_link"
|
|
276
|
+
eval_id = "39"
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
class IsEmail(EvalTemplate):
|
|
280
|
+
eval_name = "is_email"
|
|
281
|
+
eval_id = "40"
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
class ContainsCode(EvalTemplate):
|
|
285
|
+
"""New in the api revamp."""
|
|
286
|
+
|
|
287
|
+
eval_name = "contains_code"
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
# ---------------------------------------------------------------------------
|
|
291
|
+
# Comparison / matching
|
|
292
|
+
# ---------------------------------------------------------------------------
|
|
293
|
+
|
|
294
|
+
class Ranking(EvalTemplate):
|
|
295
|
+
eval_name = "eval_ranking"
|
|
296
|
+
eval_id = "61"
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
class SummaryQuality(EvalTemplate):
|
|
300
|
+
eval_name = "summary_quality"
|
|
301
|
+
eval_id = "64"
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
class IsGoodSummary(EvalTemplate):
|
|
305
|
+
eval_name = "is_good_summary"
|
|
306
|
+
eval_id = "94"
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
class FuzzyMatch(EvalTemplate):
|
|
310
|
+
eval_name = "fuzzy_match"
|
|
311
|
+
eval_id = "87"
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
class BleuScore(EvalTemplate):
|
|
315
|
+
eval_name = "bleu_score"
|
|
316
|
+
eval_id = "101"
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
class TranslationAccuracy(EvalTemplate):
|
|
320
|
+
eval_name = "translation_accuracy"
|
|
321
|
+
eval_id = "67"
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
class CulturalSensitivity(EvalTemplate):
|
|
325
|
+
eval_name = "cultural_sensitivity"
|
|
326
|
+
eval_id = "68"
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
class BiasDetection(EvalTemplate):
|
|
330
|
+
eval_name = "bias_detection"
|
|
331
|
+
eval_id = "69"
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
class GroundTruthMatch(EvalTemplate):
|
|
335
|
+
"""New in the api revamp."""
|
|
336
|
+
|
|
337
|
+
eval_name = "ground_truth_match"
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
# ---------------------------------------------------------------------------
|
|
341
|
+
# Function calling
|
|
342
|
+
# ---------------------------------------------------------------------------
|
|
343
|
+
|
|
344
|
+
class EvaluateFunctionCalling(EvalTemplate):
|
|
345
|
+
eval_name = "evaluate_function_calling"
|
|
346
|
+
eval_id = "98"
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
# Alias: old class name → current eval.
|
|
350
|
+
LLMFunctionCalling = EvaluateFunctionCalling
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
# ---------------------------------------------------------------------------
|
|
354
|
+
# Domain / agent / multimodal
|
|
355
|
+
# ---------------------------------------------------------------------------
|
|
356
|
+
|
|
357
|
+
class TextToSQL(EvalTemplate):
|
|
358
|
+
"""New in the api revamp."""
|
|
359
|
+
|
|
360
|
+
eval_name = "text_to_sql"
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
class PromptInstructionAdherence(EvalTemplate):
|
|
364
|
+
"""New in the api revamp."""
|
|
365
|
+
|
|
366
|
+
eval_name = "prompt_instruction_adherence"
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
class ProtectFlash(EvalTemplate):
|
|
370
|
+
"""New in the api revamp — lightweight prompt-injection check."""
|
|
371
|
+
|
|
372
|
+
eval_name = "protect_flash"
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
class ImageInstructionAdherence(EvalTemplate):
|
|
376
|
+
"""New in the api revamp."""
|
|
377
|
+
|
|
378
|
+
eval_name = "image_instruction_adherence"
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
class SyntheticImageEvaluator(EvalTemplate):
|
|
382
|
+
"""New in the api revamp."""
|
|
383
|
+
|
|
384
|
+
eval_name = "synthetic_image_evaluator"
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
class OCREvaluation(EvalTemplate):
|
|
388
|
+
"""New in the api revamp."""
|
|
389
|
+
|
|
390
|
+
eval_name = "ocr_evaluation"
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
class ASRAccuracy(EvalTemplate):
|
|
394
|
+
"""Speech-to-text accuracy eval.
|
|
395
|
+
|
|
396
|
+
Replaced the old ``AudioTranscriptionEvaluator`` in the revamp.
|
|
397
|
+
"""
|
|
398
|
+
|
|
399
|
+
eval_name = "ASR/STT_accuracy"
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
# Alias: old class name → current eval.
|
|
403
|
+
AudioTranscriptionEvaluator = ASRAccuracy
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
class TTSAccuracy(EvalTemplate):
|
|
407
|
+
"""New in the api revamp."""
|
|
408
|
+
|
|
409
|
+
eval_name = "TTS_accuracy"
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
class AudioQualityEvaluator(EvalTemplate):
|
|
413
|
+
eval_name = "audio_quality"
|
|
414
|
+
eval_id = "75"
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
class CaptionHallucination(EvalTemplate):
|
|
418
|
+
eval_name = "caption_hallucination"
|
|
419
|
+
eval_id = "100"
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
class IsCompliant(EvalTemplate):
|
|
423
|
+
eval_name = "is_compliant"
|
|
424
|
+
eval_id = "96"
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
# ---------------------------------------------------------------------------
|
|
428
|
+
# Customer-agent family (new in revamp)
|
|
429
|
+
# ---------------------------------------------------------------------------
|
|
430
|
+
|
|
431
|
+
class CustomerAgentClarificationSeeking(EvalTemplate):
|
|
432
|
+
eval_name = "customer_agent_clarification_seeking"
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
class CustomerAgentContextRetention(EvalTemplate):
|
|
436
|
+
eval_name = "customer_agent_context_retention"
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
class CustomerAgentConversationQuality(EvalTemplate):
|
|
440
|
+
eval_name = "customer_agent_conversation_quality"
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
class CustomerAgentHumanEscalation(EvalTemplate):
|
|
444
|
+
eval_name = "customer_agent_human_escalation"
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
class CustomerAgentInterruptionHandling(EvalTemplate):
|
|
448
|
+
eval_name = "customer_agent_interruption_handling"
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
class CustomerAgentLanguageHandling(EvalTemplate):
|
|
452
|
+
eval_name = "customer_agent_language_handling"
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
class CustomerAgentLoopDetection(EvalTemplate):
|
|
456
|
+
eval_name = "customer_agent_loop_detection"
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
class CustomerAgentObjectionHandling(EvalTemplate):
|
|
460
|
+
eval_name = "customer_agent_objection_handling"
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
class CustomerAgentPromptConformance(EvalTemplate):
|
|
464
|
+
eval_name = "customer_agent_prompt_conformance"
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
class CustomerAgentQueryHandling(EvalTemplate):
|
|
468
|
+
eval_name = "customer_agent_query_handling"
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
class CustomerAgentTerminationHandling(EvalTemplate):
|
|
472
|
+
eval_name = "customer_agent_termination_handling"
|