agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from typing import Any, Dict, List, Optional
|
|
3
|
+
|
|
4
|
+
from ..base_metric import BaseMetric
|
|
5
|
+
from ...types import TextMetricInput
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Regex(BaseMetric[TextMetricInput]):
|
|
10
|
+
"""Checks if a regex pattern is found in the response text."""
|
|
11
|
+
|
|
12
|
+
@property
|
|
13
|
+
def metric_name(self) -> str:
|
|
14
|
+
return "regex"
|
|
15
|
+
|
|
16
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None) -> None:
|
|
17
|
+
super().__init__(config)
|
|
18
|
+
self.pattern = self.config.get("pattern")
|
|
19
|
+
if not self.pattern:
|
|
20
|
+
raise ValueError("Regex metric requires a 'pattern' in its config.")
|
|
21
|
+
|
|
22
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
23
|
+
match = re.search(self.pattern, inputs.response)
|
|
24
|
+
if match:
|
|
25
|
+
return {
|
|
26
|
+
"output": 1.0, # Using 1.0 for success (True)
|
|
27
|
+
"reason": f"Regex pattern '{self.pattern}' found in response.",
|
|
28
|
+
}
|
|
29
|
+
else:
|
|
30
|
+
return {
|
|
31
|
+
"output": 0.0, # Using 0.0 for failure (False)
|
|
32
|
+
"reason": f"Regex pattern '{self.pattern}' not found in response.",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class Contains(BaseMetric[TextMetricInput]):
|
|
37
|
+
@property
|
|
38
|
+
def metric_name(self) -> str:
|
|
39
|
+
return "contains"
|
|
40
|
+
|
|
41
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
42
|
+
super().__init__(config)
|
|
43
|
+
self.keyword = self.config.get("keyword")
|
|
44
|
+
self.case_sensitive = self.config.get("case_sensitive", False)
|
|
45
|
+
if not self.keyword:
|
|
46
|
+
raise ValueError("Contains metric requires a 'keyword' config.")
|
|
47
|
+
|
|
48
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
49
|
+
text, kw = (
|
|
50
|
+
(inputs.response, self.keyword)
|
|
51
|
+
if self.case_sensitive
|
|
52
|
+
else (inputs.response.lower(), self.keyword.lower())
|
|
53
|
+
)
|
|
54
|
+
is_present = kw in text
|
|
55
|
+
return {
|
|
56
|
+
"output": 1.0 if is_present else 0.0,
|
|
57
|
+
"reason": f"Keyword '{self.keyword}' found"
|
|
58
|
+
if is_present
|
|
59
|
+
else f"Keyword '{self.keyword}' not found",
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class _BaseContainsKeywords(BaseMetric[TextMetricInput]):
|
|
64
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
65
|
+
super().__init__(config)
|
|
66
|
+
self.keywords = self.config.get("keywords")
|
|
67
|
+
self.case_sensitive = self.config.get("case_sensitive", False)
|
|
68
|
+
if not self.keywords or not isinstance(self.keywords, list):
|
|
69
|
+
raise ValueError(
|
|
70
|
+
f"{self.metric_name} metric requires a 'keywords' list in config."
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
def _get_found_keywords(self, text: str) -> List[str]:
|
|
74
|
+
text_to_check = text if self.case_sensitive else text.lower()
|
|
75
|
+
found = [
|
|
76
|
+
kw
|
|
77
|
+
for kw in self.keywords
|
|
78
|
+
if (kw if self.case_sensitive else kw.lower()) in text_to_check
|
|
79
|
+
]
|
|
80
|
+
return found
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class ContainsAll(_BaseContainsKeywords):
|
|
84
|
+
@property
|
|
85
|
+
def metric_name(self) -> str:
|
|
86
|
+
return "contains_all"
|
|
87
|
+
|
|
88
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
89
|
+
found = self._get_found_keywords(inputs.response)
|
|
90
|
+
if len(found) == len(self.keywords):
|
|
91
|
+
return {
|
|
92
|
+
"output": 1.0,
|
|
93
|
+
"reason": f"All {len(self.keywords)} keywords found.",
|
|
94
|
+
}
|
|
95
|
+
missing = [kw for kw in self.keywords if kw not in found]
|
|
96
|
+
return {"output": 0.0, "reason": f"Missing keywords: {', '.join(missing)}"}
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class ContainsAny(_BaseContainsKeywords):
|
|
100
|
+
"""Checks if the response text contains any of the provided keywords."""
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def metric_name(self) -> str:
|
|
104
|
+
return "contains_any"
|
|
105
|
+
|
|
106
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
107
|
+
found_keywords = self._get_found_keywords(inputs.response)
|
|
108
|
+
if found_keywords:
|
|
109
|
+
return {
|
|
110
|
+
"output": 1.0,
|
|
111
|
+
"reason": f"Found keywords: {', '.join(found_keywords)}",
|
|
112
|
+
}
|
|
113
|
+
return {"output": 0.0, "reason": "No keywords found in response."}
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class ContainsNone(_BaseContainsKeywords):
|
|
117
|
+
"""Checks if the response text contains none of the provided keywords."""
|
|
118
|
+
|
|
119
|
+
@property
|
|
120
|
+
def metric_name(self) -> str:
|
|
121
|
+
return "contains_none"
|
|
122
|
+
|
|
123
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
124
|
+
found_keywords = self._get_found_keywords(inputs.response)
|
|
125
|
+
if not found_keywords:
|
|
126
|
+
return {"output": 1.0, "reason": "No forbidden keywords found."}
|
|
127
|
+
return {
|
|
128
|
+
"output": 0.0,
|
|
129
|
+
"reason": f"Found forbidden keywords: {', '.join(found_keywords)}",
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _standardize_url(url: str) -> str:
|
|
134
|
+
if url.startswith("http://") or url.startswith("https://"):
|
|
135
|
+
return url
|
|
136
|
+
return f"http://{url}"
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class OneLine(BaseMetric[TextMetricInput]):
|
|
140
|
+
"""Checks if the text is a single line."""
|
|
141
|
+
|
|
142
|
+
@property
|
|
143
|
+
def metric_name(self) -> str:
|
|
144
|
+
return "one_line"
|
|
145
|
+
|
|
146
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
147
|
+
is_one_line = "\n" not in inputs.response.strip()
|
|
148
|
+
return {
|
|
149
|
+
"output": 1.0 if is_one_line else 0.0,
|
|
150
|
+
"reason": "Response is a single line."
|
|
151
|
+
if is_one_line
|
|
152
|
+
else "Response contains multiple lines.",
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class ContainsEmail(Regex):
|
|
157
|
+
"""Checks if the text contains an email address."""
|
|
158
|
+
|
|
159
|
+
@property
|
|
160
|
+
def metric_name(self) -> str:
|
|
161
|
+
return "contains_email"
|
|
162
|
+
|
|
163
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
164
|
+
# Pass the specific regex pattern to the parent Regex class
|
|
165
|
+
super().__init__({"pattern": r"[a-zA-Z0-9_.+-]+@[a-zA-Z0-9-]+\.[a-zA-Z0-9-.]+"})
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
class IsEmail(Regex):
|
|
169
|
+
"""Checks if the entire text is a valid email address."""
|
|
170
|
+
|
|
171
|
+
@property
|
|
172
|
+
def metric_name(self) -> str:
|
|
173
|
+
return "is_email"
|
|
174
|
+
|
|
175
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
176
|
+
super().__init__(
|
|
177
|
+
{"pattern": r"^[a-zA-Z0-9_.+-]+@[a-zA-Z0-9-]+\.[a-zA-Z0-9-.]+$"}
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
class ContainsLink(Regex):
|
|
182
|
+
"""Checks if the text contains a link."""
|
|
183
|
+
|
|
184
|
+
@property
|
|
185
|
+
def metric_name(self) -> str:
|
|
186
|
+
return "contains_link"
|
|
187
|
+
|
|
188
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
189
|
+
super().__init__({"pattern": r"(?!.*@)(?:https?://)?(?:www\.)?\S+\.\S+"})
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
class ContainsValidLink(BaseMetric[TextMetricInput]):
|
|
193
|
+
"""Checks if the text contains a link that returns a 2xx status code."""
|
|
194
|
+
|
|
195
|
+
@property
|
|
196
|
+
def metric_name(self) -> str:
|
|
197
|
+
return "contains_valid_link"
|
|
198
|
+
|
|
199
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
200
|
+
pattern = r"(?!.*@)(?:https?://)?(?:www\.)?\S+\.\S+"
|
|
201
|
+
match = re.search(pattern, inputs.response)
|
|
202
|
+
if not match:
|
|
203
|
+
return {"output": 0.0, "reason": "No link found in response."}
|
|
204
|
+
|
|
205
|
+
url = _standardize_url(match.group(0))
|
|
206
|
+
try:
|
|
207
|
+
response = requests.head(url, timeout=5)
|
|
208
|
+
if 200 <= response.status_code < 300:
|
|
209
|
+
return {
|
|
210
|
+
"output": 1.0,
|
|
211
|
+
"reason": f"Valid link '{url}' found (Status: {response.status_code})",
|
|
212
|
+
}
|
|
213
|
+
else:
|
|
214
|
+
return {
|
|
215
|
+
"output": 0.0,
|
|
216
|
+
"reason": f"Invalid link '{url}' found (Status: {response.status_code})",
|
|
217
|
+
}
|
|
218
|
+
except requests.RequestException as e:
|
|
219
|
+
return {
|
|
220
|
+
"output": 0.0,
|
|
221
|
+
"reason": f"Unreachable link '{url}' found. Error: {e}",
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class Equals(BaseMetric[TextMetricInput]):
|
|
226
|
+
"""Checks if the response text exactly matches the expected text."""
|
|
227
|
+
|
|
228
|
+
@property
|
|
229
|
+
def metric_name(self) -> str:
|
|
230
|
+
return "equals"
|
|
231
|
+
|
|
232
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
233
|
+
super().__init__(config)
|
|
234
|
+
self.case_sensitive = self.config.get("case_sensitive", False)
|
|
235
|
+
|
|
236
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
237
|
+
if inputs.expected_response is None:
|
|
238
|
+
raise ValueError("Equals metric requires 'expected_response' to be provided.")
|
|
239
|
+
if not isinstance(inputs.expected_response, str):
|
|
240
|
+
raise TypeError("Equals metric requires 'expected_response' to be a string.")
|
|
241
|
+
resp, expected = (
|
|
242
|
+
(inputs.response, inputs.expected_response)
|
|
243
|
+
if self.case_sensitive
|
|
244
|
+
else (inputs.response.lower(), inputs.expected_response.lower())
|
|
245
|
+
)
|
|
246
|
+
return {
|
|
247
|
+
"output": 1.0 if resp == expected else 0.0,
|
|
248
|
+
"reason": "Response matches expected text."
|
|
249
|
+
if resp == expected
|
|
250
|
+
else "Response does not match.",
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
class StartsWith(BaseMetric[TextMetricInput]):
|
|
255
|
+
"""Checks if the response text starts with the expected text."""
|
|
256
|
+
|
|
257
|
+
@property
|
|
258
|
+
def metric_name(self) -> str:
|
|
259
|
+
return "starts_with"
|
|
260
|
+
|
|
261
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
262
|
+
super().__init__(config)
|
|
263
|
+
self.case_sensitive = self.config.get("case_sensitive", False)
|
|
264
|
+
|
|
265
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
266
|
+
if inputs.expected_response is None:
|
|
267
|
+
raise ValueError("StartsWith metric requires 'expected_response' to be provided.")
|
|
268
|
+
if not isinstance(inputs.expected_response, str):
|
|
269
|
+
raise TypeError("StartsWith requires 'expected_response' to be a string.")
|
|
270
|
+
resp, prefix = (
|
|
271
|
+
(inputs.response, inputs.expected_response)
|
|
272
|
+
if self.case_sensitive
|
|
273
|
+
else (inputs.response.lower(), inputs.expected_response.lower())
|
|
274
|
+
)
|
|
275
|
+
starts = resp.startswith(prefix)
|
|
276
|
+
return {
|
|
277
|
+
"output": 1.0 if starts else 0.0,
|
|
278
|
+
"reason": f"Response starts with '{inputs.expected_response}'."
|
|
279
|
+
if starts
|
|
280
|
+
else f"Response does not start with '{inputs.expected_response}'.",
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
class EndsWith(BaseMetric[TextMetricInput]):
|
|
285
|
+
"""Checks if the response text ends with the expected text."""
|
|
286
|
+
|
|
287
|
+
@property
|
|
288
|
+
def metric_name(self) -> str:
|
|
289
|
+
return "ends_with"
|
|
290
|
+
|
|
291
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
292
|
+
super().__init__(config)
|
|
293
|
+
self.case_sensitive = self.config.get("case_sensitive", False)
|
|
294
|
+
|
|
295
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
296
|
+
if inputs.expected_response is None:
|
|
297
|
+
raise ValueError("EndsWith metric requires 'expected_response' to be provided.")
|
|
298
|
+
if not isinstance(inputs.expected_response, str):
|
|
299
|
+
raise TypeError("EndsWith requires 'expected_response' to be a string.")
|
|
300
|
+
resp, suffix = (
|
|
301
|
+
(inputs.response, inputs.expected_response)
|
|
302
|
+
if self.case_sensitive
|
|
303
|
+
else (inputs.response.lower(), inputs.expected_response.lower())
|
|
304
|
+
)
|
|
305
|
+
ends = resp.endswith(suffix)
|
|
306
|
+
return {
|
|
307
|
+
"output": 1.0 if ends else 0.0,
|
|
308
|
+
"reason": f"Response ends with '{inputs.expected_response}'."
|
|
309
|
+
if ends
|
|
310
|
+
else f"Response does not end with '{inputs.expected_response}'.",
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
# --- Length Metrics ---
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
class LengthLessThan(BaseMetric[TextMetricInput]):
|
|
318
|
+
"""Checks if text length is less than a max_length."""
|
|
319
|
+
|
|
320
|
+
@property
|
|
321
|
+
def metric_name(self) -> str:
|
|
322
|
+
return "length_less_than"
|
|
323
|
+
|
|
324
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
325
|
+
super().__init__(config)
|
|
326
|
+
self.max_length = self.config.get("max_length")
|
|
327
|
+
if not isinstance(self.max_length, int):
|
|
328
|
+
raise ValueError(
|
|
329
|
+
"LengthLessThan metric requires an integer 'max_length' config."
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
333
|
+
is_less = len(inputs.response) < self.max_length
|
|
334
|
+
return {
|
|
335
|
+
"output": 1.0 if is_less else 0.0,
|
|
336
|
+
"reason": f"Length {len(inputs.response)} < {self.max_length}"
|
|
337
|
+
if is_less
|
|
338
|
+
else f"Length {len(inputs.response)} >= {self.max_length}",
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
class LengthGreaterThan(BaseMetric[TextMetricInput]):
|
|
343
|
+
"""Checks if text length is greater than a min_length."""
|
|
344
|
+
|
|
345
|
+
@property
|
|
346
|
+
def metric_name(self) -> str:
|
|
347
|
+
return "length_greater_than"
|
|
348
|
+
|
|
349
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
350
|
+
super().__init__(config)
|
|
351
|
+
self.min_length = self.config.get("min_length")
|
|
352
|
+
if not isinstance(self.min_length, int):
|
|
353
|
+
raise ValueError(
|
|
354
|
+
"LengthGreaterThan metric requires an integer 'min_length' config."
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
358
|
+
is_greater = len(inputs.response) > self.min_length
|
|
359
|
+
return {
|
|
360
|
+
"output": 1.0 if is_greater else 0.0,
|
|
361
|
+
"reason": f"Length {len(inputs.response)} > {self.min_length}"
|
|
362
|
+
if is_greater
|
|
363
|
+
else f"Length {len(inputs.response)} <= {self.min_length}",
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
class LengthBetween(BaseMetric[TextMetricInput]):
|
|
368
|
+
"""Checks if text length is between a min_length and max_length."""
|
|
369
|
+
|
|
370
|
+
@property
|
|
371
|
+
def metric_name(self) -> str:
|
|
372
|
+
return "length_between"
|
|
373
|
+
|
|
374
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
375
|
+
super().__init__(config)
|
|
376
|
+
self.min_length = self.config.get("min_length")
|
|
377
|
+
self.max_length = self.config.get("max_length")
|
|
378
|
+
if not isinstance(self.min_length, int) or not isinstance(self.max_length, int):
|
|
379
|
+
raise ValueError(
|
|
380
|
+
"LengthBetween requires integer 'min_length' and 'max_length' configs."
|
|
381
|
+
)
|
|
382
|
+
|
|
383
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
384
|
+
length = len(inputs.response)
|
|
385
|
+
is_between = self.min_length <= length <= self.max_length
|
|
386
|
+
reason = (
|
|
387
|
+
f"Length {length} is between [{self.min_length}, {self.max_length}]"
|
|
388
|
+
if is_between
|
|
389
|
+
else f"Length {length} is not between [{self.min_length}, {self.max_length}]"
|
|
390
|
+
)
|
|
391
|
+
return {"output": 1.0 if is_between else 0.0, "reason": reason}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from .custom_judge.metric import CustomLLMJudge
|
|
2
|
+
from .types import (
|
|
3
|
+
CustomInput,
|
|
4
|
+
BaseLLMJudgeInput,
|
|
5
|
+
LLMFewShotExample,
|
|
6
|
+
LLMMessage,
|
|
7
|
+
DefaultJudgeOutput,
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"CustomLLMJudge",
|
|
12
|
+
"CustomInput",
|
|
13
|
+
"BaseLLMJudgeInput",
|
|
14
|
+
"LLMFewShotExample",
|
|
15
|
+
"LLMMessage",
|
|
16
|
+
"DefaultJudgeOutput",
|
|
17
|
+
]
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Any, Dict, List, Type
|
|
3
|
+
from pydantic import BaseModel
|
|
4
|
+
from jinja2 import Environment, BaseLoader
|
|
5
|
+
|
|
6
|
+
from ...base_llm_metric import BaseLLMJudgeMetric
|
|
7
|
+
from ..types import CustomInput, DefaultJudgeOutput
|
|
8
|
+
from ....llm.base_llm_provider import LLMProvider
|
|
9
|
+
from .prompts import DEFAULT_USER_PROMPT_TEMPLATE
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class CustomLLMJudge(BaseLLMJudgeMetric[CustomInput]):
|
|
13
|
+
"""
|
|
14
|
+
A smart, user-configurable LLM-as-a-judge metric that prioritizes ease of use.
|
|
15
|
+
|
|
16
|
+
For the most common use cases, the user only needs to provide their
|
|
17
|
+
grading criteria. The judge provides sensible defaults for the prompt
|
|
18
|
+
template and output format, which can be optionally overridden for
|
|
19
|
+
advanced customization.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def metric_name(self) -> str:
|
|
24
|
+
return self.config.get("name", "custom_llm_judge")
|
|
25
|
+
|
|
26
|
+
def __init__(self, provider: LLMProvider, config: Dict[str, Any], **litellm_kwargs):
|
|
27
|
+
# The ONLY required key is now 'grading_criteria'
|
|
28
|
+
if "grading_criteria" not in config:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
"CustomLLMJudge config must contain a 'grading_criteria' key."
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
super().__init__(provider, config, **litellm_kwargs)
|
|
34
|
+
|
|
35
|
+
# Explicitly set the input model, as this class is generic
|
|
36
|
+
self.input_model = CustomInput
|
|
37
|
+
|
|
38
|
+
# Smartly decide which Pydantic model to uset
|
|
39
|
+
self._output_model = DefaultJudgeOutput
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def output_pydantic_model(self) -> Type[BaseModel]:
|
|
43
|
+
return self._output_model
|
|
44
|
+
|
|
45
|
+
# Keys whose values are media URLs that should be sent as content parts
|
|
46
|
+
_IMAGE_KEYS = {"image_url", "input_image_url", "output_image_url", "image"}
|
|
47
|
+
_AUDIO_KEYS = {"audio_url", "input_audio_url", "audio"}
|
|
48
|
+
|
|
49
|
+
def _create_prompt_messages(self, inputs: CustomInput) -> List[Dict[str, Any]]:
|
|
50
|
+
jinja_env = Environment(loader=BaseLoader())
|
|
51
|
+
jinja_env.filters["tojson"] = json.dumps
|
|
52
|
+
|
|
53
|
+
# Use the user-provided template if it exists, otherwise use the default
|
|
54
|
+
template_str = self.config.get(
|
|
55
|
+
"user_prompt_template", DEFAULT_USER_PROMPT_TEMPLATE
|
|
56
|
+
)
|
|
57
|
+
template = jinja_env.from_string(template_str)
|
|
58
|
+
|
|
59
|
+
render_context = {
|
|
60
|
+
"grading_criteria": self.config["grading_criteria"],
|
|
61
|
+
"few_shot_examples": self.config.get("few_shot_examples", []),
|
|
62
|
+
"task_input": inputs.model_dump(),
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
user_prompt = template.render(render_context)
|
|
66
|
+
|
|
67
|
+
system_prompt = self.config.get(
|
|
68
|
+
"system_prompt",
|
|
69
|
+
"You are an expert AI evaluator. Follow the user's instructions and output format precisely.",
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
# Detect multimodal inputs and build content parts
|
|
73
|
+
user_content = self._build_content(user_prompt, inputs.model_dump())
|
|
74
|
+
|
|
75
|
+
return [
|
|
76
|
+
{"role": "system", "content": system_prompt},
|
|
77
|
+
{"role": "user", "content": user_content},
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
def _build_content(self, text: str, input_data: Dict[str, Any]):
|
|
81
|
+
"""Build message content — plain string or list of content parts with media."""
|
|
82
|
+
media_parts = []
|
|
83
|
+
|
|
84
|
+
for key, value in input_data.items():
|
|
85
|
+
if not value or not isinstance(value, str):
|
|
86
|
+
continue
|
|
87
|
+
if key in self._IMAGE_KEYS:
|
|
88
|
+
media_parts.append({
|
|
89
|
+
"type": "image_url",
|
|
90
|
+
"image_url": {"url": value},
|
|
91
|
+
})
|
|
92
|
+
elif key in self._AUDIO_KEYS:
|
|
93
|
+
# LiteLLM translates image_url type to the correct provider
|
|
94
|
+
# format for audio URLs too (Gemini, OpenAI, etc.)
|
|
95
|
+
media_parts.append({
|
|
96
|
+
"type": "image_url",
|
|
97
|
+
"image_url": {"url": value},
|
|
98
|
+
})
|
|
99
|
+
|
|
100
|
+
if not media_parts:
|
|
101
|
+
return text
|
|
102
|
+
|
|
103
|
+
return [{"type": "text", "text": text}] + media_parts
|
|
104
|
+
|
|
105
|
+
def _normalize_score(self, parsed_output: BaseModel) -> Dict[str, Any]:
|
|
106
|
+
"""Normalizes the score from the validated Pydantic output."""
|
|
107
|
+
output_dict = parsed_output.model_dump()
|
|
108
|
+
|
|
109
|
+
# Prioritize finding a "score" field, as it's our default
|
|
110
|
+
score_val = output_dict.get("score", 1.0)
|
|
111
|
+
|
|
112
|
+
return {"output": float(score_val), "reason": json.dumps(output_dict, indent=2)}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
DEFAULT_USER_PROMPT_TEMPLATE = """
|
|
2
|
+
### GRADING CRITERIA ###
|
|
3
|
+
{{ grading_criteria }}
|
|
4
|
+
|
|
5
|
+
### OUTPUT FORMAT ###
|
|
6
|
+
You MUST return a valid JSON object with two keys: "score" (a float between 0.0 and 1.0) and "reason" (a brief explanation of your score).
|
|
7
|
+
|
|
8
|
+
{% if few_shot_examples %}
|
|
9
|
+
### EXAMPLES ###
|
|
10
|
+
{% for example in few_shot_examples -%}
|
|
11
|
+
---
|
|
12
|
+
INPUT:
|
|
13
|
+
{{ example.inputs | tojson(indent=2) }}
|
|
14
|
+
|
|
15
|
+
EXPECTED JUDGEMENT:
|
|
16
|
+
{{ example.output }}
|
|
17
|
+
---
|
|
18
|
+
{% endfor %}
|
|
19
|
+
{% endif %}
|
|
20
|
+
|
|
21
|
+
### TASK ###
|
|
22
|
+
Based on the grading criteria, please evaluate the following input.
|
|
23
|
+
|
|
24
|
+
### INPUT ###
|
|
25
|
+
{{ task_input | tojson(indent=2) }}
|
|
26
|
+
"""
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from typing import Any, Dict, List, Optional
|
|
2
|
+
from pydantic import BaseModel, Field
|
|
3
|
+
from ...types import BaseMetricInput
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class BaseLLMJudgeInput(BaseMetricInput):
|
|
7
|
+
pass
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class LLMFewShotExample(BaseModel):
|
|
11
|
+
inputs: Dict[str, Any] = Field(
|
|
12
|
+
..., description="A dictionary representing an input model"
|
|
13
|
+
)
|
|
14
|
+
output: str = Field(
|
|
15
|
+
...,
|
|
16
|
+
description="The ideal JSON string the judge LLM should produce for these inputs.",
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class CustomInput(BaseLLMJudgeInput):
|
|
21
|
+
"""A flexible input model for the CustomLLMJudge that allows any field."""
|
|
22
|
+
|
|
23
|
+
class Config:
|
|
24
|
+
extra = "allow"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class DefaultJudgeOutput(BaseModel):
|
|
28
|
+
"""The default output format for a custom judge."""
|
|
29
|
+
|
|
30
|
+
score: float = Field(
|
|
31
|
+
...,
|
|
32
|
+
ge=0.0,
|
|
33
|
+
le=1.0,
|
|
34
|
+
description="The normalized evaluation score from 0.0 to 1.0.",
|
|
35
|
+
)
|
|
36
|
+
reason: str = Field(..., description="A brief explanation of the score.")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class LLMMessage(BaseModel):
|
|
40
|
+
role: str
|
|
41
|
+
content: str
|
|
42
|
+
name: Optional[str]
|
|
43
|
+
function_call: Optional[str]
|
|
44
|
+
tool_call_id: Optional[str]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class ConversationInput(BaseLLMJudgeInput):
|
|
48
|
+
messages: List[LLMMessage]
|