agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Natural Language Inference utilities for hallucination detection.
|
|
3
|
+
|
|
4
|
+
Provides entailment checking between claims and context.
|
|
5
|
+
Uses transformer-based NLI when available (pip install ai-evaluation[nli]),
|
|
6
|
+
falls back to word-overlap heuristic with a warning.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
import warnings
|
|
11
|
+
from typing import Tuple, List, Optional
|
|
12
|
+
from enum import Enum
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
_NLI_MODEL = "cross-encoder/nli-deberta-v3-base"
|
|
16
|
+
|
|
17
|
+
# Try to import transformers
|
|
18
|
+
_NLI_AVAILABLE = False
|
|
19
|
+
try:
|
|
20
|
+
from transformers import pipeline as _hf_pipeline
|
|
21
|
+
_NLI_AVAILABLE = True
|
|
22
|
+
except ImportError:
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
_nli_pipeline = None
|
|
26
|
+
_heuristic_warning_issued = False
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class NLILabel(Enum):
|
|
30
|
+
"""NLI classification labels."""
|
|
31
|
+
|
|
32
|
+
ENTAILMENT = "entailment"
|
|
33
|
+
CONTRADICTION = "contradiction"
|
|
34
|
+
NEUTRAL = "neutral"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _get_nli_pipeline():
|
|
38
|
+
"""Lazy-load NLI pipeline. Returns the pipeline or False on failure."""
|
|
39
|
+
global _nli_pipeline
|
|
40
|
+
if _nli_pipeline is not None:
|
|
41
|
+
return _nli_pipeline
|
|
42
|
+
|
|
43
|
+
if not _NLI_AVAILABLE:
|
|
44
|
+
_nli_pipeline = False
|
|
45
|
+
return False
|
|
46
|
+
|
|
47
|
+
try:
|
|
48
|
+
_nli_pipeline = _hf_pipeline(
|
|
49
|
+
"text-classification",
|
|
50
|
+
model=_NLI_MODEL,
|
|
51
|
+
device=-1, # CPU
|
|
52
|
+
)
|
|
53
|
+
except Exception as exc:
|
|
54
|
+
warnings.warn(
|
|
55
|
+
f"Failed to load NLI model '{_NLI_MODEL}': {exc}. "
|
|
56
|
+
"Falling back to word-overlap heuristic. "
|
|
57
|
+
"Install with: pip install ai-evaluation[nli]",
|
|
58
|
+
RuntimeWarning,
|
|
59
|
+
stacklevel=2,
|
|
60
|
+
)
|
|
61
|
+
_nli_pipeline = False
|
|
62
|
+
|
|
63
|
+
return _nli_pipeline
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _warn_heuristic_fallback():
|
|
67
|
+
"""Issue a one-time warning that we're using the heuristic fallback."""
|
|
68
|
+
global _heuristic_warning_issued
|
|
69
|
+
if not _heuristic_warning_issued:
|
|
70
|
+
_heuristic_warning_issued = True
|
|
71
|
+
warnings.warn(
|
|
72
|
+
"NLI model not available — using word-overlap heuristic for "
|
|
73
|
+
"hallucination detection. Results will be approximate. "
|
|
74
|
+
"For accurate NLI, install: pip install ai-evaluation[nli]",
|
|
75
|
+
RuntimeWarning,
|
|
76
|
+
stacklevel=3,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
_LABEL_MAP = {
|
|
81
|
+
"ENTAILMENT": NLILabel.ENTAILMENT,
|
|
82
|
+
"CONTRADICTION": NLILabel.CONTRADICTION,
|
|
83
|
+
"NEUTRAL": NLILabel.NEUTRAL,
|
|
84
|
+
"entailment": NLILabel.ENTAILMENT,
|
|
85
|
+
"contradiction": NLILabel.CONTRADICTION,
|
|
86
|
+
"neutral": NLILabel.NEUTRAL,
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def check_entailment(premise: str, hypothesis: str) -> Tuple[NLILabel, float]:
|
|
91
|
+
"""
|
|
92
|
+
Check if premise entails hypothesis using NLI model.
|
|
93
|
+
|
|
94
|
+
Args:
|
|
95
|
+
premise: The source text (context)
|
|
96
|
+
hypothesis: The claim to verify
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
Tuple of (NLI label, confidence score)
|
|
100
|
+
"""
|
|
101
|
+
nli = _get_nli_pipeline()
|
|
102
|
+
|
|
103
|
+
if not nli:
|
|
104
|
+
_warn_heuristic_fallback()
|
|
105
|
+
return check_entailment_heuristic(premise, hypothesis)
|
|
106
|
+
|
|
107
|
+
try:
|
|
108
|
+
result = nli(
|
|
109
|
+
{"text": premise, "text_pair": hypothesis},
|
|
110
|
+
truncation=True,
|
|
111
|
+
max_length=512,
|
|
112
|
+
)
|
|
113
|
+
# Dict input returns a single dict, not a list
|
|
114
|
+
entry = result[0] if isinstance(result, list) else result
|
|
115
|
+
label = _LABEL_MAP.get(entry["label"], NLILabel.NEUTRAL)
|
|
116
|
+
score = entry["score"]
|
|
117
|
+
return label, score
|
|
118
|
+
except Exception:
|
|
119
|
+
return check_entailment_heuristic(premise, hypothesis)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
# ---------------------------------------------------------------------------
|
|
123
|
+
# Heuristic fallback
|
|
124
|
+
# ---------------------------------------------------------------------------
|
|
125
|
+
|
|
126
|
+
_STOPWORDS = frozenset({
|
|
127
|
+
"the", "a", "an", "is", "are", "was", "were", "be", "been", "being",
|
|
128
|
+
"have", "has", "had", "do", "does", "did", "will", "would", "could",
|
|
129
|
+
"should", "may", "might", "must", "shall", "can", "to", "of", "in",
|
|
130
|
+
"for", "on", "with", "at", "by", "from", "as", "and", "or", "but",
|
|
131
|
+
"if", "that", "this", "it", "its", "they", "their", "he", "she",
|
|
132
|
+
"him", "her", "his", "we", "our", "you", "your",
|
|
133
|
+
})
|
|
134
|
+
|
|
135
|
+
_NEGATIONS = frozenset({
|
|
136
|
+
"not", "n't", "never", "no", "none", "neither", "nor", "cannot",
|
|
137
|
+
})
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _tokenize(text: str) -> set:
|
|
141
|
+
"""Tokenize text into content words, stripping punctuation."""
|
|
142
|
+
return set(re.findall(r"\b\w+\b", text.lower()))
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def check_entailment_heuristic(
|
|
146
|
+
premise: str, hypothesis: str
|
|
147
|
+
) -> Tuple[NLILabel, float]:
|
|
148
|
+
"""
|
|
149
|
+
Heuristic entailment check using word overlap and similarity.
|
|
150
|
+
|
|
151
|
+
Fallback when NLI model is not available. Uses:
|
|
152
|
+
- Content word overlap for entailment signal
|
|
153
|
+
- Negation asymmetry for contradiction detection
|
|
154
|
+
- Numeric mismatch detection
|
|
155
|
+
|
|
156
|
+
Args:
|
|
157
|
+
premise: The source text (context)
|
|
158
|
+
hypothesis: The claim to verify
|
|
159
|
+
|
|
160
|
+
Returns:
|
|
161
|
+
Tuple of (NLI label, confidence score)
|
|
162
|
+
"""
|
|
163
|
+
premise_words = _tokenize(premise)
|
|
164
|
+
hypothesis_words = _tokenize(hypothesis)
|
|
165
|
+
|
|
166
|
+
premise_content = premise_words - _STOPWORDS
|
|
167
|
+
hypothesis_content = hypothesis_words - _STOPWORDS
|
|
168
|
+
|
|
169
|
+
if not hypothesis_content:
|
|
170
|
+
return NLILabel.NEUTRAL, 0.5
|
|
171
|
+
|
|
172
|
+
overlap = len(premise_content & hypothesis_content)
|
|
173
|
+
coverage = overlap / len(hypothesis_content)
|
|
174
|
+
|
|
175
|
+
# Check for negation asymmetry
|
|
176
|
+
premise_has_neg = bool(premise_words & _NEGATIONS)
|
|
177
|
+
hypothesis_has_neg = bool(hypothesis_words & _NEGATIONS)
|
|
178
|
+
|
|
179
|
+
if premise_has_neg != hypothesis_has_neg and coverage > 0.5:
|
|
180
|
+
return NLILabel.CONTRADICTION, 0.6
|
|
181
|
+
|
|
182
|
+
# Check for numeric mismatch — only when high overlap + claim has numbers not in premise
|
|
183
|
+
premise_numbers = set(re.findall(r"\b\d+(?:\.\d+)?\b", premise))
|
|
184
|
+
hypothesis_numbers = set(re.findall(r"\b\d+(?:\.\d+)?\b", hypothesis))
|
|
185
|
+
|
|
186
|
+
if hypothesis_numbers and premise_numbers and coverage > 0.5:
|
|
187
|
+
novel_numbers = hypothesis_numbers - premise_numbers
|
|
188
|
+
if novel_numbers and len(novel_numbers) <= 2:
|
|
189
|
+
return NLILabel.CONTRADICTION, 0.55
|
|
190
|
+
|
|
191
|
+
if coverage >= 0.65:
|
|
192
|
+
return NLILabel.ENTAILMENT, coverage
|
|
193
|
+
elif coverage >= 0.3:
|
|
194
|
+
return NLILabel.NEUTRAL, coverage
|
|
195
|
+
else:
|
|
196
|
+
return NLILabel.NEUTRAL, coverage
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def check_contradiction(claim: str, context: str) -> Tuple[bool, float]:
|
|
200
|
+
"""
|
|
201
|
+
Check if claim contradicts the context.
|
|
202
|
+
|
|
203
|
+
Args:
|
|
204
|
+
claim: The claim to check
|
|
205
|
+
context: The context to check against
|
|
206
|
+
|
|
207
|
+
Returns:
|
|
208
|
+
Tuple of (is_contradiction, confidence)
|
|
209
|
+
"""
|
|
210
|
+
label, score = check_entailment(context, claim)
|
|
211
|
+
|
|
212
|
+
if label == NLILabel.CONTRADICTION:
|
|
213
|
+
return True, score
|
|
214
|
+
|
|
215
|
+
return False, 0.0
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def nli_score_for_claim(
|
|
219
|
+
claim: str, contexts: List[str]
|
|
220
|
+
) -> Tuple[NLILabel, float, Optional[str]]:
|
|
221
|
+
"""
|
|
222
|
+
Get the best NLI score for a claim against multiple contexts.
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
claim: The claim to verify
|
|
226
|
+
contexts: List of context passages
|
|
227
|
+
|
|
228
|
+
Returns:
|
|
229
|
+
Tuple of (best_label, best_score, best_context_snippet)
|
|
230
|
+
"""
|
|
231
|
+
best_label = NLILabel.NEUTRAL
|
|
232
|
+
best_score = 0.0
|
|
233
|
+
best_context = None
|
|
234
|
+
|
|
235
|
+
for ctx in contexts:
|
|
236
|
+
label, score = check_entailment(ctx, claim)
|
|
237
|
+
|
|
238
|
+
if label == NLILabel.CONTRADICTION and score > 0.5:
|
|
239
|
+
# Contradiction takes priority if confident
|
|
240
|
+
snippet = ctx[:200] + "..." if len(ctx) > 200 else ctx
|
|
241
|
+
return label, score, snippet
|
|
242
|
+
|
|
243
|
+
if label == NLILabel.ENTAILMENT and score > best_score:
|
|
244
|
+
best_label = label
|
|
245
|
+
best_score = score
|
|
246
|
+
best_context = ctx[:200] + "..." if len(ctx) > 200 else ctx
|
|
247
|
+
elif label == NLILabel.NEUTRAL and best_label != NLILabel.ENTAILMENT:
|
|
248
|
+
if score > best_score:
|
|
249
|
+
best_label = label
|
|
250
|
+
best_score = score
|
|
251
|
+
best_context = ctx[:200] + "..." if len(ctx) > 200 else ctx
|
|
252
|
+
|
|
253
|
+
return best_label, best_score, best_context
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Hallucination Sentinel — fast pre-screening before full NLI.
|
|
3
|
+
|
|
4
|
+
Provides rule-based screening to quickly flag responses
|
|
5
|
+
that are likely or unlikely to contain hallucinations,
|
|
6
|
+
avoiding expensive NLI inference for obvious cases.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from typing import Dict, List, Literal, Tuple
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
RiskLevel = Literal["low", "medium", "high"]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# Patterns that indicate high hallucination risk
|
|
17
|
+
_HIGH_RISK_PATTERNS = [
|
|
18
|
+
r"\baccording to (?:recent|latest|new)\b",
|
|
19
|
+
r"\bstudies (?:show|prove|confirm|suggest)\b",
|
|
20
|
+
r"\bresearch (?:shows|proves|confirms|suggests)\b",
|
|
21
|
+
r"\bstatistics (?:show|indicate|reveal)\b",
|
|
22
|
+
r"\b\d+(?:\.\d+)?%\b", # Specific percentages
|
|
23
|
+
r"\bin \d{4}\b", # Specific years
|
|
24
|
+
r"\bexactly \d+\b", # Exact numbers
|
|
25
|
+
r"\bproven (?:fact|to be)\b",
|
|
26
|
+
r"\bit is (?:well[- ]known|widely accepted|universally agreed)\b",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
# Patterns that indicate the response is hedging (lower risk)
|
|
30
|
+
_HEDGE_PATTERNS = [
|
|
31
|
+
r"\bI (?:think|believe|am not sure)\b",
|
|
32
|
+
r"\b(?:might|may|could|possibly|perhaps|probably)\b",
|
|
33
|
+
r"\b(?:it seems|it appears|it looks like)\b",
|
|
34
|
+
r"\bI don't (?:know|have)\b",
|
|
35
|
+
r"\bnot (?:certain|sure|clear)\b",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class HallucinationSentinel:
|
|
40
|
+
"""Fast rule-based screening for hallucination risk."""
|
|
41
|
+
|
|
42
|
+
def __init__(
|
|
43
|
+
self,
|
|
44
|
+
extra_risk_patterns: List[str] = None,
|
|
45
|
+
):
|
|
46
|
+
self.risk_patterns = _HIGH_RISK_PATTERNS + (extra_risk_patterns or [])
|
|
47
|
+
|
|
48
|
+
def screen(
|
|
49
|
+
self, response: str, context: str
|
|
50
|
+
) -> Tuple[RiskLevel, Dict]:
|
|
51
|
+
"""
|
|
52
|
+
Screen a response for hallucination risk.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
response: The LLM response to screen
|
|
56
|
+
context: The source context
|
|
57
|
+
|
|
58
|
+
Returns:
|
|
59
|
+
Tuple of (risk_level, details dict)
|
|
60
|
+
"""
|
|
61
|
+
response_lower = response.lower()
|
|
62
|
+
context_lower = context.lower()
|
|
63
|
+
|
|
64
|
+
details: Dict = {
|
|
65
|
+
"risk_signals": [],
|
|
66
|
+
"hedge_signals": [],
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
# Check risk patterns
|
|
70
|
+
risk_count = 0
|
|
71
|
+
for pattern in self.risk_patterns:
|
|
72
|
+
matches = re.findall(pattern, response_lower, re.IGNORECASE)
|
|
73
|
+
if matches:
|
|
74
|
+
risk_count += len(matches)
|
|
75
|
+
details["risk_signals"].append(pattern)
|
|
76
|
+
|
|
77
|
+
# Check hedging patterns
|
|
78
|
+
hedge_count = 0
|
|
79
|
+
for pattern in _HEDGE_PATTERNS:
|
|
80
|
+
if re.search(pattern, response_lower, re.IGNORECASE):
|
|
81
|
+
hedge_count += 1
|
|
82
|
+
details["hedge_signals"].append(pattern)
|
|
83
|
+
|
|
84
|
+
# Check if claims reference things not in context
|
|
85
|
+
# Simple: response words not found in context
|
|
86
|
+
response_words = set(re.findall(r"\b\w{4,}\b", response_lower))
|
|
87
|
+
context_words = set(re.findall(r"\b\w{4,}\b", context_lower))
|
|
88
|
+
novel_ratio = len(response_words - context_words) / max(len(response_words), 1)
|
|
89
|
+
details["novel_word_ratio"] = round(novel_ratio, 3)
|
|
90
|
+
|
|
91
|
+
# Determine risk level
|
|
92
|
+
if risk_count >= 3 or (risk_count >= 1 and novel_ratio > 0.6):
|
|
93
|
+
risk_level: RiskLevel = "high"
|
|
94
|
+
elif risk_count >= 1 or novel_ratio > 0.5:
|
|
95
|
+
risk_level = "medium"
|
|
96
|
+
else:
|
|
97
|
+
risk_level = "low"
|
|
98
|
+
|
|
99
|
+
# Hedging reduces risk
|
|
100
|
+
if hedge_count >= 2 and risk_level == "high":
|
|
101
|
+
risk_level = "medium"
|
|
102
|
+
|
|
103
|
+
details["risk_count"] = risk_count
|
|
104
|
+
details["hedge_count"] = hedge_count
|
|
105
|
+
|
|
106
|
+
return risk_level, details
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Types for Hallucination Detection.
|
|
3
|
+
|
|
4
|
+
These types support NLI-based and semantic analysis for
|
|
5
|
+
detecting hallucinations in LLM outputs.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Any, Dict, List, Literal, Optional, Union
|
|
9
|
+
from pydantic import BaseModel, Field
|
|
10
|
+
|
|
11
|
+
from ...types import BaseMetricInput
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Claim(BaseModel):
|
|
15
|
+
"""Represents a single claim extracted from text."""
|
|
16
|
+
|
|
17
|
+
text: str = Field(..., description="The claim text")
|
|
18
|
+
source_span: Optional[str] = Field(
|
|
19
|
+
default=None,
|
|
20
|
+
description="Original text span the claim was extracted from"
|
|
21
|
+
)
|
|
22
|
+
confidence: Optional[float] = Field(
|
|
23
|
+
default=None,
|
|
24
|
+
description="Confidence score for claim extraction (0-1)"
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class HallucinationInput(BaseMetricInput):
|
|
29
|
+
"""
|
|
30
|
+
Input for hallucination detection metrics.
|
|
31
|
+
|
|
32
|
+
Evaluates whether the response contains claims not supported
|
|
33
|
+
by the provided context/source.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
# The LLM response to check for hallucinations
|
|
37
|
+
response: str = Field(
|
|
38
|
+
...,
|
|
39
|
+
description="The LLM response to evaluate for hallucinations."
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
# The source/context that the response should be faithful to
|
|
43
|
+
context: Union[str, List[str]] = Field(
|
|
44
|
+
...,
|
|
45
|
+
description="Source context(s) the response should be grounded in."
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
# Optional: pre-extracted claims from response
|
|
49
|
+
claims: Optional[List[Claim]] = Field(
|
|
50
|
+
default=None,
|
|
51
|
+
description="Pre-extracted claims from response. If not provided, claims are extracted automatically."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
# Optional: the query that generated the response
|
|
55
|
+
query: Optional[str] = Field(
|
|
56
|
+
default=None,
|
|
57
|
+
description="The original query/question that generated the response."
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class ClaimExtractionInput(BaseMetricInput):
|
|
62
|
+
"""
|
|
63
|
+
Input for claim extraction from text.
|
|
64
|
+
|
|
65
|
+
Extracts atomic, verifiable claims from a text passage.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
response: str = Field(
|
|
69
|
+
...,
|
|
70
|
+
description="The text to extract claims from."
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# Extraction granularity
|
|
74
|
+
granularity: Literal["sentence", "clause", "atomic"] = Field(
|
|
75
|
+
default="sentence",
|
|
76
|
+
description="Granularity of claim extraction: sentence, clause, or atomic (finest)."
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class FactualConsistencyInput(BaseMetricInput):
|
|
81
|
+
"""
|
|
82
|
+
Input for factual consistency checking.
|
|
83
|
+
|
|
84
|
+
Evaluates whether claims in the response are consistent with
|
|
85
|
+
known facts or a reference text.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
response: str = Field(
|
|
89
|
+
...,
|
|
90
|
+
description="The response to check for factual consistency."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# Reference for fact-checking
|
|
94
|
+
reference: Optional[str] = Field(
|
|
95
|
+
default=None,
|
|
96
|
+
description="Reference text containing ground truth facts."
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
# Pre-extracted claims
|
|
100
|
+
claims: Optional[List[Claim]] = Field(
|
|
101
|
+
default=None,
|
|
102
|
+
description="Pre-extracted claims from the response."
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class NLIResult(BaseModel):
|
|
107
|
+
"""Result of Natural Language Inference classification."""
|
|
108
|
+
|
|
109
|
+
premise: str = Field(..., description="The premise (context)")
|
|
110
|
+
hypothesis: str = Field(..., description="The hypothesis (claim)")
|
|
111
|
+
label: Literal["entailment", "neutral", "contradiction"] = Field(
|
|
112
|
+
...,
|
|
113
|
+
description="NLI classification label"
|
|
114
|
+
)
|
|
115
|
+
scores: Dict[str, float] = Field(
|
|
116
|
+
default_factory=dict,
|
|
117
|
+
description="Probability scores for each label"
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class HallucinationResult(BaseModel):
|
|
122
|
+
"""Detailed result of hallucination detection."""
|
|
123
|
+
|
|
124
|
+
score: float = Field(..., description="Overall hallucination score (0=hallucinated, 1=faithful)")
|
|
125
|
+
claims_analyzed: int = Field(..., description="Number of claims analyzed")
|
|
126
|
+
supported_claims: int = Field(..., description="Number of claims supported by context")
|
|
127
|
+
unsupported_claims: int = Field(..., description="Number of unsupported claims")
|
|
128
|
+
contradicted_claims: int = Field(..., description="Number of contradicted claims")
|
|
129
|
+
claim_details: List[Dict[str, Any]] = Field(
|
|
130
|
+
default_factory=list,
|
|
131
|
+
description="Detailed analysis for each claim"
|
|
132
|
+
)
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Any, Dict, List, Optional
|
|
3
|
+
|
|
4
|
+
from ..base_metric import BaseMetric, BaseMetricInputType
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class AggregatedMetric(BaseMetric[BaseMetricInputType]):
|
|
8
|
+
"""
|
|
9
|
+
Combines multiple metric evaluators into a single aggregated score.
|
|
10
|
+
|
|
11
|
+
This metric assumes all sub-metrics can operate on the same input type.
|
|
12
|
+
|
|
13
|
+
Config:
|
|
14
|
+
- aggregator (str): 'average' or 'weighted_average'.
|
|
15
|
+
- metrics (List[BaseMetricInput]): A list of instantiated metric objects.
|
|
16
|
+
- weights (List[float]): Required if aggregator is 'weighted_average'.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
SUPPORTED_AGGREGATORS = ["average", "weighted_average"]
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def metric_name(self) -> str:
|
|
23
|
+
return "aggregated_metric"
|
|
24
|
+
|
|
25
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
26
|
+
super().__init__(config)
|
|
27
|
+
self.aggregator = self.config.get("aggregator", "average")
|
|
28
|
+
self.metrics: List[BaseMetric] = self.config.get("metrics", [])
|
|
29
|
+
self.weights: List[float] = self.config.get("weights", [])
|
|
30
|
+
|
|
31
|
+
if self.aggregator not in self.SUPPORTED_AGGREGATORS:
|
|
32
|
+
raise ValueError(f"Unsupported aggregator: {self.aggregator}")
|
|
33
|
+
if not self.metrics:
|
|
34
|
+
raise ValueError(
|
|
35
|
+
"AggregatedMetric requires at least one metric in its config."
|
|
36
|
+
)
|
|
37
|
+
if not all(isinstance(m, BaseMetric) for m in self.metrics):
|
|
38
|
+
raise TypeError("All items in 'metrics' must be instances of BaseMetric.")
|
|
39
|
+
|
|
40
|
+
# Explicitly set the input model based on the first sub-metric
|
|
41
|
+
self.input_model = self.metrics[0].input_model
|
|
42
|
+
|
|
43
|
+
if self.aggregator == "weighted_average":
|
|
44
|
+
if not self.weights or len(self.weights) != len(self.metrics):
|
|
45
|
+
raise ValueError(
|
|
46
|
+
"Weights are required for 'weighted_average' and must match the number of metrics."
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
def _normalize_score(self, value: Any) -> float:
|
|
50
|
+
"""Converts various score types to a float, clamping between 0 and 1."""
|
|
51
|
+
if isinstance(value, bool):
|
|
52
|
+
return 1.0 if value else 0.0
|
|
53
|
+
try:
|
|
54
|
+
float_value = float(value)
|
|
55
|
+
return max(0.0, min(1.0, float_value))
|
|
56
|
+
except (ValueError, TypeError):
|
|
57
|
+
return 0.0
|
|
58
|
+
|
|
59
|
+
def compute_one(self, inputs: BaseMetricInputType) -> Dict[str, Any]:
|
|
60
|
+
metric_scores = []
|
|
61
|
+
metric_details = {}
|
|
62
|
+
|
|
63
|
+
for metric in self.metrics:
|
|
64
|
+
try:
|
|
65
|
+
result_dict = metric.compute_one(inputs)
|
|
66
|
+
score = self._normalize_score(result_dict.get("output", 0.0))
|
|
67
|
+
except Exception:
|
|
68
|
+
# If a sub-metric fails, record a score of 0.0 for it
|
|
69
|
+
score = 0.0
|
|
70
|
+
|
|
71
|
+
metric_scores.append(score)
|
|
72
|
+
metric_details[metric.metric_name] = score
|
|
73
|
+
|
|
74
|
+
if not metric_scores:
|
|
75
|
+
return {"output": 0.0, "reason": "No metric scores were produced."}
|
|
76
|
+
|
|
77
|
+
if self.aggregator == "average":
|
|
78
|
+
aggregated_score = sum(metric_scores) / len(metric_scores)
|
|
79
|
+
else: # weighted_average
|
|
80
|
+
weighted_sum = sum(w * s for w, s in zip(self.weights, metric_scores))
|
|
81
|
+
total_weight = sum(self.weights)
|
|
82
|
+
aggregated_score = weighted_sum / total_weight if total_weight > 0 else 0.0
|
|
83
|
+
|
|
84
|
+
reason = f"Aggregated score calculated using '{self.aggregator}'. Details: {json.dumps(metric_details)}"
|
|
85
|
+
return {"output": aggregated_score, "reason": reason}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Any, Dict
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
# This would ideally be in a separate helper file
|
|
6
|
+
from jsonschema import validate
|
|
7
|
+
from jsonschema.exceptions import ValidationError as JsonSchemaValidationError
|
|
8
|
+
|
|
9
|
+
from ..base_metric import BaseMetric
|
|
10
|
+
from ...types import TextMetricInput, JsonMetricInput
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ContainsJson(BaseMetric[TextMetricInput]):
|
|
14
|
+
"""Checks if the response text contains a valid JSON object or array."""
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def metric_name(self) -> str:
|
|
18
|
+
return "contains_json"
|
|
19
|
+
|
|
20
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
21
|
+
text = inputs.response.strip()
|
|
22
|
+
# Simple regex to find potential JSON candidates
|
|
23
|
+
json_candidates = re.findall(r"\{.*\}|\[.*\]", text, re.DOTALL)
|
|
24
|
+
for candidate in json_candidates:
|
|
25
|
+
try:
|
|
26
|
+
json.loads(candidate)
|
|
27
|
+
return {
|
|
28
|
+
"output": 1.0,
|
|
29
|
+
"reason": "A valid JSON entity was found in the response.",
|
|
30
|
+
}
|
|
31
|
+
except json.JSONDecodeError:
|
|
32
|
+
continue
|
|
33
|
+
return {"output": 0.0, "reason": "No valid JSON entity found in the response."}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class IsJson(BaseMetric[TextMetricInput]):
|
|
37
|
+
"""Checks if the entire response text is a single, valid JSON object."""
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def metric_name(self) -> str:
|
|
41
|
+
return "is_json"
|
|
42
|
+
|
|
43
|
+
def compute_one(self, inputs: TextMetricInput) -> Dict[str, Any]:
|
|
44
|
+
try:
|
|
45
|
+
json.loads(inputs.response)
|
|
46
|
+
return {"output": 1.0, "reason": "Response is a valid JSON object."}
|
|
47
|
+
except json.JSONDecodeError as e:
|
|
48
|
+
return {
|
|
49
|
+
"output": 0.0,
|
|
50
|
+
"reason": f"Response is not a valid JSON object: {e}",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class JsonSchema(BaseMetric[JsonMetricInput]):
|
|
55
|
+
"""Validates the `response` against a provided JSON schema."""
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def metric_name(self) -> str:
|
|
59
|
+
return "json_schema"
|
|
60
|
+
|
|
61
|
+
def compute_one(self, inputs: JsonMetricInput) -> Dict[str, Any]:
|
|
62
|
+
if not inputs.schema:
|
|
63
|
+
raise ValueError("JsonSchema metric requires 'schema' to be provided.")
|
|
64
|
+
try:
|
|
65
|
+
actual_data = (
|
|
66
|
+
json.loads(inputs.response)
|
|
67
|
+
if isinstance(inputs.response, str)
|
|
68
|
+
else inputs.response
|
|
69
|
+
)
|
|
70
|
+
except json.JSONDecodeError as e:
|
|
71
|
+
return {"output": 0.0, "reason": f"Actual JSON is invalid: {e}"}
|
|
72
|
+
try:
|
|
73
|
+
schema_data = (
|
|
74
|
+
json.loads(inputs.schema)
|
|
75
|
+
if isinstance(inputs.schema, str)
|
|
76
|
+
else inputs.schema
|
|
77
|
+
)
|
|
78
|
+
except json.JSONDecodeError as e:
|
|
79
|
+
return {"output": 0.0, "reason": f"Schema JSON is invalid: {e}"}
|
|
80
|
+
try:
|
|
81
|
+
validate(instance=actual_data, schema=schema_data)
|
|
82
|
+
return {"output": 1.0, "reason": "JSON conforms to the schema."}
|
|
83
|
+
except JsonSchemaValidationError as e:
|
|
84
|
+
return {
|
|
85
|
+
"output": 0.0,
|
|
86
|
+
"reason": f"JSON schema validation failed: {e.message}",
|
|
87
|
+
}
|