agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
"""
|
|
2
|
+
evaluate() — the unified entrypoint for all evaluations.
|
|
3
|
+
|
|
4
|
+
Usage:
|
|
5
|
+
from fi.evals import evaluate
|
|
6
|
+
|
|
7
|
+
# Local metric (auto-detected)
|
|
8
|
+
result = evaluate("contains", output="hello world", keyword="hello")
|
|
9
|
+
|
|
10
|
+
# Cloud template (turing model → auto-routes to Turing)
|
|
11
|
+
result = evaluate("toxicity", output="hello world", model="turing_flash")
|
|
12
|
+
|
|
13
|
+
# Custom prompt on any LLM (use engine="llm", not "turing")
|
|
14
|
+
result = evaluate(
|
|
15
|
+
prompt="Rate the clarity of: {output}",
|
|
16
|
+
output="ML is a subset of AI.",
|
|
17
|
+
engine="llm",
|
|
18
|
+
model="gemini/gemini-2.0-flash",
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
# LLM-augmented: local heuristic first, then LLM refines
|
|
22
|
+
result = evaluate(
|
|
23
|
+
"faithfulness",
|
|
24
|
+
output="The capital of France is Paris.",
|
|
25
|
+
context="Paris is the capital of France.",
|
|
26
|
+
model="gemini/gemini-2.5-flash",
|
|
27
|
+
augment=True,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
# Multiple evals
|
|
31
|
+
results = evaluate(
|
|
32
|
+
["toxicity", "factual_accuracy"],
|
|
33
|
+
output="Paris is the capital of France",
|
|
34
|
+
model="turing_flash",
|
|
35
|
+
)
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
import warnings
|
|
39
|
+
from typing import Any, Dict, List, Optional, Union
|
|
40
|
+
|
|
41
|
+
from .registry import resolve_engine as _resolve_engine, is_turing_model
|
|
42
|
+
from .result import BatchResult, EvalResult
|
|
43
|
+
from .engines import LocalEngine, TuringEngine, LLMEngine, Engine
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def evaluate(
|
|
47
|
+
eval_name: Optional[Union[str, List[str]]] = None,
|
|
48
|
+
*,
|
|
49
|
+
prompt: Optional[str] = None,
|
|
50
|
+
engine: Optional[str] = None,
|
|
51
|
+
model: Optional[str] = None,
|
|
52
|
+
augment: Optional[bool] = None,
|
|
53
|
+
config: Optional[Dict[str, Any]] = None,
|
|
54
|
+
feedback_store: Optional[Any] = None,
|
|
55
|
+
generate_prompt: bool = False,
|
|
56
|
+
# Turing credentials (optional overrides)
|
|
57
|
+
fi_api_key: Optional[str] = None,
|
|
58
|
+
fi_secret_key: Optional[str] = None,
|
|
59
|
+
fi_base_url: Optional[str] = None,
|
|
60
|
+
**inputs,
|
|
61
|
+
) -> Union[EvalResult, BatchResult]:
|
|
62
|
+
"""Run one or more evaluations with automatic engine routing.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
eval_name: Metric/template name or list of names. Can be None when
|
|
66
|
+
using a custom prompt.
|
|
67
|
+
prompt: Custom evaluation prompt with {output}/{context}/{input}
|
|
68
|
+
placeholders. Requires explicit engine or a model hint.
|
|
69
|
+
Note: custom prompts only work with engine='llm'.
|
|
70
|
+
engine: Force a specific engine — "local", "turing", or "llm".
|
|
71
|
+
model: Model to use. Turing models (e.g. "turing_flash") auto-route
|
|
72
|
+
to the Turing engine. Other model strings (e.g.
|
|
73
|
+
"gemini/gemini-2.0-flash") auto-route to the LLM engine.
|
|
74
|
+
augment: When True, run the local heuristic first, then pass its
|
|
75
|
+
scores + reasoning to the LLM (specified by model=) for
|
|
76
|
+
refinement. Requires model= and a metric that supports
|
|
77
|
+
LLM augmentation (supports_llm_judge = True).
|
|
78
|
+
config: Optional metric/judge config dict.
|
|
79
|
+
feedback_store: Optional FeedbackStore for retrieving few-shot
|
|
80
|
+
examples from developer feedback. When provided with
|
|
81
|
+
augment=True, similar past feedback is injected into
|
|
82
|
+
the LLM judge prompt.
|
|
83
|
+
generate_prompt: When True, treat `prompt` as a short description
|
|
84
|
+
and auto-generate detailed grading criteria via LLM.
|
|
85
|
+
Requires model= and prompt=.
|
|
86
|
+
fi_api_key: Override FI_API_KEY for Turing engine.
|
|
87
|
+
fi_secret_key: Override FI_SECRET_KEY for Turing engine.
|
|
88
|
+
fi_base_url: Override FI_BASE_URL for Turing engine.
|
|
89
|
+
**inputs: Evaluation inputs (output, context, input, keyword, …).
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
EvalResult for a single eval, BatchResult for multiple.
|
|
93
|
+
"""
|
|
94
|
+
# --- Batch case: list of eval names --------------------------------
|
|
95
|
+
if isinstance(eval_name, list):
|
|
96
|
+
results = []
|
|
97
|
+
for name in eval_name:
|
|
98
|
+
r = evaluate(
|
|
99
|
+
name,
|
|
100
|
+
prompt=prompt,
|
|
101
|
+
engine=engine,
|
|
102
|
+
model=model,
|
|
103
|
+
augment=augment,
|
|
104
|
+
config=config,
|
|
105
|
+
feedback_store=feedback_store,
|
|
106
|
+
generate_prompt=generate_prompt,
|
|
107
|
+
fi_api_key=fi_api_key,
|
|
108
|
+
fi_secret_key=fi_secret_key,
|
|
109
|
+
fi_base_url=fi_base_url,
|
|
110
|
+
**inputs,
|
|
111
|
+
)
|
|
112
|
+
results.append(r)
|
|
113
|
+
return BatchResult(results=results)
|
|
114
|
+
|
|
115
|
+
# --- Single eval ---------------------------------------------------
|
|
116
|
+
# Auto-generate grading criteria from a short description
|
|
117
|
+
if generate_prompt and prompt:
|
|
118
|
+
if not model:
|
|
119
|
+
raise ValueError(
|
|
120
|
+
"generate_prompt=True requires a model= parameter "
|
|
121
|
+
"(e.g. model='gemini/gemini-2.5-flash')."
|
|
122
|
+
)
|
|
123
|
+
from .prompt_generator import generate_grading_criteria
|
|
124
|
+
prompt = generate_grading_criteria(prompt, model, inputs)
|
|
125
|
+
|
|
126
|
+
# Custom prompt with no eval_name
|
|
127
|
+
effective_name = eval_name or "custom_prompt"
|
|
128
|
+
|
|
129
|
+
# When augment=True, force local engine for initial run — model is for LLM step
|
|
130
|
+
resolved_engine = _resolve_engine(
|
|
131
|
+
eval_name,
|
|
132
|
+
model=None if augment else model,
|
|
133
|
+
prompt=prompt,
|
|
134
|
+
engine=engine,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
if resolved_engine is None:
|
|
138
|
+
raise ValueError(
|
|
139
|
+
f"Cannot auto-detect engine for '{eval_name}'. "
|
|
140
|
+
"Specify engine='local', engine='turing', or engine='llm', "
|
|
141
|
+
"or provide a model (turing models → turing, others → llm)."
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
eng = _get_engine(
|
|
145
|
+
resolved_engine,
|
|
146
|
+
fi_api_key=fi_api_key,
|
|
147
|
+
fi_secret_key=fi_secret_key,
|
|
148
|
+
fi_base_url=fi_base_url,
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
# Run with OTEL span if tracing is enabled
|
|
152
|
+
result = _run_with_tracing(
|
|
153
|
+
eng, effective_name, inputs,
|
|
154
|
+
model=model, prompt=prompt, config=config,
|
|
155
|
+
engine_type=resolved_engine,
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
# LLM augmentation: only when explicitly requested via augment=True
|
|
159
|
+
if augment:
|
|
160
|
+
result = _augment_with_llm(
|
|
161
|
+
result,
|
|
162
|
+
effective_name=effective_name,
|
|
163
|
+
inputs=inputs,
|
|
164
|
+
model=model,
|
|
165
|
+
resolved_engine=resolved_engine,
|
|
166
|
+
feedback_store=feedback_store,
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
return result
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
_ENGINE_FACTORIES = {
|
|
173
|
+
"local": lambda **_: LocalEngine(),
|
|
174
|
+
"turing": lambda **kw: TuringEngine(
|
|
175
|
+
fi_api_key=kw.get("fi_api_key"),
|
|
176
|
+
fi_secret_key=kw.get("fi_secret_key"),
|
|
177
|
+
fi_base_url=kw.get("fi_base_url"),
|
|
178
|
+
),
|
|
179
|
+
"llm": lambda **_: LLMEngine(),
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _get_engine(
|
|
184
|
+
engine_type: str,
|
|
185
|
+
*,
|
|
186
|
+
fi_api_key: Optional[str] = None,
|
|
187
|
+
fi_secret_key: Optional[str] = None,
|
|
188
|
+
fi_base_url: Optional[str] = None,
|
|
189
|
+
) -> Engine:
|
|
190
|
+
factory = _ENGINE_FACTORIES.get(engine_type.lower())
|
|
191
|
+
if factory is None:
|
|
192
|
+
raise ValueError(
|
|
193
|
+
f"Unknown engine: '{engine_type}'. "
|
|
194
|
+
f"Use one of: {', '.join(sorted(_ENGINE_FACTORIES))}."
|
|
195
|
+
)
|
|
196
|
+
return factory(fi_api_key=fi_api_key, fi_secret_key=fi_secret_key, fi_base_url=fi_base_url)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _run_with_tracing(
|
|
200
|
+
eng: Engine,
|
|
201
|
+
eval_name: str,
|
|
202
|
+
inputs: Dict[str, Any],
|
|
203
|
+
*,
|
|
204
|
+
model: Optional[str] = None,
|
|
205
|
+
prompt: Optional[str] = None,
|
|
206
|
+
config: Optional[Dict[str, Any]] = None,
|
|
207
|
+
engine_type: str = "",
|
|
208
|
+
) -> EvalResult:
|
|
209
|
+
"""Run an engine with optional OTEL tracing."""
|
|
210
|
+
try:
|
|
211
|
+
from fi.evals.otel.enrichment import (
|
|
212
|
+
is_auto_enrichment_enabled,
|
|
213
|
+
create_evaluation_span,
|
|
214
|
+
enrich_span_with_evaluation,
|
|
215
|
+
)
|
|
216
|
+
if not is_auto_enrichment_enabled():
|
|
217
|
+
raise ImportError # fall through to untraced path
|
|
218
|
+
|
|
219
|
+
with create_evaluation_span(eval_name) as span:
|
|
220
|
+
result = eng.run(eval_name, inputs, model=model, prompt=prompt, config=config)
|
|
221
|
+
# Enrich the span with the result
|
|
222
|
+
if hasattr(span, "set_attribute"):
|
|
223
|
+
span.set_attribute("gen_ai.span.kind", "EVALUATOR")
|
|
224
|
+
span.set_attribute("gen_ai.evaluation.name", eval_name)
|
|
225
|
+
enrich_span_with_evaluation(
|
|
226
|
+
metric_name=result.eval_name,
|
|
227
|
+
score=result.score if result.score is not None else 0.0,
|
|
228
|
+
reason=result.reason,
|
|
229
|
+
latency_ms=result.latency_ms,
|
|
230
|
+
span=span if hasattr(span, "set_attribute") else None,
|
|
231
|
+
)
|
|
232
|
+
return result
|
|
233
|
+
except ImportError:
|
|
234
|
+
pass
|
|
235
|
+
|
|
236
|
+
return eng.run(eval_name, inputs, model=model, prompt=prompt, config=config)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _augment_with_llm(
|
|
240
|
+
result: EvalResult,
|
|
241
|
+
*,
|
|
242
|
+
effective_name: str,
|
|
243
|
+
inputs: Dict[str, Any],
|
|
244
|
+
model: Optional[str],
|
|
245
|
+
resolved_engine: str,
|
|
246
|
+
feedback_store: Optional[Any] = None,
|
|
247
|
+
) -> EvalResult:
|
|
248
|
+
"""Augment a local heuristic result with LLM judgment.
|
|
249
|
+
|
|
250
|
+
Called only when augment=True. Validates preconditions and raises
|
|
251
|
+
clear errors instead of silently skipping.
|
|
252
|
+
"""
|
|
253
|
+
if not model:
|
|
254
|
+
raise ValueError(
|
|
255
|
+
"augment=True requires a model= parameter "
|
|
256
|
+
"(e.g. model='gemini/gemini-2.5-flash')."
|
|
257
|
+
)
|
|
258
|
+
if is_turing_model(model):
|
|
259
|
+
raise ValueError(
|
|
260
|
+
f"augment=True is not compatible with Turing models (got '{model}'). "
|
|
261
|
+
"Use a LiteLLM model string like 'gemini/gemini-2.5-flash'."
|
|
262
|
+
)
|
|
263
|
+
if resolved_engine != "local":
|
|
264
|
+
raise ValueError(
|
|
265
|
+
f"augment=True only works with local metrics, but engine resolved "
|
|
266
|
+
f"to '{resolved_engine}'. Remove engine= or set engine='local'."
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
from ..local.registry import get_registry
|
|
270
|
+
metric_cls = get_registry().get(effective_name)
|
|
271
|
+
|
|
272
|
+
if metric_cls is None or not getattr(metric_cls, "supports_llm_judge", False):
|
|
273
|
+
raise ValueError(
|
|
274
|
+
f"Metric '{effective_name}' does not support LLM augmentation "
|
|
275
|
+
f"(supports_llm_judge is False). Only judgment metrics like "
|
|
276
|
+
f"faithfulness, hallucination_score, task_completion, etc. can be augmented."
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
if result.status != "completed":
|
|
280
|
+
result.metadata["engine"] = "local"
|
|
281
|
+
return result
|
|
282
|
+
|
|
283
|
+
description = getattr(metric_cls, "judge_description", "") or ""
|
|
284
|
+
|
|
285
|
+
from .judge_prompt import build_judge_prompt
|
|
286
|
+
judge_prompt = build_judge_prompt(effective_name, description, inputs, result)
|
|
287
|
+
|
|
288
|
+
# Retrieve feedback-based few-shot examples if available
|
|
289
|
+
llm_config = None
|
|
290
|
+
try:
|
|
291
|
+
from ..feedback.hooks import retrieve_feedback_config
|
|
292
|
+
llm_config = retrieve_feedback_config(
|
|
293
|
+
metric_name=effective_name,
|
|
294
|
+
inputs=inputs,
|
|
295
|
+
store=feedback_store,
|
|
296
|
+
)
|
|
297
|
+
except ImportError:
|
|
298
|
+
pass # feedback module not installed / not configured
|
|
299
|
+
|
|
300
|
+
llm_eng = LLMEngine()
|
|
301
|
+
try:
|
|
302
|
+
augmented = llm_eng.run(
|
|
303
|
+
effective_name, inputs,
|
|
304
|
+
model=model, prompt=judge_prompt, config=llm_config,
|
|
305
|
+
)
|
|
306
|
+
augmented.metadata["engine"] = "local+llm"
|
|
307
|
+
if llm_config and llm_config.get("few_shot_examples"):
|
|
308
|
+
augmented.metadata["feedback_examples_used"] = len(llm_config["few_shot_examples"])
|
|
309
|
+
return augmented
|
|
310
|
+
except Exception as exc:
|
|
311
|
+
warnings.warn(
|
|
312
|
+
f"LLM augmentation failed for '{effective_name}', "
|
|
313
|
+
f"falling back to local heuristic result: {exc}",
|
|
314
|
+
RuntimeWarning,
|
|
315
|
+
stacklevel=3,
|
|
316
|
+
)
|
|
317
|
+
result.metadata["engine"] = "local"
|
|
318
|
+
result.metadata["augment_error"] = str(exc)
|
|
319
|
+
return result
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Generic LLM judge prompt builder for augmenting local metric results.
|
|
3
|
+
|
|
4
|
+
When a local metric supports LLM augmentation (supports_llm_judge = True)
|
|
5
|
+
and the user passes augment=True, the local heuristic runs first, then
|
|
6
|
+
this module builds a prompt that feeds the heuristic scores + reasoning
|
|
7
|
+
to the LLM for refinement.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from typing import Any, Dict
|
|
12
|
+
|
|
13
|
+
from .result import EvalResult
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
_PROMPT_TEMPLATE = """\
|
|
17
|
+
You are an expert AI evaluator. Your task is to evaluate: **{metric_name}**
|
|
18
|
+
|
|
19
|
+
## What this metric measures
|
|
20
|
+
{description}
|
|
21
|
+
|
|
22
|
+
## Local analysis (heuristic pre-screening)
|
|
23
|
+
The following analysis was produced by a fast, deterministic heuristic.
|
|
24
|
+
Use it as a starting point — it may be accurate, but it cannot reason
|
|
25
|
+
about semantics the way you can.
|
|
26
|
+
|
|
27
|
+
Score: {local_score}
|
|
28
|
+
Reasoning: {local_reason}
|
|
29
|
+
|
|
30
|
+
## Raw data
|
|
31
|
+
{formatted_inputs}
|
|
32
|
+
|
|
33
|
+
## Instructions
|
|
34
|
+
Using the local analysis as a starting point and the raw data for verification,
|
|
35
|
+
provide your refined judgment.
|
|
36
|
+
|
|
37
|
+
- If the heuristic score seems correct, confirm it with your own reasoning.
|
|
38
|
+
- If you find the heuristic missed something or was too harsh/lenient, adjust.
|
|
39
|
+
- Score from 0.0 (worst) to 1.0 (best).
|
|
40
|
+
|
|
41
|
+
Return ONLY a JSON object: {{"score": <float>, "reason": "<brief explanation>"}}\
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _format_inputs(inputs: Dict[str, Any]) -> str:
|
|
46
|
+
"""Format evaluation inputs for the prompt, keeping it concise."""
|
|
47
|
+
parts = []
|
|
48
|
+
for key, value in inputs.items():
|
|
49
|
+
if value is None:
|
|
50
|
+
continue
|
|
51
|
+
if isinstance(value, str):
|
|
52
|
+
display = value if len(value) <= 1000 else value[:1000] + "..."
|
|
53
|
+
parts.append(f"**{key}**:\n{display}")
|
|
54
|
+
elif isinstance(value, (list, dict)):
|
|
55
|
+
try:
|
|
56
|
+
dumped = json.dumps(value, indent=2, default=str)
|
|
57
|
+
if len(dumped) > 1500:
|
|
58
|
+
dumped = dumped[:1500] + "\n..."
|
|
59
|
+
parts.append(f"**{key}**:\n```json\n{dumped}\n```")
|
|
60
|
+
except (TypeError, ValueError):
|
|
61
|
+
parts.append(f"**{key}**: {str(value)[:500]}")
|
|
62
|
+
else:
|
|
63
|
+
parts.append(f"**{key}**: {value}")
|
|
64
|
+
return "\n\n".join(parts) if parts else "(no inputs)"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def build_judge_prompt(
|
|
68
|
+
metric_name: str,
|
|
69
|
+
description: str,
|
|
70
|
+
inputs: Dict[str, Any],
|
|
71
|
+
local_result: EvalResult,
|
|
72
|
+
) -> str:
|
|
73
|
+
"""Build an LLM judge prompt that includes local heuristic results.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
metric_name: The metric being evaluated (e.g. "faithfulness").
|
|
77
|
+
description: What this metric measures (from metric_cls.judge_description).
|
|
78
|
+
inputs: The raw evaluation inputs (output, context, trajectory, etc.).
|
|
79
|
+
local_result: The EvalResult from the local heuristic engine.
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
A formatted prompt string ready for the LLM engine.
|
|
83
|
+
"""
|
|
84
|
+
return _PROMPT_TEMPLATE.format(
|
|
85
|
+
metric_name=metric_name,
|
|
86
|
+
description=description or f"Evaluate the quality of the output for '{metric_name}'.",
|
|
87
|
+
local_score=local_result.score if local_result.score is not None else "N/A",
|
|
88
|
+
local_reason=local_result.reason or "No reasoning provided.",
|
|
89
|
+
formatted_inputs=_format_inputs(inputs),
|
|
90
|
+
)
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Auto-generate grading criteria from a short description.
|
|
3
|
+
|
|
4
|
+
Usage:
|
|
5
|
+
from fi.evals import evaluate
|
|
6
|
+
|
|
7
|
+
# Instead of writing a detailed rubric yourself:
|
|
8
|
+
result = evaluate(
|
|
9
|
+
prompt="product description accuracy for e-commerce images",
|
|
10
|
+
output="A red cotton t-shirt with v-neck",
|
|
11
|
+
image_url="https://example.com/tshirt.jpg",
|
|
12
|
+
engine="llm",
|
|
13
|
+
model="gemini/gemini-2.5-flash",
|
|
14
|
+
generate_prompt=True,
|
|
15
|
+
)
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
from typing import Any, Dict
|
|
20
|
+
|
|
21
|
+
_CACHE: Dict[str, str] = {}
|
|
22
|
+
|
|
23
|
+
_META_PROMPT = """\
|
|
24
|
+
You are an expert prompt engineer specializing in LLM evaluation rubrics.
|
|
25
|
+
|
|
26
|
+
Given a short description of what to evaluate, generate a detailed grading \
|
|
27
|
+
criteria that an LLM judge can use to score inputs on a 0.0–1.0 scale.
|
|
28
|
+
|
|
29
|
+
The criteria MUST:
|
|
30
|
+
- Be specific and actionable (not vague)
|
|
31
|
+
- Define what 1.0, 0.5, and 0.0 look like
|
|
32
|
+
- Reference the input fields the judge will receive: {input_keys}
|
|
33
|
+
- Be 4-8 sentences maximum
|
|
34
|
+
|
|
35
|
+
Description of what to evaluate:
|
|
36
|
+
{description}
|
|
37
|
+
|
|
38
|
+
Return ONLY the grading criteria text. No JSON, no markdown, no preamble.\
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def generate_grading_criteria(
|
|
43
|
+
description: str,
|
|
44
|
+
model: str,
|
|
45
|
+
inputs: Dict[str, Any],
|
|
46
|
+
*,
|
|
47
|
+
cache: bool = True,
|
|
48
|
+
) -> str:
|
|
49
|
+
"""Generate a detailed grading criteria from a short description.
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
description: Short description of what to evaluate
|
|
53
|
+
(e.g. "product description accuracy for images").
|
|
54
|
+
model: LiteLLM model string (e.g. "gemini/gemini-2.5-flash").
|
|
55
|
+
inputs: The evaluation inputs dict — used to tell the generator
|
|
56
|
+
which fields the judge will receive.
|
|
57
|
+
cache: Cache results per (description, model) for the session.
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
A detailed grading criteria string.
|
|
61
|
+
"""
|
|
62
|
+
cache_key = hashlib.md5(f"{description}:{model}".encode()).hexdigest()
|
|
63
|
+
if cache and cache_key in _CACHE:
|
|
64
|
+
return _CACHE[cache_key]
|
|
65
|
+
|
|
66
|
+
input_keys = ", ".join(sorted(inputs.keys())) or "output"
|
|
67
|
+
|
|
68
|
+
prompt = _META_PROMPT.format(
|
|
69
|
+
description=description,
|
|
70
|
+
input_keys=input_keys,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
import litellm
|
|
74
|
+
response = litellm.completion(
|
|
75
|
+
model=model,
|
|
76
|
+
messages=[{"role": "user", "content": prompt}],
|
|
77
|
+
)
|
|
78
|
+
criteria = response.choices[0].message.content.strip()
|
|
79
|
+
|
|
80
|
+
if cache:
|
|
81
|
+
_CACHE[cache_key] = criteria
|
|
82
|
+
|
|
83
|
+
return criteria
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Unified registry — resolves eval names to engines.
|
|
3
|
+
|
|
4
|
+
Routing is purely based on user-provided kwargs:
|
|
5
|
+
1. Explicit engine= → use that
|
|
6
|
+
2. Turing model → "turing"
|
|
7
|
+
3. Any other model → "llm"
|
|
8
|
+
4. No model → "local" (default)
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class Turing(str, Enum):
|
|
16
|
+
"""Model options for the Turing (FutureAGI) cloud engine."""
|
|
17
|
+
|
|
18
|
+
FLASH = "turing_flash"
|
|
19
|
+
SMALL = "turing_small"
|
|
20
|
+
LARGE = "turing_large"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
TURING_MODEL_PREFIXES = ("turing",)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def is_turing_model(model: Optional[str]) -> bool:
|
|
27
|
+
"""Check if the model string indicates a Turing platform model."""
|
|
28
|
+
if not model:
|
|
29
|
+
return False
|
|
30
|
+
return model.lower().startswith(TURING_MODEL_PREFIXES)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def resolve_engine(
|
|
34
|
+
name: Optional[str] = None,
|
|
35
|
+
*,
|
|
36
|
+
model: Optional[str] = None,
|
|
37
|
+
prompt: Optional[str] = None,
|
|
38
|
+
engine: Optional[str] = None,
|
|
39
|
+
) -> str:
|
|
40
|
+
"""Auto-detect which engine based on user-provided kwargs.
|
|
41
|
+
|
|
42
|
+
Priority:
|
|
43
|
+
1. Explicit engine kwarg
|
|
44
|
+
2. Turing model → "turing"
|
|
45
|
+
3. Any other model → "llm"
|
|
46
|
+
4. No model → "local" (default, fails gracefully if metric not found)
|
|
47
|
+
"""
|
|
48
|
+
if engine:
|
|
49
|
+
return engine
|
|
50
|
+
|
|
51
|
+
if is_turing_model(model):
|
|
52
|
+
return "turing"
|
|
53
|
+
|
|
54
|
+
if model:
|
|
55
|
+
return "llm"
|
|
56
|
+
|
|
57
|
+
return "local"
|
fi/evals/core/result.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Unified result types for all evaluations.
|
|
3
|
+
|
|
4
|
+
EvalResult is the ONE result type returned by evaluate() and all engines.
|
|
5
|
+
BatchResult wraps multiple EvalResults when running several evals at once.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Any, Dict, Iterator, List, Optional
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class EvalResult:
|
|
14
|
+
"""The unified result type returned by evaluate() and all engines."""
|
|
15
|
+
|
|
16
|
+
eval_name: str
|
|
17
|
+
score: Optional[float] = None
|
|
18
|
+
passed: Optional[bool] = None
|
|
19
|
+
reason: str = ""
|
|
20
|
+
latency_ms: float = 0.0
|
|
21
|
+
status: str = "completed"
|
|
22
|
+
error: Optional[str] = None
|
|
23
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
24
|
+
|
|
25
|
+
def __post_init__(self):
|
|
26
|
+
# Auto-derive passed from score if not explicitly set
|
|
27
|
+
if self.passed is None and self.score is not None:
|
|
28
|
+
self.passed = self.score >= 0.5
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class BatchResult:
|
|
33
|
+
"""Returned when multiple evals are run via evaluate()."""
|
|
34
|
+
|
|
35
|
+
results: List[EvalResult] = field(default_factory=list)
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def success_rate(self) -> float:
|
|
39
|
+
if not self.results:
|
|
40
|
+
return 0.0
|
|
41
|
+
completed = sum(1 for r in self.results if r.status == "completed")
|
|
42
|
+
return completed / len(self.results)
|
|
43
|
+
|
|
44
|
+
def get(self, name: str) -> Optional[EvalResult]:
|
|
45
|
+
"""Get result by eval name."""
|
|
46
|
+
for r in self.results:
|
|
47
|
+
if r.eval_name == name:
|
|
48
|
+
return r
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
def __iter__(self) -> Iterator[EvalResult]:
|
|
52
|
+
return iter(self.results)
|
|
53
|
+
|
|
54
|
+
def __len__(self) -> int:
|
|
55
|
+
return len(self.results)
|