agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Execution mode selector for evaluations.
|
|
2
|
+
|
|
3
|
+
This module defines execution modes that determine how evaluations run:
|
|
4
|
+
- LOCAL: Run all evaluations locally using heuristic metrics (no API calls)
|
|
5
|
+
- CLOUD: Run all evaluations via the cloud API
|
|
6
|
+
- HYBRID: Automatically route each evaluation to local or cloud based on metric type
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from enum import Enum
|
|
10
|
+
from typing import Set
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class RoutingMode(Enum):
|
|
14
|
+
"""Defines how evaluations should be executed."""
|
|
15
|
+
|
|
16
|
+
LOCAL = "local"
|
|
17
|
+
"""Run evaluations locally using heuristic metrics only."""
|
|
18
|
+
|
|
19
|
+
CLOUD = "cloud"
|
|
20
|
+
"""Run all evaluations via the cloud API."""
|
|
21
|
+
|
|
22
|
+
HYBRID = "hybrid"
|
|
23
|
+
"""Automatically choose local or cloud based on metric capabilities."""
|
|
24
|
+
|
|
25
|
+
def __str__(self) -> str:
|
|
26
|
+
return self.value
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Set of metric names that can be run locally (heuristic metrics)
|
|
30
|
+
LOCAL_CAPABLE_METRICS: Set[str] = {
|
|
31
|
+
# String metrics
|
|
32
|
+
"regex",
|
|
33
|
+
"contains",
|
|
34
|
+
"contains_all",
|
|
35
|
+
"contains_any",
|
|
36
|
+
"contains_none",
|
|
37
|
+
"one_line",
|
|
38
|
+
"contains_email",
|
|
39
|
+
"is_email",
|
|
40
|
+
"contains_link",
|
|
41
|
+
"contains_valid_link",
|
|
42
|
+
"equals",
|
|
43
|
+
"starts_with",
|
|
44
|
+
"ends_with",
|
|
45
|
+
"length_less_than",
|
|
46
|
+
"length_greater_than",
|
|
47
|
+
"length_between",
|
|
48
|
+
|
|
49
|
+
# JSON metrics
|
|
50
|
+
"contains_json",
|
|
51
|
+
"is_json",
|
|
52
|
+
"json_schema",
|
|
53
|
+
|
|
54
|
+
# Similarity metrics
|
|
55
|
+
"bleu_score",
|
|
56
|
+
"rouge_score",
|
|
57
|
+
"recall_score",
|
|
58
|
+
"levenshtein_similarity",
|
|
59
|
+
"numeric_similarity",
|
|
60
|
+
"embedding_similarity",
|
|
61
|
+
"semantic_list_contains",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def can_run_locally(metric_name: str) -> bool:
|
|
66
|
+
"""Check if a metric can be run locally.
|
|
67
|
+
|
|
68
|
+
Args:
|
|
69
|
+
metric_name: The name of the metric to check.
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
True if the metric can run locally, False otherwise.
|
|
73
|
+
"""
|
|
74
|
+
return metric_name.lower() in LOCAL_CAPABLE_METRICS
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def select_routing_mode(
|
|
78
|
+
metric_name: str,
|
|
79
|
+
preferred_mode: RoutingMode,
|
|
80
|
+
force_local: bool = False,
|
|
81
|
+
force_cloud: bool = False,
|
|
82
|
+
) -> RoutingMode:
|
|
83
|
+
"""Select the execution mode for a metric based on preferences and capabilities.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
metric_name: The name of the metric.
|
|
87
|
+
preferred_mode: The user's preferred execution mode.
|
|
88
|
+
force_local: If True, always try local execution (error if not possible).
|
|
89
|
+
force_cloud: If True, always use cloud execution.
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
The selected execution mode.
|
|
93
|
+
|
|
94
|
+
Raises:
|
|
95
|
+
ValueError: If force_local is True but the metric cannot run locally.
|
|
96
|
+
"""
|
|
97
|
+
if force_cloud:
|
|
98
|
+
return RoutingMode.CLOUD
|
|
99
|
+
|
|
100
|
+
if force_local:
|
|
101
|
+
if not can_run_locally(metric_name):
|
|
102
|
+
raise ValueError(
|
|
103
|
+
f"Metric '{metric_name}' cannot run locally. "
|
|
104
|
+
f"Local-capable metrics: {sorted(LOCAL_CAPABLE_METRICS)}"
|
|
105
|
+
)
|
|
106
|
+
return RoutingMode.LOCAL
|
|
107
|
+
|
|
108
|
+
if preferred_mode == RoutingMode.LOCAL:
|
|
109
|
+
if can_run_locally(metric_name):
|
|
110
|
+
return RoutingMode.LOCAL
|
|
111
|
+
# Fall back to cloud if metric can't run locally
|
|
112
|
+
return RoutingMode.CLOUD
|
|
113
|
+
|
|
114
|
+
if preferred_mode == RoutingMode.HYBRID:
|
|
115
|
+
# In hybrid mode, prefer local for capable metrics
|
|
116
|
+
if can_run_locally(metric_name):
|
|
117
|
+
return RoutingMode.LOCAL
|
|
118
|
+
return RoutingMode.CLOUD
|
|
119
|
+
|
|
120
|
+
# Default: CLOUD mode
|
|
121
|
+
return RoutingMode.CLOUD
|
fi/evals/local/llm.py
ADDED
|
@@ -0,0 +1,489 @@
|
|
|
1
|
+
"""Local LLM integration for running LLM-as-judge evaluations without cloud API.
|
|
2
|
+
|
|
3
|
+
This module provides support for local LLM inference via Ollama, enabling
|
|
4
|
+
offline LLM-based evaluations for air-gapped environments or faster iteration.
|
|
5
|
+
|
|
6
|
+
Example:
|
|
7
|
+
>>> from fi.evals.local.llm import OllamaLLM, LocalLLMConfig
|
|
8
|
+
>>>
|
|
9
|
+
>>> # Initialize with default model
|
|
10
|
+
>>> llm = OllamaLLM()
|
|
11
|
+
>>>
|
|
12
|
+
>>> # Generate completion
|
|
13
|
+
>>> response = llm.generate("What is 2+2?")
|
|
14
|
+
>>>
|
|
15
|
+
>>> # Use as LLM judge
|
|
16
|
+
>>> result = llm.judge(
|
|
17
|
+
... query="What is the capital of France?",
|
|
18
|
+
... response="The capital of France is Paris.",
|
|
19
|
+
... criteria="Evaluate if the response correctly answers the question."
|
|
20
|
+
... )
|
|
21
|
+
>>> print(result["score"])
|
|
22
|
+
0.9
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from typing import Any, Dict, List, Optional
|
|
27
|
+
import json
|
|
28
|
+
import logging
|
|
29
|
+
import re
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class LocalLLMConfig:
|
|
36
|
+
"""Configuration for local LLM.
|
|
37
|
+
|
|
38
|
+
Attributes:
|
|
39
|
+
model: The model name to use (e.g., "llama3.2", "mistral", "phi3").
|
|
40
|
+
base_url: The Ollama API base URL.
|
|
41
|
+
temperature: Sampling temperature (0.0 for deterministic).
|
|
42
|
+
max_tokens: Maximum tokens to generate.
|
|
43
|
+
timeout: Request timeout in seconds.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
model: str = "llama3.2"
|
|
47
|
+
base_url: str = "http://localhost:11434"
|
|
48
|
+
temperature: float = 0.0
|
|
49
|
+
max_tokens: int = 1024
|
|
50
|
+
timeout: int = 120
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class OllamaLLM:
|
|
54
|
+
"""Interface to Ollama for local LLM inference.
|
|
55
|
+
|
|
56
|
+
This class provides methods for text generation and LLM-as-judge
|
|
57
|
+
evaluations using locally running Ollama models.
|
|
58
|
+
|
|
59
|
+
Example:
|
|
60
|
+
>>> llm = OllamaLLM()
|
|
61
|
+
>>> llm.is_available()
|
|
62
|
+
True
|
|
63
|
+
>>> response = llm.generate("Hello, how are you?")
|
|
64
|
+
"I'm doing well, thank you for asking!"
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
def __init__(
|
|
68
|
+
self,
|
|
69
|
+
config: Optional[LocalLLMConfig] = None,
|
|
70
|
+
auto_check: bool = True,
|
|
71
|
+
) -> None:
|
|
72
|
+
"""Initialize the Ollama LLM interface.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
config: Configuration for the LLM.
|
|
76
|
+
auto_check: If True, check Ollama availability on init.
|
|
77
|
+
"""
|
|
78
|
+
self.config = config or LocalLLMConfig()
|
|
79
|
+
self._available: Optional[bool] = None
|
|
80
|
+
self._models: Optional[List[str]] = None
|
|
81
|
+
|
|
82
|
+
if auto_check:
|
|
83
|
+
self._check_availability()
|
|
84
|
+
|
|
85
|
+
def _check_availability(self) -> None:
|
|
86
|
+
"""Check if Ollama is available and the model is installed."""
|
|
87
|
+
try:
|
|
88
|
+
import requests
|
|
89
|
+
|
|
90
|
+
response = requests.get(
|
|
91
|
+
f"{self.config.base_url}/api/tags",
|
|
92
|
+
timeout=5
|
|
93
|
+
)
|
|
94
|
+
response.raise_for_status()
|
|
95
|
+
|
|
96
|
+
data = response.json()
|
|
97
|
+
self._models = [m.get("name", "") for m in data.get("models", [])]
|
|
98
|
+
self._available = True
|
|
99
|
+
|
|
100
|
+
# Check if requested model is available
|
|
101
|
+
model_base = self.config.model.split(":")[0]
|
|
102
|
+
if not any(model_base in m for m in self._models):
|
|
103
|
+
logger.warning(
|
|
104
|
+
f"Model '{self.config.model}' not found in Ollama. "
|
|
105
|
+
f"Available models: {self._models}"
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
except Exception as e:
|
|
109
|
+
self._available = False
|
|
110
|
+
logger.warning(f"Ollama not available: {e}")
|
|
111
|
+
|
|
112
|
+
def is_available(self) -> bool:
|
|
113
|
+
"""Check if Ollama is available.
|
|
114
|
+
|
|
115
|
+
Returns:
|
|
116
|
+
True if Ollama is running and accessible.
|
|
117
|
+
"""
|
|
118
|
+
if self._available is None:
|
|
119
|
+
self._check_availability()
|
|
120
|
+
return self._available or False
|
|
121
|
+
|
|
122
|
+
def list_models(self) -> List[str]:
|
|
123
|
+
"""List available models in Ollama.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
List of installed model names.
|
|
127
|
+
"""
|
|
128
|
+
if self._models is None:
|
|
129
|
+
self._check_availability()
|
|
130
|
+
return self._models or []
|
|
131
|
+
|
|
132
|
+
def generate(
|
|
133
|
+
self,
|
|
134
|
+
prompt: str,
|
|
135
|
+
system: Optional[str] = None,
|
|
136
|
+
temperature: Optional[float] = None,
|
|
137
|
+
max_tokens: Optional[int] = None,
|
|
138
|
+
) -> str:
|
|
139
|
+
"""Generate a completion using Ollama.
|
|
140
|
+
|
|
141
|
+
Args:
|
|
142
|
+
prompt: The user prompt.
|
|
143
|
+
system: Optional system prompt.
|
|
144
|
+
temperature: Override temperature for this request.
|
|
145
|
+
max_tokens: Override max tokens for this request.
|
|
146
|
+
|
|
147
|
+
Returns:
|
|
148
|
+
The generated text response.
|
|
149
|
+
|
|
150
|
+
Raises:
|
|
151
|
+
ConnectionError: If Ollama is not available.
|
|
152
|
+
RuntimeError: If generation fails.
|
|
153
|
+
"""
|
|
154
|
+
if not self.is_available():
|
|
155
|
+
raise ConnectionError(
|
|
156
|
+
f"Cannot connect to Ollama at {self.config.base_url}. "
|
|
157
|
+
"Make sure Ollama is running: `ollama serve`"
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
import requests
|
|
161
|
+
|
|
162
|
+
payload = {
|
|
163
|
+
"model": self.config.model,
|
|
164
|
+
"prompt": prompt,
|
|
165
|
+
"stream": False,
|
|
166
|
+
"options": {
|
|
167
|
+
"temperature": temperature if temperature is not None else self.config.temperature,
|
|
168
|
+
"num_predict": max_tokens if max_tokens is not None else self.config.max_tokens,
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
if system:
|
|
173
|
+
payload["system"] = system
|
|
174
|
+
|
|
175
|
+
try:
|
|
176
|
+
response = requests.post(
|
|
177
|
+
f"{self.config.base_url}/api/generate",
|
|
178
|
+
json=payload,
|
|
179
|
+
timeout=self.config.timeout,
|
|
180
|
+
)
|
|
181
|
+
response.raise_for_status()
|
|
182
|
+
|
|
183
|
+
return response.json().get("response", "")
|
|
184
|
+
|
|
185
|
+
except requests.exceptions.Timeout:
|
|
186
|
+
raise RuntimeError(
|
|
187
|
+
f"Ollama request timed out after {self.config.timeout}s"
|
|
188
|
+
)
|
|
189
|
+
except requests.exceptions.RequestException as e:
|
|
190
|
+
raise RuntimeError(f"Ollama request failed: {e}")
|
|
191
|
+
|
|
192
|
+
def chat(
|
|
193
|
+
self,
|
|
194
|
+
messages: List[Dict[str, str]],
|
|
195
|
+
temperature: Optional[float] = None,
|
|
196
|
+
max_tokens: Optional[int] = None,
|
|
197
|
+
) -> str:
|
|
198
|
+
"""Chat completion using Ollama's chat API.
|
|
199
|
+
|
|
200
|
+
Args:
|
|
201
|
+
messages: List of message dictionaries with 'role' and 'content'.
|
|
202
|
+
temperature: Override temperature for this request.
|
|
203
|
+
max_tokens: Override max tokens for this request.
|
|
204
|
+
|
|
205
|
+
Returns:
|
|
206
|
+
The assistant's response.
|
|
207
|
+
"""
|
|
208
|
+
if not self.is_available():
|
|
209
|
+
raise ConnectionError(
|
|
210
|
+
f"Cannot connect to Ollama at {self.config.base_url}. "
|
|
211
|
+
"Make sure Ollama is running: `ollama serve`"
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
import requests
|
|
215
|
+
|
|
216
|
+
payload = {
|
|
217
|
+
"model": self.config.model,
|
|
218
|
+
"messages": messages,
|
|
219
|
+
"stream": False,
|
|
220
|
+
"options": {
|
|
221
|
+
"temperature": temperature if temperature is not None else self.config.temperature,
|
|
222
|
+
"num_predict": max_tokens if max_tokens is not None else self.config.max_tokens,
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
try:
|
|
227
|
+
response = requests.post(
|
|
228
|
+
f"{self.config.base_url}/api/chat",
|
|
229
|
+
json=payload,
|
|
230
|
+
timeout=self.config.timeout,
|
|
231
|
+
)
|
|
232
|
+
response.raise_for_status()
|
|
233
|
+
|
|
234
|
+
return response.json().get("message", {}).get("content", "")
|
|
235
|
+
|
|
236
|
+
except requests.exceptions.RequestException as e:
|
|
237
|
+
raise RuntimeError(f"Ollama chat request failed: {e}")
|
|
238
|
+
|
|
239
|
+
def judge(
|
|
240
|
+
self,
|
|
241
|
+
query: str,
|
|
242
|
+
response: str,
|
|
243
|
+
criteria: str,
|
|
244
|
+
context: Optional[str] = None,
|
|
245
|
+
output_format: str = "json",
|
|
246
|
+
) -> Dict[str, Any]:
|
|
247
|
+
"""Use LLM as judge for evaluation.
|
|
248
|
+
|
|
249
|
+
Args:
|
|
250
|
+
query: The original query/question.
|
|
251
|
+
response: The response to evaluate.
|
|
252
|
+
criteria: The evaluation criteria/rubric.
|
|
253
|
+
context: Optional context/reference information.
|
|
254
|
+
output_format: Output format ("json" or "text").
|
|
255
|
+
|
|
256
|
+
Returns:
|
|
257
|
+
Dictionary with evaluation result:
|
|
258
|
+
- score: Float between 0 and 1
|
|
259
|
+
- passed: Boolean indicating if evaluation passed
|
|
260
|
+
- reason: Explanation of the evaluation
|
|
261
|
+
"""
|
|
262
|
+
system_prompt = """You are an AI evaluation judge. Evaluate the response based on the given criteria.
|
|
263
|
+
|
|
264
|
+
Your evaluation must be fair, consistent, and based solely on the criteria provided.
|
|
265
|
+
|
|
266
|
+
Output your evaluation as JSON with the following format:
|
|
267
|
+
{
|
|
268
|
+
"score": <float between 0.0 and 1.0>,
|
|
269
|
+
"passed": <true or false, based on whether score >= 0.5>,
|
|
270
|
+
"reason": "<brief explanation of your evaluation>"
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
Only output the JSON object, nothing else."""
|
|
274
|
+
|
|
275
|
+
user_prompt = f"""## Evaluation Criteria
|
|
276
|
+
{criteria}
|
|
277
|
+
|
|
278
|
+
## Query
|
|
279
|
+
{query}
|
|
280
|
+
|
|
281
|
+
## Response to Evaluate
|
|
282
|
+
{response}
|
|
283
|
+
"""
|
|
284
|
+
|
|
285
|
+
if context:
|
|
286
|
+
user_prompt += f"""
|
|
287
|
+
## Context/Reference
|
|
288
|
+
{context}
|
|
289
|
+
"""
|
|
290
|
+
|
|
291
|
+
user_prompt += """
|
|
292
|
+
## Your Evaluation (JSON only)"""
|
|
293
|
+
|
|
294
|
+
try:
|
|
295
|
+
result_text = self.generate(user_prompt, system=system_prompt)
|
|
296
|
+
return self._parse_judge_response(result_text)
|
|
297
|
+
|
|
298
|
+
except Exception as e:
|
|
299
|
+
logger.error(f"LLM judge failed: {e}")
|
|
300
|
+
return {
|
|
301
|
+
"score": 0.0,
|
|
302
|
+
"passed": False,
|
|
303
|
+
"reason": f"Evaluation failed: {str(e)}",
|
|
304
|
+
"error": True,
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
def _parse_judge_response(self, response: str) -> Dict[str, Any]:
|
|
308
|
+
"""Parse the LLM judge response into structured format.
|
|
309
|
+
|
|
310
|
+
Args:
|
|
311
|
+
response: The raw LLM response text.
|
|
312
|
+
|
|
313
|
+
Returns:
|
|
314
|
+
Parsed evaluation result dictionary.
|
|
315
|
+
"""
|
|
316
|
+
response = response.strip()
|
|
317
|
+
|
|
318
|
+
# Try direct JSON parse first
|
|
319
|
+
try:
|
|
320
|
+
result = json.loads(response)
|
|
321
|
+
return self._validate_judge_result(result)
|
|
322
|
+
except json.JSONDecodeError:
|
|
323
|
+
pass
|
|
324
|
+
|
|
325
|
+
# Try to extract JSON from markdown code block
|
|
326
|
+
code_block_pattern = r"```(?:json)?\s*([\s\S]*?)```"
|
|
327
|
+
match = re.search(code_block_pattern, response)
|
|
328
|
+
if match:
|
|
329
|
+
try:
|
|
330
|
+
result = json.loads(match.group(1).strip())
|
|
331
|
+
return self._validate_judge_result(result)
|
|
332
|
+
except json.JSONDecodeError:
|
|
333
|
+
pass
|
|
334
|
+
|
|
335
|
+
# Try to find JSON object in response
|
|
336
|
+
json_pattern = r"\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}"
|
|
337
|
+
match = re.search(json_pattern, response, re.DOTALL)
|
|
338
|
+
if match:
|
|
339
|
+
try:
|
|
340
|
+
result = json.loads(match.group())
|
|
341
|
+
return self._validate_judge_result(result)
|
|
342
|
+
except json.JSONDecodeError:
|
|
343
|
+
pass
|
|
344
|
+
|
|
345
|
+
# Fallback: try to extract score and reason from text
|
|
346
|
+
score = 0.5
|
|
347
|
+
score_pattern = r"(?:score|rating)[:\s]*([0-9]*\.?[0-9]+)"
|
|
348
|
+
score_match = re.search(score_pattern, response, re.IGNORECASE)
|
|
349
|
+
if score_match:
|
|
350
|
+
try:
|
|
351
|
+
score = float(score_match.group(1))
|
|
352
|
+
# Normalize if score is > 1 (e.g., 1-10 scale)
|
|
353
|
+
if score > 1:
|
|
354
|
+
score = score / 10
|
|
355
|
+
score = max(0.0, min(1.0, score))
|
|
356
|
+
except ValueError:
|
|
357
|
+
pass
|
|
358
|
+
|
|
359
|
+
return {
|
|
360
|
+
"score": score,
|
|
361
|
+
"passed": score >= 0.5,
|
|
362
|
+
"reason": response[:500] if response else "Unable to parse evaluation",
|
|
363
|
+
"parse_error": True,
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
def _validate_judge_result(self, result: Dict[str, Any]) -> Dict[str, Any]:
|
|
367
|
+
"""Validate and normalize the judge result.
|
|
368
|
+
|
|
369
|
+
Args:
|
|
370
|
+
result: Raw parsed result dictionary.
|
|
371
|
+
|
|
372
|
+
Returns:
|
|
373
|
+
Validated and normalized result.
|
|
374
|
+
"""
|
|
375
|
+
score = result.get("score", 0.5)
|
|
376
|
+
|
|
377
|
+
# Handle various score formats
|
|
378
|
+
if isinstance(score, str):
|
|
379
|
+
try:
|
|
380
|
+
score = float(score)
|
|
381
|
+
except ValueError:
|
|
382
|
+
score = 0.5
|
|
383
|
+
|
|
384
|
+
# Normalize score to 0-1 range
|
|
385
|
+
if score > 1:
|
|
386
|
+
score = score / 10 if score <= 10 else score / 100
|
|
387
|
+
score = max(0.0, min(1.0, float(score)))
|
|
388
|
+
|
|
389
|
+
passed = result.get("passed")
|
|
390
|
+
if passed is None:
|
|
391
|
+
passed = score >= 0.5
|
|
392
|
+
elif isinstance(passed, str):
|
|
393
|
+
passed = passed.lower() in ("true", "yes", "1", "pass")
|
|
394
|
+
|
|
395
|
+
reason = result.get("reason", result.get("explanation", ""))
|
|
396
|
+
if not isinstance(reason, str):
|
|
397
|
+
reason = str(reason)
|
|
398
|
+
|
|
399
|
+
return {
|
|
400
|
+
"score": score,
|
|
401
|
+
"passed": bool(passed),
|
|
402
|
+
"reason": reason,
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
def batch_judge(
|
|
406
|
+
self,
|
|
407
|
+
evaluations: List[Dict[str, Any]],
|
|
408
|
+
) -> List[Dict[str, Any]]:
|
|
409
|
+
"""Run multiple judge evaluations.
|
|
410
|
+
|
|
411
|
+
Args:
|
|
412
|
+
evaluations: List of evaluation specifications, each containing:
|
|
413
|
+
- query: The query
|
|
414
|
+
- response: The response to evaluate
|
|
415
|
+
- criteria: Evaluation criteria
|
|
416
|
+
- context: Optional context
|
|
417
|
+
|
|
418
|
+
Returns:
|
|
419
|
+
List of evaluation results.
|
|
420
|
+
"""
|
|
421
|
+
results = []
|
|
422
|
+
for eval_spec in evaluations:
|
|
423
|
+
result = self.judge(
|
|
424
|
+
query=eval_spec.get("query", ""),
|
|
425
|
+
response=eval_spec.get("response", ""),
|
|
426
|
+
criteria=eval_spec.get("criteria", ""),
|
|
427
|
+
context=eval_spec.get("context"),
|
|
428
|
+
)
|
|
429
|
+
results.append(result)
|
|
430
|
+
return results
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
class LocalLLMFactory:
|
|
434
|
+
"""Factory for creating local LLM instances.
|
|
435
|
+
|
|
436
|
+
This factory provides a unified interface for creating different
|
|
437
|
+
types of local LLM backends.
|
|
438
|
+
"""
|
|
439
|
+
|
|
440
|
+
_backends = {
|
|
441
|
+
"ollama": OllamaLLM,
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
@classmethod
|
|
445
|
+
def create(
|
|
446
|
+
cls,
|
|
447
|
+
backend: str = "ollama",
|
|
448
|
+
**kwargs,
|
|
449
|
+
) -> OllamaLLM:
|
|
450
|
+
"""Create a local LLM instance.
|
|
451
|
+
|
|
452
|
+
Args:
|
|
453
|
+
backend: The LLM backend to use (currently only "ollama").
|
|
454
|
+
**kwargs: Additional arguments passed to the LLM constructor.
|
|
455
|
+
|
|
456
|
+
Returns:
|
|
457
|
+
An initialized LLM instance.
|
|
458
|
+
|
|
459
|
+
Raises:
|
|
460
|
+
ValueError: If backend is not supported.
|
|
461
|
+
"""
|
|
462
|
+
if backend not in cls._backends:
|
|
463
|
+
raise ValueError(
|
|
464
|
+
f"Unsupported LLM backend: {backend}. "
|
|
465
|
+
f"Supported backends: {list(cls._backends.keys())}"
|
|
466
|
+
)
|
|
467
|
+
|
|
468
|
+
return cls._backends[backend](**kwargs)
|
|
469
|
+
|
|
470
|
+
@classmethod
|
|
471
|
+
def from_string(cls, spec: str) -> OllamaLLM:
|
|
472
|
+
"""Create a local LLM from a string specification.
|
|
473
|
+
|
|
474
|
+
Args:
|
|
475
|
+
spec: String specification in format "backend/model"
|
|
476
|
+
(e.g., "ollama/llama3.2").
|
|
477
|
+
|
|
478
|
+
Returns:
|
|
479
|
+
An initialized LLM instance.
|
|
480
|
+
"""
|
|
481
|
+
parts = spec.split("/", 1)
|
|
482
|
+
backend = parts[0].lower()
|
|
483
|
+
model = parts[1] if len(parts) > 1 else None
|
|
484
|
+
|
|
485
|
+
config = LocalLLMConfig()
|
|
486
|
+
if model:
|
|
487
|
+
config.model = model
|
|
488
|
+
|
|
489
|
+
return cls.create(backend, config=config)
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Local metrics module.
|
|
2
|
+
|
|
3
|
+
This module provides the base class and utilities for local metric implementations.
|
|
4
|
+
The actual metric implementations live in fi.evals.metrics.heuristics/ and are
|
|
5
|
+
registered via the LocalMetricRegistry.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
# Re-export commonly used types for convenience
|
|
9
|
+
from ...types import TextMetricInput, JsonMetricInput, EvalResult, BatchRunResult
|
|
10
|
+
from ...metrics.base_metric import BaseMetric
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"BaseMetric",
|
|
15
|
+
"TextMetricInput",
|
|
16
|
+
"JsonMetricInput",
|
|
17
|
+
"EvalResult",
|
|
18
|
+
"BatchRunResult",
|
|
19
|
+
]
|