agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,573 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Function Calling Evaluation Metrics.
|
|
3
|
+
|
|
4
|
+
AST-based, deterministic evaluation of LLM function calling.
|
|
5
|
+
Provides sub-10ms evaluation latency without LLM-as-judge dependency.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import ast
|
|
9
|
+
import json
|
|
10
|
+
from typing import Any, Dict, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
from ..base_metric import BaseMetric
|
|
13
|
+
from .types import FunctionCallInput, FunctionCall
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _parse_function_call(call: Union[FunctionCall, Dict, str, None]) -> Optional[FunctionCall]:
|
|
17
|
+
"""Parse various input formats into a FunctionCall object."""
|
|
18
|
+
if call is None:
|
|
19
|
+
return None
|
|
20
|
+
if isinstance(call, FunctionCall):
|
|
21
|
+
return call
|
|
22
|
+
if isinstance(call, dict):
|
|
23
|
+
# Get arguments, handling both dict and JSON string formats (OpenAI-style)
|
|
24
|
+
arguments = call.get("arguments", call.get("parameters", call.get("input", {})))
|
|
25
|
+
if isinstance(arguments, str):
|
|
26
|
+
try:
|
|
27
|
+
arguments = json.loads(arguments)
|
|
28
|
+
except json.JSONDecodeError:
|
|
29
|
+
arguments = {}
|
|
30
|
+
return FunctionCall(
|
|
31
|
+
name=call.get("name", call.get("function", "")),
|
|
32
|
+
arguments=arguments if isinstance(arguments, dict) else {}
|
|
33
|
+
)
|
|
34
|
+
if isinstance(call, str):
|
|
35
|
+
try:
|
|
36
|
+
parsed = json.loads(call)
|
|
37
|
+
return _parse_function_call(parsed)
|
|
38
|
+
except json.JSONDecodeError:
|
|
39
|
+
# Try to parse as Python AST (e.g., "get_weather(city='NYC')")
|
|
40
|
+
return _parse_ast_call(call)
|
|
41
|
+
return None
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _parse_ast_call(call_str: str) -> Optional[FunctionCall]:
|
|
45
|
+
"""Parse a function call string using AST."""
|
|
46
|
+
try:
|
|
47
|
+
# Wrap in expression for parsing
|
|
48
|
+
tree = ast.parse(call_str.strip(), mode='eval')
|
|
49
|
+
if isinstance(tree.body, ast.Call):
|
|
50
|
+
call_node = tree.body
|
|
51
|
+
func_name = ""
|
|
52
|
+
if isinstance(call_node.func, ast.Name):
|
|
53
|
+
func_name = call_node.func.id
|
|
54
|
+
elif isinstance(call_node.func, ast.Attribute):
|
|
55
|
+
func_name = call_node.func.attr
|
|
56
|
+
|
|
57
|
+
# Extract arguments
|
|
58
|
+
arguments = {}
|
|
59
|
+
|
|
60
|
+
# Handle positional arguments
|
|
61
|
+
for i, arg in enumerate(call_node.args):
|
|
62
|
+
arguments[f"__positional_{i}"] = _ast_to_value(arg)
|
|
63
|
+
|
|
64
|
+
# Handle keyword arguments
|
|
65
|
+
for kw in call_node.keywords:
|
|
66
|
+
if kw.arg:
|
|
67
|
+
arguments[kw.arg] = _ast_to_value(kw.value)
|
|
68
|
+
|
|
69
|
+
return FunctionCall(name=func_name, arguments=arguments)
|
|
70
|
+
except (SyntaxError, ValueError):
|
|
71
|
+
pass
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _ast_to_value(node: ast.expr) -> Any:
|
|
76
|
+
"""Convert an AST node to a Python value."""
|
|
77
|
+
# ast.Constant handles strings, numbers, booleans, None in Python 3.8+
|
|
78
|
+
if isinstance(node, ast.Constant):
|
|
79
|
+
return node.value
|
|
80
|
+
if isinstance(node, ast.List):
|
|
81
|
+
return [_ast_to_value(elt) for elt in node.elts]
|
|
82
|
+
if isinstance(node, ast.Dict):
|
|
83
|
+
return {
|
|
84
|
+
_ast_to_value(k): _ast_to_value(v)
|
|
85
|
+
for k, v in zip(node.keys, node.values)
|
|
86
|
+
if k is not None
|
|
87
|
+
}
|
|
88
|
+
if isinstance(node, ast.Name):
|
|
89
|
+
# Handle True, False, None as names (fallback)
|
|
90
|
+
if node.id == "True":
|
|
91
|
+
return True
|
|
92
|
+
elif node.id == "False":
|
|
93
|
+
return False
|
|
94
|
+
elif node.id == "None":
|
|
95
|
+
return None
|
|
96
|
+
return node.id
|
|
97
|
+
if isinstance(node, ast.Tuple):
|
|
98
|
+
return tuple(_ast_to_value(elt) for elt in node.elts)
|
|
99
|
+
if isinstance(node, ast.Set):
|
|
100
|
+
return {_ast_to_value(elt) for elt in node.elts}
|
|
101
|
+
return str(node)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _parse_function_calls(
|
|
105
|
+
calls: Union[FunctionCall, List[FunctionCall], Dict, str, List[Dict], None]
|
|
106
|
+
) -> List[FunctionCall]:
|
|
107
|
+
"""Parse various input formats into a list of FunctionCall objects."""
|
|
108
|
+
if calls is None:
|
|
109
|
+
return []
|
|
110
|
+
if isinstance(calls, list):
|
|
111
|
+
return [_parse_function_call(c) for c in calls if _parse_function_call(c)]
|
|
112
|
+
parsed = _parse_function_call(calls)
|
|
113
|
+
return [parsed] if parsed else []
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _types_compatible(actual: Any, expected: Any, strict: bool = False) -> bool:
|
|
117
|
+
"""Check if types are compatible."""
|
|
118
|
+
if strict:
|
|
119
|
+
return type(actual) is type(expected)
|
|
120
|
+
|
|
121
|
+
# Flexible type checking
|
|
122
|
+
if actual is None or expected is None:
|
|
123
|
+
return actual == expected
|
|
124
|
+
|
|
125
|
+
# Bool guard: bool is a subclass of int in Python, so must check before numeric
|
|
126
|
+
if isinstance(actual, bool) or isinstance(expected, bool):
|
|
127
|
+
return isinstance(actual, bool) and isinstance(expected, bool)
|
|
128
|
+
|
|
129
|
+
# Numeric compatibility (int/float)
|
|
130
|
+
if isinstance(actual, (int, float)) and isinstance(expected, (int, float)):
|
|
131
|
+
return True
|
|
132
|
+
|
|
133
|
+
# String compatibility
|
|
134
|
+
if isinstance(actual, str) and isinstance(expected, str):
|
|
135
|
+
return True
|
|
136
|
+
|
|
137
|
+
# List compatibility
|
|
138
|
+
if isinstance(actual, list) and isinstance(expected, list):
|
|
139
|
+
return True
|
|
140
|
+
|
|
141
|
+
# Dict compatibility
|
|
142
|
+
if isinstance(actual, dict) and isinstance(expected, dict):
|
|
143
|
+
return True
|
|
144
|
+
|
|
145
|
+
return type(actual) is type(expected)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _values_equal(actual: Any, expected: Any, strict_type: bool = False) -> bool:
|
|
149
|
+
"""Check if values are equal with optional type flexibility."""
|
|
150
|
+
if not _types_compatible(actual, expected, strict_type):
|
|
151
|
+
return False
|
|
152
|
+
|
|
153
|
+
# For numeric types, allow float/int comparison
|
|
154
|
+
if isinstance(actual, (int, float)) and isinstance(expected, (int, float)):
|
|
155
|
+
return float(actual) == float(expected)
|
|
156
|
+
|
|
157
|
+
# For strings, exact match (case-sensitive)
|
|
158
|
+
if isinstance(actual, str) and isinstance(expected, str):
|
|
159
|
+
return actual == expected
|
|
160
|
+
|
|
161
|
+
# For lists, recursive comparison
|
|
162
|
+
if isinstance(actual, list) and isinstance(expected, list):
|
|
163
|
+
if len(actual) != len(expected):
|
|
164
|
+
return False
|
|
165
|
+
return all(_values_equal(a, e, strict_type) for a, e in zip(actual, expected))
|
|
166
|
+
|
|
167
|
+
# For dicts, recursive comparison
|
|
168
|
+
if isinstance(actual, dict) and isinstance(expected, dict):
|
|
169
|
+
if set(actual.keys()) != set(expected.keys()):
|
|
170
|
+
return False
|
|
171
|
+
return all(
|
|
172
|
+
_values_equal(actual[k], expected[k], strict_type)
|
|
173
|
+
for k in expected.keys()
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
return actual == expected
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class FunctionNameMatch(BaseMetric[FunctionCallInput]):
|
|
180
|
+
"""
|
|
181
|
+
Evaluates if the function name matches the expected name.
|
|
182
|
+
|
|
183
|
+
Returns 1.0 if names match, 0.0 otherwise.
|
|
184
|
+
Fast, deterministic metric (~1ms).
|
|
185
|
+
"""
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def metric_name(self) -> str:
|
|
189
|
+
return "function_name_match"
|
|
190
|
+
|
|
191
|
+
def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
|
|
192
|
+
actual = _parse_function_call(inputs.response)
|
|
193
|
+
expected = _parse_function_call(inputs.expected_response)
|
|
194
|
+
|
|
195
|
+
if actual is None:
|
|
196
|
+
return {
|
|
197
|
+
"output": 0.0,
|
|
198
|
+
"reason": "Could not parse actual function call from response."
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
if expected is None:
|
|
202
|
+
return {
|
|
203
|
+
"output": 0.0,
|
|
204
|
+
"reason": "Could not parse expected function call. 'expected_response' is required."
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
if actual.name == expected.name:
|
|
208
|
+
return {
|
|
209
|
+
"output": 1.0,
|
|
210
|
+
"reason": f"Function name '{actual.name}' matches expected."
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
return {
|
|
214
|
+
"output": 0.0,
|
|
215
|
+
"reason": f"Function name mismatch: got '{actual.name}', expected '{expected.name}'."
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
class ParameterValidation(BaseMetric[FunctionCallInput]):
|
|
220
|
+
"""
|
|
221
|
+
Validates function call parameters against a schema.
|
|
222
|
+
|
|
223
|
+
Checks:
|
|
224
|
+
- Required parameters are present
|
|
225
|
+
- Parameter types match specification
|
|
226
|
+
- Enum constraints are satisfied
|
|
227
|
+
|
|
228
|
+
Returns score from 0.0 to 1.0 based on validation success.
|
|
229
|
+
"""
|
|
230
|
+
|
|
231
|
+
@property
|
|
232
|
+
def metric_name(self) -> str:
|
|
233
|
+
return "parameter_validation"
|
|
234
|
+
|
|
235
|
+
def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
|
|
236
|
+
actual = _parse_function_call(inputs.response)
|
|
237
|
+
|
|
238
|
+
if actual is None:
|
|
239
|
+
return {
|
|
240
|
+
"output": 0.0,
|
|
241
|
+
"reason": "Could not parse function call from response."
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
if not inputs.function_definitions:
|
|
245
|
+
return {
|
|
246
|
+
"output": 0.0,
|
|
247
|
+
"reason": "No function definitions provided for validation."
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
# Find the matching function definition
|
|
251
|
+
func_def = None
|
|
252
|
+
for fd in inputs.function_definitions:
|
|
253
|
+
if fd.name == actual.name:
|
|
254
|
+
func_def = fd
|
|
255
|
+
break
|
|
256
|
+
|
|
257
|
+
if func_def is None:
|
|
258
|
+
return {
|
|
259
|
+
"output": 0.0,
|
|
260
|
+
"reason": f"Function '{actual.name}' not found in definitions."
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
errors = []
|
|
264
|
+
total_checks = 0
|
|
265
|
+
passed_checks = 0
|
|
266
|
+
|
|
267
|
+
for param_spec in func_def.parameters:
|
|
268
|
+
total_checks += 1
|
|
269
|
+
|
|
270
|
+
# Check required parameters
|
|
271
|
+
if param_spec.required and param_spec.name not in actual.arguments:
|
|
272
|
+
errors.append(f"Missing required parameter: {param_spec.name}")
|
|
273
|
+
continue
|
|
274
|
+
|
|
275
|
+
if param_spec.name in actual.arguments:
|
|
276
|
+
value = actual.arguments[param_spec.name]
|
|
277
|
+
|
|
278
|
+
# Type checking
|
|
279
|
+
if not self._check_type(value, param_spec.type, inputs.strict_type_check):
|
|
280
|
+
errors.append(
|
|
281
|
+
f"Parameter '{param_spec.name}' has wrong type: "
|
|
282
|
+
f"expected {param_spec.type}, got {type(value).__name__}"
|
|
283
|
+
)
|
|
284
|
+
continue
|
|
285
|
+
|
|
286
|
+
# Enum checking
|
|
287
|
+
if param_spec.enum and value not in param_spec.enum:
|
|
288
|
+
errors.append(
|
|
289
|
+
f"Parameter '{param_spec.name}' value '{value}' not in allowed values: {param_spec.enum}"
|
|
290
|
+
)
|
|
291
|
+
continue
|
|
292
|
+
|
|
293
|
+
passed_checks += 1
|
|
294
|
+
else:
|
|
295
|
+
# Optional parameter not provided - that's fine
|
|
296
|
+
passed_checks += 1
|
|
297
|
+
|
|
298
|
+
# Check for extra parameters
|
|
299
|
+
if not inputs.ignore_extra_params:
|
|
300
|
+
expected_params = {p.name for p in func_def.parameters}
|
|
301
|
+
extra_params = set(actual.arguments.keys()) - expected_params
|
|
302
|
+
if extra_params:
|
|
303
|
+
total_checks += len(extra_params)
|
|
304
|
+
errors.append(f"Unexpected parameters: {', '.join(extra_params)}")
|
|
305
|
+
|
|
306
|
+
if total_checks == 0:
|
|
307
|
+
return {"output": 1.0, "reason": "No parameters to validate."}
|
|
308
|
+
|
|
309
|
+
score = passed_checks / total_checks if total_checks > 0 else 1.0
|
|
310
|
+
|
|
311
|
+
if errors:
|
|
312
|
+
return {
|
|
313
|
+
"output": round(score, 4),
|
|
314
|
+
"reason": "; ".join(errors)
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
return {
|
|
318
|
+
"output": 1.0,
|
|
319
|
+
"reason": f"All {total_checks} parameter checks passed."
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
def _check_type(self, value: Any, expected_type: str, strict: bool) -> bool:
|
|
323
|
+
"""Check if a value matches the expected type."""
|
|
324
|
+
type_map = {
|
|
325
|
+
"string": (str,),
|
|
326
|
+
"integer": (int,) if strict else (int, float),
|
|
327
|
+
"number": (int, float),
|
|
328
|
+
"boolean": (bool,),
|
|
329
|
+
"array": (list,),
|
|
330
|
+
"object": (dict,),
|
|
331
|
+
"null": (type(None),),
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
expected_types = type_map.get(expected_type.lower(), (str,))
|
|
335
|
+
|
|
336
|
+
# Special case: strict integer check
|
|
337
|
+
if expected_type.lower() == "integer" and strict:
|
|
338
|
+
return isinstance(value, int) and not isinstance(value, bool)
|
|
339
|
+
|
|
340
|
+
# Boolean should not match int in Python
|
|
341
|
+
if isinstance(value, bool) and expected_type.lower() != "boolean":
|
|
342
|
+
return False
|
|
343
|
+
|
|
344
|
+
return isinstance(value, expected_types)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
class FunctionCallAccuracy(BaseMetric[FunctionCallInput]):
|
|
348
|
+
"""
|
|
349
|
+
Comprehensive function call accuracy evaluation.
|
|
350
|
+
|
|
351
|
+
Evaluates:
|
|
352
|
+
- Function name match (weighted 40%)
|
|
353
|
+
- Parameter presence (weighted 30%)
|
|
354
|
+
- Parameter value accuracy (weighted 30%)
|
|
355
|
+
|
|
356
|
+
Returns overall score from 0.0 to 1.0.
|
|
357
|
+
"""
|
|
358
|
+
|
|
359
|
+
@property
|
|
360
|
+
def metric_name(self) -> str:
|
|
361
|
+
return "function_call_accuracy"
|
|
362
|
+
|
|
363
|
+
def __init__(self, config: Optional[Dict[str, Any]] = None):
|
|
364
|
+
super().__init__(config)
|
|
365
|
+
self.name_weight = self.config.get("name_weight", 0.4)
|
|
366
|
+
self.presence_weight = self.config.get("presence_weight", 0.3)
|
|
367
|
+
self.value_weight = self.config.get("value_weight", 0.3)
|
|
368
|
+
|
|
369
|
+
def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
|
|
370
|
+
actual_calls = _parse_function_calls(inputs.response)
|
|
371
|
+
expected_calls = _parse_function_calls(inputs.expected_response)
|
|
372
|
+
|
|
373
|
+
if not actual_calls:
|
|
374
|
+
return {
|
|
375
|
+
"output": 0.0,
|
|
376
|
+
"reason": "Could not parse any function calls from response."
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
if not expected_calls:
|
|
380
|
+
return {
|
|
381
|
+
"output": 0.0,
|
|
382
|
+
"reason": "No expected function calls provided."
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
# Handle single vs multiple calls
|
|
386
|
+
if len(expected_calls) == 1 and len(actual_calls) == 1:
|
|
387
|
+
return self._evaluate_single(
|
|
388
|
+
actual_calls[0], expected_calls[0], inputs
|
|
389
|
+
)
|
|
390
|
+
|
|
391
|
+
# Multiple calls - evaluate as set or sequence
|
|
392
|
+
return self._evaluate_multiple(
|
|
393
|
+
actual_calls, expected_calls, inputs
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
def _evaluate_single(
|
|
397
|
+
self,
|
|
398
|
+
actual: FunctionCall,
|
|
399
|
+
expected: FunctionCall,
|
|
400
|
+
inputs: FunctionCallInput
|
|
401
|
+
) -> Dict[str, Any]:
|
|
402
|
+
"""Evaluate a single function call pair."""
|
|
403
|
+
details = []
|
|
404
|
+
|
|
405
|
+
# Name match (40%)
|
|
406
|
+
name_score = 1.0 if actual.name == expected.name else 0.0
|
|
407
|
+
details.append(f"name: {name_score:.0%}")
|
|
408
|
+
|
|
409
|
+
# Parameter presence (30%)
|
|
410
|
+
expected_params = set(expected.arguments.keys())
|
|
411
|
+
actual_params = set(actual.arguments.keys())
|
|
412
|
+
|
|
413
|
+
if expected_params:
|
|
414
|
+
presence_score = len(expected_params & actual_params) / len(expected_params)
|
|
415
|
+
else:
|
|
416
|
+
presence_score = 1.0 if not actual_params or inputs.ignore_extra_params else 0.0
|
|
417
|
+
|
|
418
|
+
details.append(f"params: {presence_score:.0%}")
|
|
419
|
+
|
|
420
|
+
# Parameter values (30%)
|
|
421
|
+
value_matches = 0
|
|
422
|
+
value_total = len(expected.arguments)
|
|
423
|
+
|
|
424
|
+
for param_name, expected_value in expected.arguments.items():
|
|
425
|
+
if param_name in actual.arguments:
|
|
426
|
+
actual_value = actual.arguments[param_name]
|
|
427
|
+
if _values_equal(actual_value, expected_value, inputs.strict_type_check):
|
|
428
|
+
value_matches += 1
|
|
429
|
+
|
|
430
|
+
value_score = value_matches / value_total if value_total > 0 else 1.0
|
|
431
|
+
details.append(f"values: {value_score:.0%}")
|
|
432
|
+
|
|
433
|
+
# Calculate weighted score
|
|
434
|
+
total_score = (
|
|
435
|
+
name_score * self.name_weight +
|
|
436
|
+
presence_score * self.presence_weight +
|
|
437
|
+
value_score * self.value_weight
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
return {
|
|
441
|
+
"output": round(total_score, 4),
|
|
442
|
+
"reason": f"Function call evaluation: {', '.join(details)}. Overall: {total_score:.1%}"
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
def _evaluate_multiple(
|
|
446
|
+
self,
|
|
447
|
+
actual_calls: List[FunctionCall],
|
|
448
|
+
expected_calls: List[FunctionCall],
|
|
449
|
+
inputs: FunctionCallInput
|
|
450
|
+
) -> Dict[str, Any]:
|
|
451
|
+
"""Evaluate multiple function calls (parallel calling)."""
|
|
452
|
+
if inputs.order_matters:
|
|
453
|
+
# Sequence comparison
|
|
454
|
+
if len(actual_calls) != len(expected_calls):
|
|
455
|
+
return {
|
|
456
|
+
"output": 0.0,
|
|
457
|
+
"reason": f"Call count mismatch: got {len(actual_calls)}, expected {len(expected_calls)}"
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
scores = []
|
|
461
|
+
for actual, expected in zip(actual_calls, expected_calls):
|
|
462
|
+
result = self._evaluate_single(actual, expected, inputs)
|
|
463
|
+
scores.append(result["output"])
|
|
464
|
+
|
|
465
|
+
avg_score = sum(scores) / len(scores)
|
|
466
|
+
return {
|
|
467
|
+
"output": round(avg_score, 4),
|
|
468
|
+
"reason": f"Sequence evaluation: {len(scores)} calls, avg score {avg_score:.1%}"
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
# Set comparison - find best match for each expected call
|
|
472
|
+
matched_scores = []
|
|
473
|
+
unmatched_expected = []
|
|
474
|
+
|
|
475
|
+
for expected in expected_calls:
|
|
476
|
+
best_score = 0.0
|
|
477
|
+
for actual in actual_calls:
|
|
478
|
+
result = self._evaluate_single(actual, expected, inputs)
|
|
479
|
+
best_score = max(best_score, result["output"])
|
|
480
|
+
|
|
481
|
+
if best_score > 0:
|
|
482
|
+
matched_scores.append(best_score)
|
|
483
|
+
else:
|
|
484
|
+
unmatched_expected.append(expected.name)
|
|
485
|
+
|
|
486
|
+
if not matched_scores:
|
|
487
|
+
return {
|
|
488
|
+
"output": 0.0,
|
|
489
|
+
"reason": f"No expected calls matched. Expected: {[c.name for c in expected_calls]}"
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
# Penalize for missing calls
|
|
493
|
+
coverage = len(matched_scores) / len(expected_calls)
|
|
494
|
+
avg_match_score = sum(matched_scores) / len(matched_scores)
|
|
495
|
+
final_score = coverage * avg_match_score
|
|
496
|
+
|
|
497
|
+
reason = f"Matched {len(matched_scores)}/{len(expected_calls)} calls, avg accuracy {avg_match_score:.1%}"
|
|
498
|
+
if unmatched_expected:
|
|
499
|
+
reason += f". Missing: {unmatched_expected}"
|
|
500
|
+
|
|
501
|
+
return {
|
|
502
|
+
"output": round(final_score, 4),
|
|
503
|
+
"reason": reason
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
class FunctionCallExactMatch(BaseMetric[FunctionCallInput]):
|
|
508
|
+
"""
|
|
509
|
+
AST-based exact match evaluation.
|
|
510
|
+
|
|
511
|
+
Parses function calls as AST and compares structure.
|
|
512
|
+
Useful for evaluating code-style function calls.
|
|
513
|
+
|
|
514
|
+
Returns 1.0 for exact match, 0.0 otherwise.
|
|
515
|
+
"""
|
|
516
|
+
|
|
517
|
+
@property
|
|
518
|
+
def metric_name(self) -> str:
|
|
519
|
+
return "function_call_exact_match"
|
|
520
|
+
|
|
521
|
+
def compute_one(self, inputs: FunctionCallInput) -> Dict[str, Any]:
|
|
522
|
+
actual = _parse_function_call(inputs.response)
|
|
523
|
+
expected = _parse_function_call(inputs.expected_response)
|
|
524
|
+
|
|
525
|
+
if actual is None:
|
|
526
|
+
return {
|
|
527
|
+
"output": 0.0,
|
|
528
|
+
"reason": "Could not parse actual function call."
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
if expected is None:
|
|
532
|
+
return {
|
|
533
|
+
"output": 0.0,
|
|
534
|
+
"reason": "Could not parse expected function call."
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
# Compare name
|
|
538
|
+
if actual.name != expected.name:
|
|
539
|
+
return {
|
|
540
|
+
"output": 0.0,
|
|
541
|
+
"reason": f"Function name mismatch: '{actual.name}' vs '{expected.name}'"
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
# Compare arguments
|
|
545
|
+
if set(actual.arguments.keys()) != set(expected.arguments.keys()):
|
|
546
|
+
missing = set(expected.arguments.keys()) - set(actual.arguments.keys())
|
|
547
|
+
extra = set(actual.arguments.keys()) - set(expected.arguments.keys())
|
|
548
|
+
parts = []
|
|
549
|
+
if missing:
|
|
550
|
+
parts.append(f"missing: {missing}")
|
|
551
|
+
if extra and not inputs.ignore_extra_params:
|
|
552
|
+
parts.append(f"extra: {extra}")
|
|
553
|
+
if parts:
|
|
554
|
+
return {
|
|
555
|
+
"output": 0.0,
|
|
556
|
+
"reason": f"Parameter mismatch: {', '.join(parts)}"
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
# Compare values
|
|
560
|
+
for key, expected_value in expected.arguments.items():
|
|
561
|
+
if key not in actual.arguments:
|
|
562
|
+
continue
|
|
563
|
+
actual_value = actual.arguments[key]
|
|
564
|
+
if not _values_equal(actual_value, expected_value, inputs.strict_type_check):
|
|
565
|
+
return {
|
|
566
|
+
"output": 0.0,
|
|
567
|
+
"reason": f"Value mismatch for '{key}': got {actual_value!r}, expected {expected_value!r}"
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
return {
|
|
571
|
+
"output": 1.0,
|
|
572
|
+
"reason": f"Function call matches: {actual.name}({', '.join(f'{k}={v!r}' for k, v in actual.arguments.items())})"
|
|
573
|
+
}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Types for Function Calling Evaluation.
|
|
3
|
+
|
|
4
|
+
These types support the evaluation of LLM function/tool calling
|
|
5
|
+
capabilities with AST-based comparison.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Any, Dict, List, Optional, Union
|
|
9
|
+
from pydantic import BaseModel, Field
|
|
10
|
+
|
|
11
|
+
from ...types import BaseMetricInput
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ParameterSpec(BaseModel):
|
|
15
|
+
"""Specification for a function parameter."""
|
|
16
|
+
|
|
17
|
+
name: str = Field(..., description="Parameter name")
|
|
18
|
+
type: str = Field(..., description="Expected type (string, integer, number, boolean, array, object)")
|
|
19
|
+
required: bool = Field(default=True, description="Whether the parameter is required")
|
|
20
|
+
enum: Optional[List[Any]] = Field(default=None, description="Allowed values if constrained")
|
|
21
|
+
default: Optional[Any] = Field(default=None, description="Default value if not required")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class FunctionDefinition(BaseModel):
|
|
25
|
+
"""Definition of an expected function signature."""
|
|
26
|
+
|
|
27
|
+
name: str = Field(..., description="Function name")
|
|
28
|
+
parameters: List[ParameterSpec] = Field(
|
|
29
|
+
default_factory=list,
|
|
30
|
+
description="List of parameter specifications"
|
|
31
|
+
)
|
|
32
|
+
description: Optional[str] = Field(default=None, description="Function description")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class FunctionCall(BaseModel):
|
|
36
|
+
"""Represents a function call from the LLM."""
|
|
37
|
+
|
|
38
|
+
name: str = Field(..., description="Name of the function called")
|
|
39
|
+
arguments: Dict[str, Any] = Field(
|
|
40
|
+
default_factory=dict,
|
|
41
|
+
description="Arguments passed to the function"
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class FunctionCallInput(BaseMetricInput):
|
|
46
|
+
"""
|
|
47
|
+
Input for function calling evaluation metrics.
|
|
48
|
+
|
|
49
|
+
Supports evaluating:
|
|
50
|
+
- Single function call against expected
|
|
51
|
+
- Multiple function calls (parallel calling)
|
|
52
|
+
- Function call with schema validation
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
# The actual function call(s) from the LLM
|
|
56
|
+
response: Union[FunctionCall, List[FunctionCall], Dict[str, Any], str] = Field(
|
|
57
|
+
...,
|
|
58
|
+
description="The function call(s) from the LLM. Can be FunctionCall object, dict, JSON string, or list of calls."
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# The expected function call(s)
|
|
62
|
+
expected_response: Optional[Union[FunctionCall, List[FunctionCall], Dict[str, Any], str]] = Field(
|
|
63
|
+
default=None,
|
|
64
|
+
description="The expected function call(s). Required for accuracy metrics."
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# Function definitions for schema validation
|
|
68
|
+
function_definitions: Optional[List[FunctionDefinition]] = Field(
|
|
69
|
+
default=None,
|
|
70
|
+
description="Available function definitions for validation."
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# Evaluation options
|
|
74
|
+
strict_type_check: bool = Field(
|
|
75
|
+
default=False,
|
|
76
|
+
description="If True, require exact type matches. If False, allow compatible types (e.g., int/float)."
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
ignore_extra_params: bool = Field(
|
|
80
|
+
default=False,
|
|
81
|
+
description="If True, ignore extra parameters not in expected. If False, penalize extra params."
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
order_matters: bool = Field(
|
|
85
|
+
default=False,
|
|
86
|
+
description="If True, for parallel calls, order must match. If False, set comparison."
|
|
87
|
+
)
|