agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/evals/evaluator.py
ADDED
|
@@ -0,0 +1,721 @@
|
|
|
1
|
+
import inspect
|
|
2
|
+
import json
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed, TimeoutError
|
|
6
|
+
from typing import Any, Dict, List, Optional, Union
|
|
7
|
+
|
|
8
|
+
from requests import Response
|
|
9
|
+
|
|
10
|
+
from fi.api.auth import APIKeyAuth, ResponseHandler
|
|
11
|
+
from fi.api.types import HttpMethod, RequestConfig
|
|
12
|
+
from fi.evals.execution import Execution, _normalize_status
|
|
13
|
+
from fi.evals.templates import EvalTemplate
|
|
14
|
+
from fi.evals.types import BatchRunResult, EvalResult
|
|
15
|
+
from fi.utils.errors import InvalidAuthError
|
|
16
|
+
from fi.utils.routes import Routes
|
|
17
|
+
|
|
18
|
+
def _coerce_to_api_input(value: Any) -> Any:
|
|
19
|
+
"""Serialize rich native Python objects into API-supported input values."""
|
|
20
|
+
if isinstance(value, dict):
|
|
21
|
+
return json.dumps(value)
|
|
22
|
+
if isinstance(value, list):
|
|
23
|
+
if all(isinstance(v, str) for v in value):
|
|
24
|
+
return value
|
|
25
|
+
if all(
|
|
26
|
+
isinstance(v, list) and all(isinstance(x, str) for x in v)
|
|
27
|
+
for v in value
|
|
28
|
+
):
|
|
29
|
+
return value
|
|
30
|
+
return json.dumps(value)
|
|
31
|
+
return value
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class EvalResponseHandler(ResponseHandler[BatchRunResult, None]):
|
|
35
|
+
"""Handles responses for evaluation requests"""
|
|
36
|
+
|
|
37
|
+
@classmethod
|
|
38
|
+
def _parse_success(cls, response: Response) -> BatchRunResult:
|
|
39
|
+
return cls.convert_to_batch_results(response.json())
|
|
40
|
+
|
|
41
|
+
@classmethod
|
|
42
|
+
def _handle_error(cls, response: Response) -> None:
|
|
43
|
+
if response.status_code == 400:
|
|
44
|
+
raise Exception(
|
|
45
|
+
f"Evaluation failed with a 400 Bad Request. Please check your input data and evaluation configuration. Response: {response.text}"
|
|
46
|
+
)
|
|
47
|
+
elif response.status_code == 403:
|
|
48
|
+
raise InvalidAuthError()
|
|
49
|
+
else:
|
|
50
|
+
raise Exception(
|
|
51
|
+
f"Error in evaluation: {response.status_code}, response: {response.text}"
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
@classmethod
|
|
55
|
+
def convert_to_batch_results(cls, response: Dict[str, Any]) -> BatchRunResult:
|
|
56
|
+
"""
|
|
57
|
+
Convert API response to BatchRunResult
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
response: Raw API response dictionary
|
|
61
|
+
|
|
62
|
+
Returns:
|
|
63
|
+
BatchRunResult containing evaluation results
|
|
64
|
+
"""
|
|
65
|
+
eval_results = []
|
|
66
|
+
|
|
67
|
+
# The revamped backend (post 2026-04-12) returns pure snake_case:
|
|
68
|
+
# {"result": [{"evaluations": [
|
|
69
|
+
# {"name", "reason", "runtime", "output", "output_type",
|
|
70
|
+
# "eval_id", "model"?, "error_localizer_enabled"?,
|
|
71
|
+
# "error_localizer"?}
|
|
72
|
+
# ]}]}
|
|
73
|
+
# Async / error-localization paths may return the eval wrapped in
|
|
74
|
+
# {"eval_status": "...", "result": <eval>} — handle that too.
|
|
75
|
+
for result in response.get("result", []) or []:
|
|
76
|
+
if isinstance(result, dict) and "evaluations" in result:
|
|
77
|
+
entries = result.get("evaluations", []) or []
|
|
78
|
+
else:
|
|
79
|
+
entries = [result] if isinstance(result, dict) else []
|
|
80
|
+
|
|
81
|
+
for evaluation in entries:
|
|
82
|
+
if not isinstance(evaluation, dict):
|
|
83
|
+
continue
|
|
84
|
+
eval_results.append(
|
|
85
|
+
EvalResult(
|
|
86
|
+
name=evaluation.get("name", ""),
|
|
87
|
+
output=evaluation.get("output", evaluation.get("value")),
|
|
88
|
+
reason=evaluation.get("reason", ""),
|
|
89
|
+
runtime=evaluation.get("runtime", 0),
|
|
90
|
+
output_type=evaluation.get("output_type", ""),
|
|
91
|
+
eval_id=evaluation.get("eval_id", ""),
|
|
92
|
+
model=evaluation.get("model"),
|
|
93
|
+
error_localizer_enabled=evaluation.get(
|
|
94
|
+
"error_localizer_enabled"
|
|
95
|
+
),
|
|
96
|
+
error_localizer=evaluation.get("error_localizer"),
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
return BatchRunResult(eval_results=eval_results)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class EvalInfoResponseHandler(ResponseHandler[dict, None]):
|
|
104
|
+
"""Handles responses for evaluation info requests"""
|
|
105
|
+
|
|
106
|
+
@classmethod
|
|
107
|
+
def _parse_success(cls, response: Response) -> dict:
|
|
108
|
+
data = response.json()
|
|
109
|
+
if "result" in data:
|
|
110
|
+
return data["result"]
|
|
111
|
+
else:
|
|
112
|
+
raise Exception(f"Failed to get evaluation info: {data}")
|
|
113
|
+
|
|
114
|
+
@classmethod
|
|
115
|
+
def _handle_error(cls, response: Response) -> None:
|
|
116
|
+
if response.status_code == 400:
|
|
117
|
+
response.raise_for_status()
|
|
118
|
+
if response.status_code == 403:
|
|
119
|
+
raise InvalidAuthError()
|
|
120
|
+
raise Exception(f"Failed to get evaluation info: {response.status_code}")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class Evaluator(APIKeyAuth):
|
|
124
|
+
"""Client for evaluating LLM test cases"""
|
|
125
|
+
|
|
126
|
+
def __init__(
|
|
127
|
+
self,
|
|
128
|
+
fi_api_key: Optional[str] = None,
|
|
129
|
+
fi_secret_key: Optional[str] = None,
|
|
130
|
+
fi_base_url: Optional[str] = None,
|
|
131
|
+
**kwargs,
|
|
132
|
+
) -> None:
|
|
133
|
+
"""
|
|
134
|
+
Initialize the Eval Client
|
|
135
|
+
|
|
136
|
+
Args:
|
|
137
|
+
fi_api_key: API key
|
|
138
|
+
fi_secret_key: Secret key
|
|
139
|
+
fi_base_url: Base URL
|
|
140
|
+
|
|
141
|
+
Keyword Args:
|
|
142
|
+
timeout: Optional timeout value in seconds (default: 200)
|
|
143
|
+
max_queue_bound: Optional maximum queue size (default: 5000)
|
|
144
|
+
max_workers: Optional maximum number of workers (default: 8)
|
|
145
|
+
langfuse_secret_key: Optional Langfuse secret key
|
|
146
|
+
langfuse_public_key: Optional Langfuse public key
|
|
147
|
+
langfuse_host: Optional Langfuse host
|
|
148
|
+
"""
|
|
149
|
+
super().__init__(fi_api_key, fi_secret_key, fi_base_url, **kwargs)
|
|
150
|
+
self._max_workers = kwargs.get("max_workers", 8) # Default to 8 if not provided
|
|
151
|
+
|
|
152
|
+
# Handle Langfuse credentials
|
|
153
|
+
self.langfuse_secret_key = kwargs.get("langfuse_secret_key") or os.getenv("LANGFUSE_SECRET_KEY")
|
|
154
|
+
self.langfuse_public_key = kwargs.get("langfuse_public_key") or os.getenv("LANGFUSE_PUBLIC_KEY")
|
|
155
|
+
self.langfuse_host = kwargs.get("langfuse_host") or os.getenv("LANGFUSE_HOST")
|
|
156
|
+
|
|
157
|
+
# Instance-level cache for _get_eval_info results.
|
|
158
|
+
# Previously decorated with @lru_cache, but @lru_cache on a bound
|
|
159
|
+
# method (one taking `self`) holds a strong reference to the
|
|
160
|
+
# instance via the cache key, preventing the Evaluator from being
|
|
161
|
+
# garbage-collected. In long-running processes (Celery / Temporal
|
|
162
|
+
# workers, FastAPI lifespans) that creates a slow memory leak.
|
|
163
|
+
self._eval_info_cache: Dict[str, Dict[str, Any]] = {}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def evaluate(
|
|
167
|
+
self,
|
|
168
|
+
eval_templates: Union[str, type[EvalTemplate]],
|
|
169
|
+
inputs: Dict[str, Any],
|
|
170
|
+
timeout: Optional[int] = None,
|
|
171
|
+
model_name: Optional[str] = None,
|
|
172
|
+
custom_eval_name: Optional[str] = None,
|
|
173
|
+
trace_eval: Optional[bool] = False,
|
|
174
|
+
platform: Optional[str] = None,
|
|
175
|
+
is_async: Optional[bool] = False,
|
|
176
|
+
error_localizer: Optional[bool] = False,
|
|
177
|
+
eval_config: Optional[Dict[str, Any]] = None,
|
|
178
|
+
**kwargs,
|
|
179
|
+
) -> BatchRunResult:
|
|
180
|
+
"""
|
|
181
|
+
Run a single or batch of evaluations independently
|
|
182
|
+
|
|
183
|
+
Args:
|
|
184
|
+
eval_templates: Evaluation name string (e.g., "Factual Accuracy")
|
|
185
|
+
inputs: Single test case or list of test cases
|
|
186
|
+
timeout: Optional timeout value for the evaluation
|
|
187
|
+
model_name: Optional model name to use for the evaluation for Future AGI Agents
|
|
188
|
+
span_id: Optional span_id to attach to the evaluation. If not provided, it will be retrieved from the OpenTelemetry context if available.
|
|
189
|
+
custom_eval_name: Optional custom evaluation name to use for the evaluation. If not provided, eval will not be added to the span.
|
|
190
|
+
Returns:
|
|
191
|
+
BatchRunResult containing evaluation results
|
|
192
|
+
|
|
193
|
+
Raises:
|
|
194
|
+
ValidationError: If the inputs do not match the evaluation templates
|
|
195
|
+
Exception: If the API request fails
|
|
196
|
+
"""
|
|
197
|
+
if platform:
|
|
198
|
+
if isinstance(eval_templates, str) and isinstance(inputs, dict) and custom_eval_name:
|
|
199
|
+
return self._configure_evaluations(
|
|
200
|
+
eval_templates=eval_templates,
|
|
201
|
+
inputs=inputs,
|
|
202
|
+
platform=platform,
|
|
203
|
+
custom_eval_name=custom_eval_name,
|
|
204
|
+
model_name=model_name,
|
|
205
|
+
**kwargs
|
|
206
|
+
)
|
|
207
|
+
else:
|
|
208
|
+
raise ValueError("Invalid arguments for platform configuration")
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _extract_name(t) -> str | None:
|
|
212
|
+
if isinstance(t, str):
|
|
213
|
+
return t
|
|
214
|
+
if isinstance(t, EvalTemplate):
|
|
215
|
+
return t.eval_name
|
|
216
|
+
if inspect.isclass(t) and issubclass(t, EvalTemplate):
|
|
217
|
+
return t.eval_name
|
|
218
|
+
return None
|
|
219
|
+
|
|
220
|
+
eval_name = _extract_name(
|
|
221
|
+
eval_templates[0] if isinstance(eval_templates, list) else eval_templates
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
span_id = None
|
|
225
|
+
project_name = None
|
|
226
|
+
if trace_eval:
|
|
227
|
+
if not custom_eval_name:
|
|
228
|
+
trace_eval = False
|
|
229
|
+
logging.warning("Failed to trace the evaluation. Please set the custom_eval_name.")
|
|
230
|
+
else:
|
|
231
|
+
try:
|
|
232
|
+
from opentelemetry import trace
|
|
233
|
+
|
|
234
|
+
current_span = trace.get_current_span()
|
|
235
|
+
if current_span and current_span.is_recording():
|
|
236
|
+
span_context = current_span.get_span_context()
|
|
237
|
+
if span_context.is_valid:
|
|
238
|
+
span_id = format(span_context.span_id, "016x")
|
|
239
|
+
tracer_provider = trace.get_tracer_provider()
|
|
240
|
+
if hasattr(tracer_provider, "resource"):
|
|
241
|
+
attributes = tracer_provider.resource.attributes
|
|
242
|
+
project_name = attributes.get("project_name")
|
|
243
|
+
|
|
244
|
+
if not project_name:
|
|
245
|
+
trace_eval = False
|
|
246
|
+
logging.warning(
|
|
247
|
+
"Could not determine project_name from OpenTelemetry context. "
|
|
248
|
+
"Skipping trace_eval for this evaluation."
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
except ImportError:
|
|
252
|
+
logging.exception(
|
|
253
|
+
"Future AGI SDK not found. "
|
|
254
|
+
"Please install 'fi-instrumentation-otel' to automatically enrich the evaluation with project context."
|
|
255
|
+
)
|
|
256
|
+
return
|
|
257
|
+
|
|
258
|
+
if eval_name is None:
|
|
259
|
+
raise TypeError(
|
|
260
|
+
"Unsupported eval_templates argument. "
|
|
261
|
+
"Expect eval template class/obj or name str."
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
# Dynamic registry: filter user-supplied inputs to only the keys the
|
|
265
|
+
# backend currently accepts for this eval. The api rejects supersets
|
|
266
|
+
# (e.g. {output,input,context} for a template that only wants
|
|
267
|
+
# {output}), so this can't be a pass-through. If the registry fetch
|
|
268
|
+
# fails or the name is unknown, leave inputs untouched.
|
|
269
|
+
if kwargs.get("skip_input_mapping") is not True and isinstance(inputs, dict):
|
|
270
|
+
try:
|
|
271
|
+
from fi.evals.core.cloud_registry import map_inputs_to_backend
|
|
272
|
+
inputs = map_inputs_to_backend(
|
|
273
|
+
eval_name,
|
|
274
|
+
inputs,
|
|
275
|
+
base_url=self._base_url,
|
|
276
|
+
api_key=self._fi_api_key,
|
|
277
|
+
secret_key=self._fi_secret_key,
|
|
278
|
+
)
|
|
279
|
+
except Exception as exc:
|
|
280
|
+
logging.debug("Dynamic input mapping skipped: %s", exc)
|
|
281
|
+
|
|
282
|
+
# The api validator accepts only strings, list[str], or list[list[str]].
|
|
283
|
+
# JSON-serialize dict / list-of-dicts values (e.g. conversation messages)
|
|
284
|
+
# so users can pass native Python objects without manually stringifying.
|
|
285
|
+
if isinstance(inputs, dict):
|
|
286
|
+
inputs = {k: _coerce_to_api_input(v) for k, v in inputs.items()}
|
|
287
|
+
|
|
288
|
+
final_api_payload = {
|
|
289
|
+
"eval_name": eval_name,
|
|
290
|
+
"inputs": inputs,
|
|
291
|
+
"model": model_name,
|
|
292
|
+
"span_id": span_id,
|
|
293
|
+
"custom_eval_name": custom_eval_name,
|
|
294
|
+
"trace_eval": trace_eval,
|
|
295
|
+
"is_async": is_async,
|
|
296
|
+
"error_localizer": error_localizer,
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
if eval_config:
|
|
300
|
+
final_api_payload["config"] = {"params": eval_config}
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
all_results = []
|
|
304
|
+
failed_inputs = []
|
|
305
|
+
with ThreadPoolExecutor(max_workers=self._max_workers) as executor:
|
|
306
|
+
# Submit the batch only once
|
|
307
|
+
future = executor.submit(
|
|
308
|
+
self.request,
|
|
309
|
+
config=RequestConfig(
|
|
310
|
+
method=HttpMethod.POST,
|
|
311
|
+
url=f"{self._base_url}/{Routes.evaluatev2.value}",
|
|
312
|
+
json=final_api_payload,
|
|
313
|
+
timeout=timeout or self._default_timeout,
|
|
314
|
+
),
|
|
315
|
+
response_handler=EvalResponseHandler,
|
|
316
|
+
)
|
|
317
|
+
future_to_input = {future: inputs} # map single future to all inputs
|
|
318
|
+
|
|
319
|
+
for future in as_completed(future_to_input):
|
|
320
|
+
try:
|
|
321
|
+
response: BatchRunResult = future.result(timeout=timeout or self._default_timeout)
|
|
322
|
+
all_results.extend(response.eval_results)
|
|
323
|
+
except TimeoutError:
|
|
324
|
+
input_case = future_to_input[future]
|
|
325
|
+
logging.error(f"Evaluation timed out for input: {input_case}")
|
|
326
|
+
failed_inputs.append(input_case)
|
|
327
|
+
all_results.append(
|
|
328
|
+
EvalResult(
|
|
329
|
+
name=eval_name,
|
|
330
|
+
output=None,
|
|
331
|
+
reason=f"Evaluation timed out after {timeout or self._default_timeout}s",
|
|
332
|
+
runtime=0,
|
|
333
|
+
)
|
|
334
|
+
)
|
|
335
|
+
except Exception as exc:
|
|
336
|
+
input_case = future_to_input[future]
|
|
337
|
+
logging.error(f"Evaluation failed for input {input_case}: {str(exc)}")
|
|
338
|
+
failed_inputs.append(input_case)
|
|
339
|
+
all_results.append(
|
|
340
|
+
EvalResult(
|
|
341
|
+
name=eval_name,
|
|
342
|
+
output=None,
|
|
343
|
+
reason=str(exc),
|
|
344
|
+
runtime=0,
|
|
345
|
+
)
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
if failed_inputs:
|
|
349
|
+
logging.warning(f"Failed to evaluate {len(failed_inputs)} inputs out of {len(inputs)} total inputs")
|
|
350
|
+
|
|
351
|
+
# Automatically enrich current span with evaluation results
|
|
352
|
+
result = BatchRunResult(eval_results=all_results)
|
|
353
|
+
try:
|
|
354
|
+
from fi.evals.otel.enrichment import enrich_span_with_batch_result, is_auto_enrichment_enabled
|
|
355
|
+
if is_auto_enrichment_enabled():
|
|
356
|
+
enriched_count = enrich_span_with_batch_result(result)
|
|
357
|
+
if enriched_count > 0:
|
|
358
|
+
logging.debug(f"Enriched active span with {enriched_count} evaluation results")
|
|
359
|
+
except ImportError:
|
|
360
|
+
pass # OTEL enrichment not available
|
|
361
|
+
except Exception as e:
|
|
362
|
+
logging.debug(f"Failed to enrich span with evaluation results: {e}")
|
|
363
|
+
|
|
364
|
+
return result
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def get_eval_result(self, eval_id: str):
|
|
368
|
+
"""
|
|
369
|
+
Get the raw evaluation status payload by ID (unparsed).
|
|
370
|
+
|
|
371
|
+
For a higher-level handle that understands the status envelope
|
|
372
|
+
and can be awaited, see :py:meth:`get_execution`.
|
|
373
|
+
"""
|
|
374
|
+
url = f"{self._base_url}/{Routes.get_eval_result.value}"
|
|
375
|
+
response = self.request(
|
|
376
|
+
config=RequestConfig(
|
|
377
|
+
method=HttpMethod.GET,
|
|
378
|
+
url=url,
|
|
379
|
+
params={"eval_id": eval_id},
|
|
380
|
+
timeout=self._default_timeout,
|
|
381
|
+
),
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
return response.json()
|
|
385
|
+
|
|
386
|
+
# ------------------------------------------------------------------
|
|
387
|
+
# Async submission / execution handles
|
|
388
|
+
# ------------------------------------------------------------------
|
|
389
|
+
|
|
390
|
+
def submit(
|
|
391
|
+
self,
|
|
392
|
+
eval_templates: Union[str, type[EvalTemplate]],
|
|
393
|
+
inputs: Dict[str, Any],
|
|
394
|
+
*,
|
|
395
|
+
model_name: Optional[str] = None,
|
|
396
|
+
custom_eval_name: Optional[str] = None,
|
|
397
|
+
trace_eval: bool = False,
|
|
398
|
+
error_localizer: bool = False,
|
|
399
|
+
timeout: Optional[int] = None,
|
|
400
|
+
**kwargs: Any,
|
|
401
|
+
) -> Execution:
|
|
402
|
+
"""
|
|
403
|
+
Submit an eval for async execution and return an :class:`Execution`
|
|
404
|
+
handle immediately. Use ``handle.wait()`` or
|
|
405
|
+
:py:meth:`get_execution` to poll for completion.
|
|
406
|
+
|
|
407
|
+
This is the non-blocking equivalent of :py:meth:`evaluate` —
|
|
408
|
+
internally it always sets ``is_async=True`` so the backend records
|
|
409
|
+
the evaluation and starts a worker without holding the HTTP
|
|
410
|
+
connection open.
|
|
411
|
+
"""
|
|
412
|
+
|
|
413
|
+
def _extract_name(t: Any) -> Optional[str]:
|
|
414
|
+
if isinstance(t, str):
|
|
415
|
+
return t
|
|
416
|
+
if isinstance(t, EvalTemplate):
|
|
417
|
+
return t.eval_name
|
|
418
|
+
if inspect.isclass(t) and issubclass(t, EvalTemplate):
|
|
419
|
+
return t.eval_name
|
|
420
|
+
return None
|
|
421
|
+
|
|
422
|
+
eval_name = _extract_name(
|
|
423
|
+
eval_templates[0] if isinstance(eval_templates, list) else eval_templates
|
|
424
|
+
)
|
|
425
|
+
if eval_name is None:
|
|
426
|
+
raise TypeError(
|
|
427
|
+
"Unsupported eval_templates argument. "
|
|
428
|
+
"Expect eval template class/obj or name str."
|
|
429
|
+
)
|
|
430
|
+
|
|
431
|
+
payload = {
|
|
432
|
+
"eval_name": eval_name,
|
|
433
|
+
"inputs": inputs,
|
|
434
|
+
"model": model_name,
|
|
435
|
+
"span_id": kwargs.get("span_id"),
|
|
436
|
+
"custom_eval_name": custom_eval_name,
|
|
437
|
+
"trace_eval": trace_eval,
|
|
438
|
+
"is_async": True,
|
|
439
|
+
"error_localizer": error_localizer,
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
response = self.request(
|
|
443
|
+
config=RequestConfig(
|
|
444
|
+
method=HttpMethod.POST,
|
|
445
|
+
url=f"{self._base_url}/{Routes.evaluatev2.value}",
|
|
446
|
+
json=payload,
|
|
447
|
+
timeout=timeout or self._default_timeout,
|
|
448
|
+
),
|
|
449
|
+
)
|
|
450
|
+
body = response.json() if hasattr(response, "json") else {}
|
|
451
|
+
|
|
452
|
+
# Backend responds with:
|
|
453
|
+
# {"status": true, "result": [{"evaluations": [{name, output_type, eval_id}]}]}
|
|
454
|
+
# on success, or
|
|
455
|
+
# {"status": false, "result": {...error dict...}}
|
|
456
|
+
# on validation failure.
|
|
457
|
+
if not body.get("status", True):
|
|
458
|
+
raise RuntimeError(
|
|
459
|
+
f"Async submit rejected by backend: {body.get('result')}"
|
|
460
|
+
)
|
|
461
|
+
|
|
462
|
+
results = body.get("result") or []
|
|
463
|
+
if not isinstance(results, list) or not results:
|
|
464
|
+
raise RuntimeError(
|
|
465
|
+
f"Async submit did not return a result list (response: {body})"
|
|
466
|
+
)
|
|
467
|
+
evaluations = results[0].get("evaluations") or []
|
|
468
|
+
first_eval = evaluations[0] if evaluations else {}
|
|
469
|
+
execution_id = first_eval.get("eval_id")
|
|
470
|
+
if not execution_id:
|
|
471
|
+
raise RuntimeError(
|
|
472
|
+
f"Async submit did not return an eval_id (response: {body})"
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
handle = Execution(
|
|
476
|
+
id=str(execution_id),
|
|
477
|
+
kind="eval",
|
|
478
|
+
status="pending",
|
|
479
|
+
)
|
|
480
|
+
handle._refresher = lambda eid=execution_id: self._refresh_eval_execution(eid)
|
|
481
|
+
return handle
|
|
482
|
+
|
|
483
|
+
def get_execution(self, execution_id: str) -> Execution:
|
|
484
|
+
"""
|
|
485
|
+
Fetch the latest state of an async single-eval execution by ID.
|
|
486
|
+
|
|
487
|
+
Returns an :class:`Execution` handle with an attached refresher
|
|
488
|
+
closure — call ``handle.wait()`` to block until completion.
|
|
489
|
+
"""
|
|
490
|
+
handle = self._refresh_eval_execution(execution_id)
|
|
491
|
+
handle._refresher = (
|
|
492
|
+
lambda eid=execution_id: self._refresh_eval_execution(eid)
|
|
493
|
+
)
|
|
494
|
+
return handle
|
|
495
|
+
|
|
496
|
+
def _refresh_eval_execution(self, execution_id: str) -> Execution:
|
|
497
|
+
url = f"{self._base_url}/{Routes.get_eval_result.value}"
|
|
498
|
+
response = self.request(
|
|
499
|
+
config=RequestConfig(
|
|
500
|
+
method=HttpMethod.GET,
|
|
501
|
+
url=url,
|
|
502
|
+
params={"eval_id": execution_id},
|
|
503
|
+
timeout=self._default_timeout,
|
|
504
|
+
),
|
|
505
|
+
)
|
|
506
|
+
body = response.json() if hasattr(response, "json") else {}
|
|
507
|
+
# Envelope: {"status": true, "result": {"eval_status": ..., "result": <body|str>}}
|
|
508
|
+
payload = body.get("result") or {}
|
|
509
|
+
status = _normalize_status(payload.get("eval_status"))
|
|
510
|
+
raw_result = payload.get("result")
|
|
511
|
+
error_message = payload.get("error_message")
|
|
512
|
+
|
|
513
|
+
parsed_result: Any = None
|
|
514
|
+
error_localizer: Optional[Dict[str, Any]] = None
|
|
515
|
+
if isinstance(raw_result, dict):
|
|
516
|
+
# Completed / failed state — full eval record.
|
|
517
|
+
parsed_result = EvalResult(
|
|
518
|
+
name=raw_result.get("name", ""),
|
|
519
|
+
output=raw_result.get("output", raw_result.get("value")),
|
|
520
|
+
reason=raw_result.get("reason"),
|
|
521
|
+
runtime=raw_result.get("runtime", 0),
|
|
522
|
+
output_type=raw_result.get("output_type"),
|
|
523
|
+
eval_id=str(raw_result.get("eval_id", execution_id)),
|
|
524
|
+
model=raw_result.get("model"),
|
|
525
|
+
error_localizer_enabled=raw_result.get("error_localizer_enabled"),
|
|
526
|
+
error_localizer=raw_result.get("error_localizer"),
|
|
527
|
+
)
|
|
528
|
+
error_localizer = raw_result.get("error_localizer")
|
|
529
|
+
error_message = raw_result.get("error_message") or error_message
|
|
530
|
+
# else: raw_result is a human-readable string like "Evaluation is
|
|
531
|
+
# being processed." — leave parsed_result as None.
|
|
532
|
+
|
|
533
|
+
return Execution(
|
|
534
|
+
id=str(execution_id),
|
|
535
|
+
kind="eval",
|
|
536
|
+
status=status,
|
|
537
|
+
result=parsed_result,
|
|
538
|
+
error_message=error_message,
|
|
539
|
+
error_localizer=error_localizer,
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _configure_evaluations(
|
|
544
|
+
self,
|
|
545
|
+
eval_templates: str,
|
|
546
|
+
inputs: Dict[str, Any],
|
|
547
|
+
platform: str,
|
|
548
|
+
custom_eval_name: str,
|
|
549
|
+
model_name: Optional[str] = None,
|
|
550
|
+
**kwargs,
|
|
551
|
+
) -> Dict[str, Any]:
|
|
552
|
+
"""
|
|
553
|
+
Configure evaluations on a specified platform.
|
|
554
|
+
|
|
555
|
+
This will not return any evaluation results, but rather a
|
|
556
|
+
confirmation message from the backend.
|
|
557
|
+
|
|
558
|
+
Args:
|
|
559
|
+
eval_config: The evaluation configuration dictionary.
|
|
560
|
+
platform: The platform to which the evaluations should be sent.
|
|
561
|
+
timeout: Optional timeout for the API request.
|
|
562
|
+
**kwargs: Additional configuration parameters to be sent with the request.
|
|
563
|
+
|
|
564
|
+
Returns:
|
|
565
|
+
A dictionary containing the backend's response message.
|
|
566
|
+
"""
|
|
567
|
+
try:
|
|
568
|
+
from fi.evals.otel_utils import _get_current_otel_span
|
|
569
|
+
|
|
570
|
+
if platform == "langfuse":
|
|
571
|
+
kwargs["langfuse_secret_key"] = self.langfuse_secret_key
|
|
572
|
+
kwargs["langfuse_public_key"] = self.langfuse_public_key
|
|
573
|
+
kwargs["langfuse_host"] = self.langfuse_host
|
|
574
|
+
|
|
575
|
+
current_span = _get_current_otel_span()
|
|
576
|
+
if current_span:
|
|
577
|
+
span_context = current_span.get_span_context()
|
|
578
|
+
if span_context.is_valid:
|
|
579
|
+
span_id = format(span_context.span_id, "016x")
|
|
580
|
+
trace_id = format(span_context.trace_id, "032x")
|
|
581
|
+
kwargs["span_id"] = span_id
|
|
582
|
+
kwargs["trace_id"] = trace_id
|
|
583
|
+
|
|
584
|
+
# Check if span_id and trace_id are present in kwargs
|
|
585
|
+
if "span_id" not in kwargs or "trace_id" not in kwargs:
|
|
586
|
+
logging.warning(
|
|
587
|
+
"span_id and/or trace_id not found in kwargs ."
|
|
588
|
+
"Please run this function within a span context."
|
|
589
|
+
)
|
|
590
|
+
return
|
|
591
|
+
|
|
592
|
+
api_payload = {
|
|
593
|
+
"eval_config": {
|
|
594
|
+
"eval_templates": eval_templates,
|
|
595
|
+
"inputs": inputs,
|
|
596
|
+
"model_name": model_name
|
|
597
|
+
},
|
|
598
|
+
"custom_eval_name": custom_eval_name,
|
|
599
|
+
"platform": platform,
|
|
600
|
+
**kwargs,
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
response = self.request(
|
|
604
|
+
config=RequestConfig(
|
|
605
|
+
method=HttpMethod.POST,
|
|
606
|
+
url=f"{self._base_url}/{Routes.configure_evaluations.value}",
|
|
607
|
+
json=api_payload,
|
|
608
|
+
timeout=self._default_timeout,
|
|
609
|
+
),
|
|
610
|
+
)
|
|
611
|
+
|
|
612
|
+
if response.status_code != 200:
|
|
613
|
+
logging.warning(
|
|
614
|
+
f"Received non-200 status code from backend: {response.status_code}. "
|
|
615
|
+
f"Response: {response.text}"
|
|
616
|
+
)
|
|
617
|
+
|
|
618
|
+
return response.json()
|
|
619
|
+
|
|
620
|
+
except ImportError:
|
|
621
|
+
logging.exception(
|
|
622
|
+
"Future AGI SDK not found. "
|
|
623
|
+
"Please install 'fi-instrumentation-otel' to use these evaluations."
|
|
624
|
+
)
|
|
625
|
+
return
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def _get_eval_info(self, eval_name: str) -> Dict[str, Any]:
|
|
629
|
+
cached = self._eval_info_cache.get(eval_name)
|
|
630
|
+
if cached is not None:
|
|
631
|
+
return cached
|
|
632
|
+
|
|
633
|
+
url = (
|
|
634
|
+
self._base_url
|
|
635
|
+
+ "/"
|
|
636
|
+
+ Routes.get_eval_templates.value
|
|
637
|
+
)
|
|
638
|
+
response = self.request(
|
|
639
|
+
config=RequestConfig(method=HttpMethod.GET, url=url),
|
|
640
|
+
response_handler=EvalInfoResponseHandler,
|
|
641
|
+
)
|
|
642
|
+
eval_info = next((item for item in response if item["name"] == eval_name), None)
|
|
643
|
+
if eval_info is None:
|
|
644
|
+
raise KeyError(f"Evaluation template '{eval_name}' not found in registry")
|
|
645
|
+
if not eval_info:
|
|
646
|
+
raise Exception(f"Evaluation template with name '{eval_name}' not found")
|
|
647
|
+
self._eval_info_cache[eval_name] = eval_info
|
|
648
|
+
return eval_info
|
|
649
|
+
|
|
650
|
+
def list_evaluations(self):
|
|
651
|
+
"""
|
|
652
|
+
Fetch information about all available evaluation templates by getting eval_info
|
|
653
|
+
for each template class defined in templates.py.
|
|
654
|
+
|
|
655
|
+
Returns:
|
|
656
|
+
List[Dict[str, Any]]: List of evaluation template information dictionaries
|
|
657
|
+
"""
|
|
658
|
+
config = RequestConfig(method=HttpMethod.GET,
|
|
659
|
+
url=f"{self._base_url}/{Routes.get_eval_templates.value}")
|
|
660
|
+
|
|
661
|
+
response = self.request(config=config, response_handler=EvalInfoResponseHandler)
|
|
662
|
+
|
|
663
|
+
return response
|
|
664
|
+
|
|
665
|
+
|
|
666
|
+
def evaluate_pipeline(
|
|
667
|
+
self,
|
|
668
|
+
project_name: str,
|
|
669
|
+
version : str,
|
|
670
|
+
eval_data : List[Dict[str, Any]],
|
|
671
|
+
):
|
|
672
|
+
api_payload = {
|
|
673
|
+
"project_name": project_name,
|
|
674
|
+
"version": version,
|
|
675
|
+
"eval_data": eval_data
|
|
676
|
+
}
|
|
677
|
+
|
|
678
|
+
response = self.request(
|
|
679
|
+
config=RequestConfig(
|
|
680
|
+
method=HttpMethod.POST,
|
|
681
|
+
url=f"{self._base_url}/{Routes.evaluate_pipeline.value}",
|
|
682
|
+
json=api_payload,
|
|
683
|
+
timeout=self._default_timeout,
|
|
684
|
+
),
|
|
685
|
+
)
|
|
686
|
+
|
|
687
|
+
return response.json()
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def get_pipeline_results(
|
|
691
|
+
self,
|
|
692
|
+
project_name: str,
|
|
693
|
+
versions : List[str],
|
|
694
|
+
):
|
|
695
|
+
|
|
696
|
+
if not isinstance(versions, list) or not all(isinstance(v, str) for v in versions):
|
|
697
|
+
raise TypeError("versions must be a list of strings")
|
|
698
|
+
|
|
699
|
+
api_payload = {
|
|
700
|
+
"project_name": project_name,
|
|
701
|
+
"versions": ",".join(versions),
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
response = self.request(
|
|
705
|
+
config=RequestConfig(
|
|
706
|
+
method=HttpMethod.GET,
|
|
707
|
+
url=f"{self._base_url}/{Routes.evaluate_pipeline.value}",
|
|
708
|
+
params=api_payload,
|
|
709
|
+
timeout=self._default_timeout,
|
|
710
|
+
),
|
|
711
|
+
)
|
|
712
|
+
|
|
713
|
+
return response.json()
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
# Top-level convenience for the common "list everything" case.
|
|
717
|
+
# The main ``evaluate()`` entrypoint is imported from ``fi.evals.core``.
|
|
718
|
+
def list_evaluations():
|
|
719
|
+
return Evaluator().list_evaluations()
|
|
720
|
+
|
|
721
|
+
|