agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,577 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Non-blocking evaluation executor.
|
|
3
|
+
|
|
4
|
+
Provides async/background evaluation with zero latency impact on the main thread.
|
|
5
|
+
Uses thread pools for local execution with context propagation.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Dict, Any, List, Optional, Callable
|
|
9
|
+
from concurrent.futures import ThreadPoolExecutor, Future
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from datetime import datetime, timezone
|
|
12
|
+
import threading
|
|
13
|
+
import time
|
|
14
|
+
|
|
15
|
+
from ..types import FrameworkEvalResult as EvalResult, EvalStatus, BatchEvalResult
|
|
16
|
+
from ..context import EvalContext
|
|
17
|
+
from ..protocols import BaseEvaluation
|
|
18
|
+
from ..registry import register_current_span
|
|
19
|
+
from ..propagation import ContextCarrier
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class EvalFuture:
|
|
24
|
+
"""
|
|
25
|
+
Future representing a pending evaluation.
|
|
26
|
+
|
|
27
|
+
Wraps the underlying executor future and provides convenience methods.
|
|
28
|
+
|
|
29
|
+
Example:
|
|
30
|
+
future = evaluator.evaluate(inputs)
|
|
31
|
+
# Do other work...
|
|
32
|
+
result = future.result() # Block until complete
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
future: Future
|
|
36
|
+
eval_name: str
|
|
37
|
+
eval_version: str
|
|
38
|
+
context: Optional[EvalContext] = None
|
|
39
|
+
submitted_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
|
40
|
+
|
|
41
|
+
def done(self) -> bool:
|
|
42
|
+
"""Check if evaluation is complete."""
|
|
43
|
+
return self.future.done()
|
|
44
|
+
|
|
45
|
+
def result(self, timeout: Optional[float] = None) -> EvalResult:
|
|
46
|
+
"""
|
|
47
|
+
Get the evaluation result, blocking if necessary.
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
timeout: Maximum seconds to wait (None = wait forever)
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
EvalResult from the evaluation
|
|
54
|
+
|
|
55
|
+
Raises:
|
|
56
|
+
TimeoutError: If timeout exceeded
|
|
57
|
+
Exception: If evaluation raised an exception
|
|
58
|
+
"""
|
|
59
|
+
return self.future.result(timeout=timeout)
|
|
60
|
+
|
|
61
|
+
def cancel(self) -> bool:
|
|
62
|
+
"""
|
|
63
|
+
Attempt to cancel the evaluation.
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
True if cancelled, False if already running/complete
|
|
67
|
+
"""
|
|
68
|
+
return self.future.cancel()
|
|
69
|
+
|
|
70
|
+
def cancelled(self) -> bool:
|
|
71
|
+
"""Check if evaluation was cancelled."""
|
|
72
|
+
return self.future.cancelled()
|
|
73
|
+
|
|
74
|
+
def add_done_callback(self, fn: Callable[["EvalFuture"], None]) -> None:
|
|
75
|
+
"""
|
|
76
|
+
Add callback to run when evaluation completes.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
fn: Callback function taking this EvalFuture
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
def wrapper(f):
|
|
83
|
+
fn(self)
|
|
84
|
+
|
|
85
|
+
self.future.add_done_callback(wrapper)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass
|
|
89
|
+
class BatchEvalFuture:
|
|
90
|
+
"""Future representing a batch of evaluations."""
|
|
91
|
+
|
|
92
|
+
futures: List[EvalFuture]
|
|
93
|
+
submitted_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
|
94
|
+
|
|
95
|
+
def done(self) -> bool:
|
|
96
|
+
"""Check if all evaluations are complete."""
|
|
97
|
+
return all(f.done() for f in self.futures)
|
|
98
|
+
|
|
99
|
+
def results(self, timeout: Optional[float] = None) -> BatchEvalResult:
|
|
100
|
+
"""
|
|
101
|
+
Get all results, blocking if necessary.
|
|
102
|
+
|
|
103
|
+
Args:
|
|
104
|
+
timeout: Maximum seconds to wait for ALL results
|
|
105
|
+
|
|
106
|
+
Returns:
|
|
107
|
+
BatchEvalResult with all evaluation results
|
|
108
|
+
"""
|
|
109
|
+
results = []
|
|
110
|
+
for f in self.futures:
|
|
111
|
+
try:
|
|
112
|
+
results.append(f.result(timeout=timeout))
|
|
113
|
+
except Exception as e:
|
|
114
|
+
# Create failure result for exceptions
|
|
115
|
+
results.append(
|
|
116
|
+
EvalResult.failure(
|
|
117
|
+
eval_name=f.eval_name,
|
|
118
|
+
eval_version=f.eval_version,
|
|
119
|
+
error=str(e),
|
|
120
|
+
)
|
|
121
|
+
)
|
|
122
|
+
return BatchEvalResult.from_results(results)
|
|
123
|
+
|
|
124
|
+
def cancel_all(self) -> int:
|
|
125
|
+
"""
|
|
126
|
+
Cancel all pending evaluations.
|
|
127
|
+
|
|
128
|
+
Returns:
|
|
129
|
+
Number of successfully cancelled evaluations
|
|
130
|
+
"""
|
|
131
|
+
return sum(1 for f in self.futures if f.cancel())
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class NonBlockingEvaluator:
|
|
135
|
+
"""
|
|
136
|
+
Non-blocking evaluation executor using thread pool.
|
|
137
|
+
|
|
138
|
+
Runs evaluations in background threads with zero latency impact.
|
|
139
|
+
Automatically enriches spans with results when complete.
|
|
140
|
+
|
|
141
|
+
Example:
|
|
142
|
+
evaluator = NonBlockingEvaluator(
|
|
143
|
+
evaluations=[ToxicityEval(), BiasEval()],
|
|
144
|
+
max_workers=4,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
# Returns immediately
|
|
148
|
+
future = evaluator.evaluate({"response": "..."})
|
|
149
|
+
|
|
150
|
+
# Later, check results
|
|
151
|
+
if future.done():
|
|
152
|
+
result = future.result()
|
|
153
|
+
|
|
154
|
+
# Or wait for result
|
|
155
|
+
result = future.result(timeout=5.0)
|
|
156
|
+
|
|
157
|
+
Thread Safety:
|
|
158
|
+
This class is thread-safe. Multiple threads can call evaluate()
|
|
159
|
+
concurrently.
|
|
160
|
+
"""
|
|
161
|
+
|
|
162
|
+
def __init__(
|
|
163
|
+
self,
|
|
164
|
+
evaluations: Optional[List[BaseEvaluation]] = None,
|
|
165
|
+
max_workers: int = 4,
|
|
166
|
+
auto_enrich_span: bool = True,
|
|
167
|
+
fail_fast: bool = False,
|
|
168
|
+
validate_inputs: bool = True,
|
|
169
|
+
executor: Optional[ThreadPoolExecutor] = None,
|
|
170
|
+
):
|
|
171
|
+
"""
|
|
172
|
+
Initialize the non-blocking evaluator.
|
|
173
|
+
|
|
174
|
+
Args:
|
|
175
|
+
evaluations: List of evaluations to run
|
|
176
|
+
max_workers: Maximum concurrent evaluations
|
|
177
|
+
auto_enrich_span: Whether to auto-enrich spans with results
|
|
178
|
+
fail_fast: Stop on first failure (for batch operations)
|
|
179
|
+
validate_inputs: Whether to validate inputs before evaluation
|
|
180
|
+
executor: Custom executor (uses internal pool if None)
|
|
181
|
+
"""
|
|
182
|
+
self.evaluations = list(evaluations) if evaluations else []
|
|
183
|
+
self.max_workers = max_workers
|
|
184
|
+
self.auto_enrich_span = auto_enrich_span
|
|
185
|
+
self.fail_fast = fail_fast
|
|
186
|
+
self.validate_inputs = validate_inputs
|
|
187
|
+
|
|
188
|
+
# Use provided executor or create our own
|
|
189
|
+
self._executor = executor
|
|
190
|
+
self._owns_executor = executor is None
|
|
191
|
+
self._lock = threading.Lock()
|
|
192
|
+
|
|
193
|
+
@property
|
|
194
|
+
def executor(self) -> ThreadPoolExecutor:
|
|
195
|
+
"""Get or create the thread pool executor."""
|
|
196
|
+
if self._executor is None:
|
|
197
|
+
with self._lock:
|
|
198
|
+
if self._executor is None:
|
|
199
|
+
self._executor = ThreadPoolExecutor(
|
|
200
|
+
max_workers=self.max_workers,
|
|
201
|
+
thread_name_prefix="eval_worker_",
|
|
202
|
+
)
|
|
203
|
+
return self._executor
|
|
204
|
+
|
|
205
|
+
def add_evaluation(self, evaluation: BaseEvaluation) -> "NonBlockingEvaluator":
|
|
206
|
+
"""
|
|
207
|
+
Add an evaluation to run.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
evaluation: The evaluation to add
|
|
211
|
+
|
|
212
|
+
Returns:
|
|
213
|
+
Self for chaining
|
|
214
|
+
"""
|
|
215
|
+
self.evaluations.append(evaluation)
|
|
216
|
+
return self
|
|
217
|
+
|
|
218
|
+
def evaluate(
|
|
219
|
+
self,
|
|
220
|
+
inputs: Dict[str, Any],
|
|
221
|
+
evaluations: Optional[List[BaseEvaluation]] = None,
|
|
222
|
+
context: Optional[EvalContext] = None,
|
|
223
|
+
callback: Optional[Callable[[EvalResult], None]] = None,
|
|
224
|
+
) -> BatchEvalFuture:
|
|
225
|
+
"""
|
|
226
|
+
Run evaluations in background, returning immediately.
|
|
227
|
+
|
|
228
|
+
Args:
|
|
229
|
+
inputs: Input data for evaluations
|
|
230
|
+
evaluations: Override evaluations (uses instance evals if None)
|
|
231
|
+
context: Trace context for span enrichment
|
|
232
|
+
callback: Optional callback for each result
|
|
233
|
+
|
|
234
|
+
Returns:
|
|
235
|
+
BatchEvalFuture to track/retrieve results
|
|
236
|
+
|
|
237
|
+
Raises:
|
|
238
|
+
ValueError: If no evaluations configured
|
|
239
|
+
"""
|
|
240
|
+
evals_to_run = evaluations if evaluations is not None else self.evaluations
|
|
241
|
+
if not evals_to_run:
|
|
242
|
+
raise ValueError("No evaluations to run")
|
|
243
|
+
|
|
244
|
+
# Capture context for span enrichment
|
|
245
|
+
carrier = ContextCarrier.capture() if context is None else ContextCarrier(context)
|
|
246
|
+
|
|
247
|
+
# Register current span if enrichment enabled
|
|
248
|
+
if self.auto_enrich_span:
|
|
249
|
+
register_current_span()
|
|
250
|
+
|
|
251
|
+
# Submit all evaluations
|
|
252
|
+
futures = []
|
|
253
|
+
for evaluation in evals_to_run:
|
|
254
|
+
future = self._submit_evaluation(
|
|
255
|
+
evaluation=evaluation,
|
|
256
|
+
inputs=inputs,
|
|
257
|
+
carrier=carrier,
|
|
258
|
+
callback=callback,
|
|
259
|
+
)
|
|
260
|
+
futures.append(future)
|
|
261
|
+
|
|
262
|
+
return BatchEvalFuture(futures=futures)
|
|
263
|
+
|
|
264
|
+
def evaluate_single(
|
|
265
|
+
self,
|
|
266
|
+
evaluation: BaseEvaluation,
|
|
267
|
+
inputs: Dict[str, Any],
|
|
268
|
+
context: Optional[EvalContext] = None,
|
|
269
|
+
callback: Optional[Callable[[EvalResult], None]] = None,
|
|
270
|
+
) -> EvalFuture:
|
|
271
|
+
"""
|
|
272
|
+
Run a single evaluation in background.
|
|
273
|
+
|
|
274
|
+
Args:
|
|
275
|
+
evaluation: The evaluation to run
|
|
276
|
+
inputs: Input data
|
|
277
|
+
context: Trace context
|
|
278
|
+
callback: Optional result callback
|
|
279
|
+
|
|
280
|
+
Returns:
|
|
281
|
+
EvalFuture to track/retrieve result
|
|
282
|
+
"""
|
|
283
|
+
carrier = ContextCarrier.capture() if context is None else ContextCarrier(context)
|
|
284
|
+
|
|
285
|
+
if self.auto_enrich_span:
|
|
286
|
+
register_current_span()
|
|
287
|
+
|
|
288
|
+
return self._submit_evaluation(
|
|
289
|
+
evaluation=evaluation,
|
|
290
|
+
inputs=inputs,
|
|
291
|
+
carrier=carrier,
|
|
292
|
+
callback=callback,
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
def _submit_evaluation(
|
|
296
|
+
self,
|
|
297
|
+
evaluation: BaseEvaluation,
|
|
298
|
+
inputs: Dict[str, Any],
|
|
299
|
+
carrier: ContextCarrier,
|
|
300
|
+
callback: Optional[Callable[[EvalResult], None]] = None,
|
|
301
|
+
) -> EvalFuture:
|
|
302
|
+
"""Submit a single evaluation to the thread pool."""
|
|
303
|
+
eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
|
|
304
|
+
eval_version = getattr(evaluation, "version", "1.0.0")
|
|
305
|
+
|
|
306
|
+
future = self.executor.submit(
|
|
307
|
+
self._run_evaluation,
|
|
308
|
+
evaluation=evaluation,
|
|
309
|
+
inputs=inputs,
|
|
310
|
+
carrier=carrier,
|
|
311
|
+
callback=callback,
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
return EvalFuture(
|
|
315
|
+
future=future,
|
|
316
|
+
eval_name=eval_name,
|
|
317
|
+
eval_version=eval_version,
|
|
318
|
+
context=carrier.context,
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
def _run_evaluation(
|
|
322
|
+
self,
|
|
323
|
+
evaluation: BaseEvaluation,
|
|
324
|
+
inputs: Dict[str, Any],
|
|
325
|
+
carrier: ContextCarrier,
|
|
326
|
+
callback: Optional[Callable[[EvalResult], None]] = None,
|
|
327
|
+
) -> EvalResult:
|
|
328
|
+
"""
|
|
329
|
+
Run evaluation in background thread.
|
|
330
|
+
|
|
331
|
+
Handles:
|
|
332
|
+
- Input validation
|
|
333
|
+
- Timing
|
|
334
|
+
- Error handling
|
|
335
|
+
- Span enrichment
|
|
336
|
+
- Callback invocation
|
|
337
|
+
"""
|
|
338
|
+
eval_name = getattr(evaluation, "name", evaluation.__class__.__name__)
|
|
339
|
+
eval_version = getattr(evaluation, "version", "1.0.0")
|
|
340
|
+
|
|
341
|
+
start_time = time.perf_counter()
|
|
342
|
+
|
|
343
|
+
try:
|
|
344
|
+
# Validate inputs if enabled
|
|
345
|
+
if self.validate_inputs:
|
|
346
|
+
validate = getattr(evaluation, "validate_inputs", None)
|
|
347
|
+
if validate:
|
|
348
|
+
errors = validate(inputs)
|
|
349
|
+
if errors:
|
|
350
|
+
return EvalResult.failure(
|
|
351
|
+
eval_name=eval_name,
|
|
352
|
+
eval_version=eval_version,
|
|
353
|
+
error=f"Validation errors: {errors}",
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
# Run the evaluation
|
|
357
|
+
value = evaluation.evaluate(inputs)
|
|
358
|
+
|
|
359
|
+
latency_ms = (time.perf_counter() - start_time) * 1000
|
|
360
|
+
|
|
361
|
+
result = EvalResult(
|
|
362
|
+
value=value,
|
|
363
|
+
eval_name=eval_name,
|
|
364
|
+
eval_version=eval_version,
|
|
365
|
+
latency_ms=latency_ms,
|
|
366
|
+
status=EvalStatus.COMPLETED,
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
# Enrich span with results
|
|
370
|
+
if self.auto_enrich_span:
|
|
371
|
+
self._enrich_span(evaluation, result, carrier)
|
|
372
|
+
|
|
373
|
+
# Invoke callback
|
|
374
|
+
if callback:
|
|
375
|
+
try:
|
|
376
|
+
callback(result)
|
|
377
|
+
except Exception:
|
|
378
|
+
pass # Don't fail evaluation for callback errors
|
|
379
|
+
|
|
380
|
+
return result
|
|
381
|
+
|
|
382
|
+
except Exception as e:
|
|
383
|
+
latency_ms = (time.perf_counter() - start_time) * 1000
|
|
384
|
+
|
|
385
|
+
result = EvalResult.failure(
|
|
386
|
+
eval_name=eval_name,
|
|
387
|
+
eval_version=eval_version,
|
|
388
|
+
error=str(e),
|
|
389
|
+
)
|
|
390
|
+
result.latency_ms = latency_ms
|
|
391
|
+
|
|
392
|
+
# Enrich span with failure
|
|
393
|
+
if self.auto_enrich_span:
|
|
394
|
+
self._enrich_span_failure(eval_name, result, carrier)
|
|
395
|
+
|
|
396
|
+
# Invoke callback even on failure
|
|
397
|
+
if callback:
|
|
398
|
+
try:
|
|
399
|
+
callback(result)
|
|
400
|
+
except Exception:
|
|
401
|
+
pass
|
|
402
|
+
|
|
403
|
+
return result
|
|
404
|
+
|
|
405
|
+
def _enrich_span(
|
|
406
|
+
self,
|
|
407
|
+
evaluation: BaseEvaluation,
|
|
408
|
+
result: EvalResult,
|
|
409
|
+
carrier: ContextCarrier,
|
|
410
|
+
) -> bool:
|
|
411
|
+
"""Enrich the original span with evaluation results."""
|
|
412
|
+
eval_name = result.eval_name.replace("-", "_").replace(" ", "_")
|
|
413
|
+
prefix = f"eval.{eval_name}"
|
|
414
|
+
|
|
415
|
+
# Get span attributes from evaluation
|
|
416
|
+
get_attrs = getattr(evaluation, "get_span_attributes", None)
|
|
417
|
+
if get_attrs:
|
|
418
|
+
try:
|
|
419
|
+
eval_attrs = get_attrs(result.value)
|
|
420
|
+
except Exception:
|
|
421
|
+
eval_attrs = {}
|
|
422
|
+
else:
|
|
423
|
+
eval_attrs = {}
|
|
424
|
+
|
|
425
|
+
# Build attributes
|
|
426
|
+
attributes = {
|
|
427
|
+
f"{prefix}.status": result.status.value,
|
|
428
|
+
f"{prefix}.latency_ms": result.latency_ms,
|
|
429
|
+
f"{prefix}.version": result.eval_version,
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
# Add eval-specific attributes
|
|
433
|
+
for key, value in eval_attrs.items():
|
|
434
|
+
if isinstance(value, (str, int, float, bool)):
|
|
435
|
+
attributes[f"{prefix}.{key}"] = value
|
|
436
|
+
|
|
437
|
+
return carrier.enrich_span(attributes)
|
|
438
|
+
|
|
439
|
+
def _enrich_span_failure(
|
|
440
|
+
self,
|
|
441
|
+
eval_name: str,
|
|
442
|
+
result: EvalResult,
|
|
443
|
+
carrier: ContextCarrier,
|
|
444
|
+
) -> bool:
|
|
445
|
+
"""Enrich span with failure information."""
|
|
446
|
+
safe_name = eval_name.replace("-", "_").replace(" ", "_")
|
|
447
|
+
prefix = f"eval.{safe_name}"
|
|
448
|
+
|
|
449
|
+
attributes = {
|
|
450
|
+
f"{prefix}.status": result.status.value,
|
|
451
|
+
f"{prefix}.latency_ms": result.latency_ms,
|
|
452
|
+
f"{prefix}.error": result.error or "Unknown error",
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
return carrier.enrich_span(attributes)
|
|
456
|
+
|
|
457
|
+
def shutdown(self, wait: bool = True) -> None:
|
|
458
|
+
"""
|
|
459
|
+
Shutdown the executor.
|
|
460
|
+
|
|
461
|
+
Args:
|
|
462
|
+
wait: Whether to wait for pending tasks to complete
|
|
463
|
+
"""
|
|
464
|
+
if self._executor and self._owns_executor:
|
|
465
|
+
self._executor.shutdown(wait=wait)
|
|
466
|
+
self._executor = None
|
|
467
|
+
|
|
468
|
+
def __enter__(self) -> "NonBlockingEvaluator":
|
|
469
|
+
return self
|
|
470
|
+
|
|
471
|
+
def __exit__(self, exc_type, exc_val, exc_tb) -> None:
|
|
472
|
+
self.shutdown(wait=True)
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def non_blocking_evaluate(
|
|
476
|
+
inputs: Dict[str, Any],
|
|
477
|
+
*evaluations: BaseEvaluation,
|
|
478
|
+
max_workers: int = 4,
|
|
479
|
+
auto_enrich_span: bool = True,
|
|
480
|
+
callback: Optional[Callable[[EvalResult], None]] = None,
|
|
481
|
+
) -> BatchEvalFuture:
|
|
482
|
+
"""
|
|
483
|
+
Convenience function for non-blocking evaluation.
|
|
484
|
+
|
|
485
|
+
Runs evaluations in background and returns immediately.
|
|
486
|
+
The internal executor is automatically shut down when all futures complete.
|
|
487
|
+
|
|
488
|
+
Args:
|
|
489
|
+
inputs: Input data for evaluations
|
|
490
|
+
*evaluations: Evaluations to run
|
|
491
|
+
max_workers: Maximum concurrent evaluations
|
|
492
|
+
auto_enrich_span: Whether to enrich spans with results
|
|
493
|
+
callback: Optional callback for each result
|
|
494
|
+
|
|
495
|
+
Returns:
|
|
496
|
+
BatchEvalFuture to track/retrieve results
|
|
497
|
+
|
|
498
|
+
Example:
|
|
499
|
+
future = non_blocking_evaluate(
|
|
500
|
+
{"response": "..."},
|
|
501
|
+
ToxicityEval(),
|
|
502
|
+
BiasEval(),
|
|
503
|
+
)
|
|
504
|
+
# Do other work...
|
|
505
|
+
results = future.results()
|
|
506
|
+
"""
|
|
507
|
+
evaluator = NonBlockingEvaluator(
|
|
508
|
+
evaluations=list(evaluations),
|
|
509
|
+
max_workers=max_workers,
|
|
510
|
+
auto_enrich_span=auto_enrich_span,
|
|
511
|
+
)
|
|
512
|
+
batch_future = evaluator.evaluate(inputs, callback=callback)
|
|
513
|
+
|
|
514
|
+
# Auto-shutdown executor when all futures complete
|
|
515
|
+
pending = [f.future for f in batch_future.futures]
|
|
516
|
+
remaining = {id(f) for f in pending}
|
|
517
|
+
|
|
518
|
+
def _on_done(f):
|
|
519
|
+
remaining.discard(id(f))
|
|
520
|
+
if not remaining:
|
|
521
|
+
evaluator.shutdown(wait=False)
|
|
522
|
+
|
|
523
|
+
for f in pending:
|
|
524
|
+
f.add_done_callback(_on_done)
|
|
525
|
+
|
|
526
|
+
return batch_future
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
class EvalResultAggregator:
|
|
530
|
+
"""
|
|
531
|
+
Aggregates results from multiple async evaluations.
|
|
532
|
+
|
|
533
|
+
Useful for collecting results from different evaluation runs.
|
|
534
|
+
|
|
535
|
+
Example:
|
|
536
|
+
aggregator = EvalResultAggregator()
|
|
537
|
+
|
|
538
|
+
# Add results as they come in
|
|
539
|
+
aggregator.add(future1.result())
|
|
540
|
+
aggregator.add(future2.result())
|
|
541
|
+
|
|
542
|
+
# Get aggregated results
|
|
543
|
+
batch = aggregator.to_batch()
|
|
544
|
+
print(f"Success rate: {batch.success_rate}")
|
|
545
|
+
"""
|
|
546
|
+
|
|
547
|
+
def __init__(self):
|
|
548
|
+
self._results: List[EvalResult] = []
|
|
549
|
+
self._lock = threading.Lock()
|
|
550
|
+
|
|
551
|
+
def add(self, result: EvalResult) -> None:
|
|
552
|
+
"""Add a result to the aggregator."""
|
|
553
|
+
with self._lock:
|
|
554
|
+
self._results.append(result)
|
|
555
|
+
|
|
556
|
+
def add_all(self, results: List[EvalResult]) -> None:
|
|
557
|
+
"""Add multiple results."""
|
|
558
|
+
with self._lock:
|
|
559
|
+
self._results.extend(results)
|
|
560
|
+
|
|
561
|
+
def to_batch(self) -> BatchEvalResult:
|
|
562
|
+
"""Get aggregated results as a batch."""
|
|
563
|
+
with self._lock:
|
|
564
|
+
return BatchEvalResult.from_results(list(self._results))
|
|
565
|
+
|
|
566
|
+
def clear(self) -> int:
|
|
567
|
+
"""Clear all results and return count cleared."""
|
|
568
|
+
with self._lock:
|
|
569
|
+
count = len(self._results)
|
|
570
|
+
self._results.clear()
|
|
571
|
+
return count
|
|
572
|
+
|
|
573
|
+
@property
|
|
574
|
+
def count(self) -> int:
|
|
575
|
+
"""Get current result count."""
|
|
576
|
+
with self._lock:
|
|
577
|
+
return len(self._results)
|