agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,433 @@
|
|
|
1
|
+
"""
|
|
2
|
+
ResilientBackend wrapper for adding resilience to any backend.
|
|
3
|
+
|
|
4
|
+
Combines circuit breaker, rate limiting, retry, and fallback capabilities.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import Any, Callable, Dict, List, Optional, TypeVar
|
|
10
|
+
|
|
11
|
+
from fi.evals.framework.backends.base import (
|
|
12
|
+
Backend,
|
|
13
|
+
BackendConfig,
|
|
14
|
+
TaskHandle,
|
|
15
|
+
TaskStatus,
|
|
16
|
+
)
|
|
17
|
+
from .circuit_breaker import CircuitBreaker
|
|
18
|
+
from .degradation import FallbackChain
|
|
19
|
+
from .health import HealthChecker
|
|
20
|
+
from .rate_limiter import TokenBucketRateLimiter
|
|
21
|
+
from .retry import RetryHandler
|
|
22
|
+
from .types import (
|
|
23
|
+
CircuitBreakerConfig,
|
|
24
|
+
CircuitOpenError,
|
|
25
|
+
DegradationConfig,
|
|
26
|
+
EventCallback,
|
|
27
|
+
HealthCheckConfig,
|
|
28
|
+
RateLimitConfig,
|
|
29
|
+
RateLimitExceededError,
|
|
30
|
+
ResilienceConfig,
|
|
31
|
+
ResilienceEvent,
|
|
32
|
+
ResilienceEventType,
|
|
33
|
+
RetryConfig,
|
|
34
|
+
RetryExhaustedError,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
T = TypeVar("T")
|
|
38
|
+
logger = logging.getLogger(__name__)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class ResilientBackendConfig(BackendConfig):
|
|
43
|
+
"""Configuration for ResilientBackend wrapper."""
|
|
44
|
+
|
|
45
|
+
resilience: ResilienceConfig = field(default_factory=ResilienceConfig)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ResilientBackend(Backend):
|
|
49
|
+
"""
|
|
50
|
+
Wraps any backend with resilience features.
|
|
51
|
+
|
|
52
|
+
Applies the following protections in order:
|
|
53
|
+
1. Rate limiting - prevent overwhelming the backend
|
|
54
|
+
2. Circuit breaker - fail fast if backend is unhealthy
|
|
55
|
+
3. Retry - retry transient failures
|
|
56
|
+
4. Fallback - use alternative if all else fails
|
|
57
|
+
|
|
58
|
+
Example:
|
|
59
|
+
from fi.evals.framework.backends import ThreadPoolBackend
|
|
60
|
+
from fi.evals.framework.resilience import (
|
|
61
|
+
ResilientBackend,
|
|
62
|
+
CircuitBreakerConfig,
|
|
63
|
+
RateLimitConfig,
|
|
64
|
+
ResilienceConfig,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
config = ResilienceConfig(
|
|
68
|
+
circuit_breaker=CircuitBreakerConfig(failure_threshold=5),
|
|
69
|
+
rate_limit=RateLimitConfig(requests_per_second=10),
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
underlying = ThreadPoolBackend()
|
|
73
|
+
backend = ResilientBackend(underlying, config)
|
|
74
|
+
|
|
75
|
+
# Use normally - resilience is transparent
|
|
76
|
+
handle = backend.submit(my_func, args=(1, 2))
|
|
77
|
+
result = backend.get_result(handle)
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
name: str = "resilient"
|
|
81
|
+
|
|
82
|
+
def __init__(
|
|
83
|
+
self,
|
|
84
|
+
underlying: Backend,
|
|
85
|
+
config: Optional[ResilienceConfig] = None,
|
|
86
|
+
fallback_backend: Optional[Backend] = None,
|
|
87
|
+
event_callback: Optional[EventCallback] = None,
|
|
88
|
+
):
|
|
89
|
+
"""
|
|
90
|
+
Initialize ResilientBackend wrapper.
|
|
91
|
+
|
|
92
|
+
Args:
|
|
93
|
+
underlying: The backend to wrap
|
|
94
|
+
config: Resilience configuration
|
|
95
|
+
fallback_backend: Optional fallback backend
|
|
96
|
+
event_callback: Callback for resilience events
|
|
97
|
+
"""
|
|
98
|
+
self.underlying = underlying
|
|
99
|
+
self.config = config or ResilienceConfig()
|
|
100
|
+
self.fallback_backend = fallback_backend
|
|
101
|
+
self.event_callback = event_callback
|
|
102
|
+
self.name = f"resilient({underlying.name})"
|
|
103
|
+
|
|
104
|
+
# Initialize components based on config
|
|
105
|
+
self._rate_limiter: Optional[TokenBucketRateLimiter] = None
|
|
106
|
+
self._circuit_breaker: Optional[CircuitBreaker] = None
|
|
107
|
+
self._retry_handler: Optional[RetryHandler] = None
|
|
108
|
+
self._fallback_chain: Optional[FallbackChain] = None
|
|
109
|
+
self._health_checker: Optional[HealthChecker] = None
|
|
110
|
+
|
|
111
|
+
self._setup_components()
|
|
112
|
+
|
|
113
|
+
def _setup_components(self) -> None:
|
|
114
|
+
"""Initialize resilience components based on config."""
|
|
115
|
+
# Rate limiter
|
|
116
|
+
if self.config.rate_limit:
|
|
117
|
+
self._rate_limiter = TokenBucketRateLimiter(
|
|
118
|
+
name=self.name,
|
|
119
|
+
config=self.config.rate_limit,
|
|
120
|
+
event_callback=self.event_callback,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
# Circuit breaker
|
|
124
|
+
if self.config.circuit_breaker:
|
|
125
|
+
self._circuit_breaker = CircuitBreaker(
|
|
126
|
+
name=self.name,
|
|
127
|
+
config=self.config.circuit_breaker,
|
|
128
|
+
event_callback=self.event_callback,
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
# Retry handler
|
|
132
|
+
if self.config.retry:
|
|
133
|
+
self._retry_handler = RetryHandler(
|
|
134
|
+
name=self.name,
|
|
135
|
+
config=self.config.retry,
|
|
136
|
+
event_callback=self.event_callback,
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
# Fallback chain (if fallback backend configured)
|
|
140
|
+
if self.fallback_backend and self.config.degradation:
|
|
141
|
+
self._fallback_chain = FallbackChain(
|
|
142
|
+
name=self.name,
|
|
143
|
+
config=self.config.degradation,
|
|
144
|
+
event_callback=self.event_callback,
|
|
145
|
+
)
|
|
146
|
+
self._fallback_chain.add_fallback(
|
|
147
|
+
"fallback_backend",
|
|
148
|
+
lambda: None, # Placeholder - actual fallback handled in submit
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
# Health checker
|
|
152
|
+
if self.config.health_check:
|
|
153
|
+
self._health_checker = HealthChecker(
|
|
154
|
+
name=self.name,
|
|
155
|
+
check_func=self._health_check,
|
|
156
|
+
config=self.config.health_check,
|
|
157
|
+
event_callback=self.event_callback,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def _health_check(self) -> bool:
|
|
161
|
+
"""Health check for the underlying backend."""
|
|
162
|
+
# If circuit breaker exists, check its state
|
|
163
|
+
if self._circuit_breaker:
|
|
164
|
+
return self._circuit_breaker.is_closed
|
|
165
|
+
|
|
166
|
+
# Default: assume healthy
|
|
167
|
+
return True
|
|
168
|
+
|
|
169
|
+
def submit(
|
|
170
|
+
self,
|
|
171
|
+
fn: Callable[..., T],
|
|
172
|
+
args: tuple = (),
|
|
173
|
+
kwargs: Optional[Dict[str, Any]] = None,
|
|
174
|
+
context: Optional[Dict[str, Any]] = None,
|
|
175
|
+
) -> TaskHandle[T]:
|
|
176
|
+
"""
|
|
177
|
+
Submit a task with resilience protections.
|
|
178
|
+
|
|
179
|
+
Applies rate limiting, circuit breaker, retry, and fallback.
|
|
180
|
+
|
|
181
|
+
Args:
|
|
182
|
+
fn: The function to execute
|
|
183
|
+
args: Positional arguments
|
|
184
|
+
kwargs: Keyword arguments
|
|
185
|
+
context: Trace context
|
|
186
|
+
|
|
187
|
+
Returns:
|
|
188
|
+
TaskHandle to track the task
|
|
189
|
+
|
|
190
|
+
Raises:
|
|
191
|
+
RateLimitExceededError: If rate limit exceeded and blocking disabled
|
|
192
|
+
CircuitOpenError: If circuit breaker is open
|
|
193
|
+
RetryExhaustedError: If all retries fail
|
|
194
|
+
"""
|
|
195
|
+
kwargs = kwargs or {}
|
|
196
|
+
|
|
197
|
+
def do_submit() -> TaskHandle[T]:
|
|
198
|
+
"""Perform the actual submission."""
|
|
199
|
+
return self.underlying.submit(fn, args, kwargs, context)
|
|
200
|
+
|
|
201
|
+
def submit_with_protections() -> TaskHandle[T]:
|
|
202
|
+
"""Apply all resilience protections."""
|
|
203
|
+
# 1. Rate limiting
|
|
204
|
+
if self._rate_limiter:
|
|
205
|
+
if not self._rate_limiter.acquire():
|
|
206
|
+
raise RateLimitExceededError(
|
|
207
|
+
self.name,
|
|
208
|
+
self._rate_limiter.get_wait_time(),
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
# 2. Circuit breaker
|
|
212
|
+
if self._circuit_breaker:
|
|
213
|
+
return self._circuit_breaker.execute(do_submit)
|
|
214
|
+
|
|
215
|
+
return do_submit()
|
|
216
|
+
|
|
217
|
+
def submit_with_retry() -> TaskHandle[T]:
|
|
218
|
+
"""Apply retry logic."""
|
|
219
|
+
if self._retry_handler:
|
|
220
|
+
return self._retry_handler.execute(submit_with_protections)
|
|
221
|
+
return submit_with_protections()
|
|
222
|
+
|
|
223
|
+
# 3. Retry + fallback
|
|
224
|
+
try:
|
|
225
|
+
return submit_with_retry()
|
|
226
|
+
except (CircuitOpenError, RateLimitExceededError, RetryExhaustedError):
|
|
227
|
+
# Try fallback if available
|
|
228
|
+
if self.fallback_backend and self.config.degradation:
|
|
229
|
+
if self.config.degradation.fallback_on_circuit_open:
|
|
230
|
+
self._emit_fallback_event("circuit_open_or_rate_limit")
|
|
231
|
+
return self.fallback_backend.submit(fn, args, kwargs, context)
|
|
232
|
+
raise
|
|
233
|
+
|
|
234
|
+
def get_result(
|
|
235
|
+
self,
|
|
236
|
+
handle: TaskHandle[T],
|
|
237
|
+
timeout: Optional[float] = None,
|
|
238
|
+
) -> T:
|
|
239
|
+
"""
|
|
240
|
+
Get the result of a submitted task.
|
|
241
|
+
|
|
242
|
+
Args:
|
|
243
|
+
handle: The task handle from submit()
|
|
244
|
+
timeout: Maximum seconds to wait
|
|
245
|
+
|
|
246
|
+
Returns:
|
|
247
|
+
The task result
|
|
248
|
+
|
|
249
|
+
Raises:
|
|
250
|
+
TimeoutError: If timeout exceeded
|
|
251
|
+
Exception: If task failed
|
|
252
|
+
"""
|
|
253
|
+
# Determine which backend to use based on handle metadata
|
|
254
|
+
backend = self._get_backend_for_handle(handle)
|
|
255
|
+
return backend.get_result(handle, timeout)
|
|
256
|
+
|
|
257
|
+
def get_status(self, handle: TaskHandle) -> TaskStatus:
|
|
258
|
+
"""Get current status of a task."""
|
|
259
|
+
backend = self._get_backend_for_handle(handle)
|
|
260
|
+
return backend.get_status(handle)
|
|
261
|
+
|
|
262
|
+
def cancel(self, handle: TaskHandle) -> bool:
|
|
263
|
+
"""Attempt to cancel a task."""
|
|
264
|
+
backend = self._get_backend_for_handle(handle)
|
|
265
|
+
return backend.cancel(handle)
|
|
266
|
+
|
|
267
|
+
def _get_backend_for_handle(self, handle: TaskHandle) -> Backend:
|
|
268
|
+
"""Determine which backend handles this task."""
|
|
269
|
+
# Check if handle is from fallback backend
|
|
270
|
+
if (
|
|
271
|
+
self.fallback_backend
|
|
272
|
+
and handle.backend_name == self.fallback_backend.name
|
|
273
|
+
):
|
|
274
|
+
return self.fallback_backend
|
|
275
|
+
return self.underlying
|
|
276
|
+
|
|
277
|
+
def submit_batch(
|
|
278
|
+
self,
|
|
279
|
+
tasks: List[tuple],
|
|
280
|
+
) -> List[TaskHandle]:
|
|
281
|
+
"""Submit multiple tasks with resilience protections."""
|
|
282
|
+
handles = []
|
|
283
|
+
for fn, args, kwargs, context in tasks:
|
|
284
|
+
handle = self.submit(fn, args, kwargs or {}, context)
|
|
285
|
+
handles.append(handle)
|
|
286
|
+
return handles
|
|
287
|
+
|
|
288
|
+
def shutdown(self, wait: bool = True) -> None:
|
|
289
|
+
"""Shutdown the backend and all components."""
|
|
290
|
+
# Stop health checker if running
|
|
291
|
+
if self._health_checker and self._health_checker.is_running:
|
|
292
|
+
self._health_checker.stop()
|
|
293
|
+
|
|
294
|
+
# Shutdown underlying backend
|
|
295
|
+
self.underlying.shutdown(wait)
|
|
296
|
+
|
|
297
|
+
# Shutdown fallback if exists
|
|
298
|
+
if self.fallback_backend:
|
|
299
|
+
self.fallback_backend.shutdown(wait)
|
|
300
|
+
|
|
301
|
+
def _emit_fallback_event(self, reason: str) -> None:
|
|
302
|
+
"""Emit fallback event."""
|
|
303
|
+
if not self.event_callback:
|
|
304
|
+
return
|
|
305
|
+
|
|
306
|
+
event = ResilienceEvent(
|
|
307
|
+
event_type=ResilienceEventType.FALLBACK_USED,
|
|
308
|
+
backend_name=self.name,
|
|
309
|
+
metadata={
|
|
310
|
+
"reason": reason,
|
|
311
|
+
"fallback_backend": (
|
|
312
|
+
self.fallback_backend.name if self.fallback_backend else None
|
|
313
|
+
),
|
|
314
|
+
},
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
try:
|
|
318
|
+
self.event_callback(event)
|
|
319
|
+
except Exception:
|
|
320
|
+
pass
|
|
321
|
+
|
|
322
|
+
# === Component access for testing/monitoring ===
|
|
323
|
+
|
|
324
|
+
@property
|
|
325
|
+
def rate_limiter(self) -> Optional[TokenBucketRateLimiter]:
|
|
326
|
+
"""Get rate limiter component."""
|
|
327
|
+
return self._rate_limiter
|
|
328
|
+
|
|
329
|
+
@property
|
|
330
|
+
def circuit_breaker(self) -> Optional[CircuitBreaker]:
|
|
331
|
+
"""Get circuit breaker component."""
|
|
332
|
+
return self._circuit_breaker
|
|
333
|
+
|
|
334
|
+
@property
|
|
335
|
+
def retry_handler(self) -> Optional[RetryHandler]:
|
|
336
|
+
"""Get retry handler component."""
|
|
337
|
+
return self._retry_handler
|
|
338
|
+
|
|
339
|
+
@property
|
|
340
|
+
def health_checker(self) -> Optional[HealthChecker]:
|
|
341
|
+
"""Get health checker component."""
|
|
342
|
+
return self._health_checker
|
|
343
|
+
|
|
344
|
+
def get_stats(self) -> dict:
|
|
345
|
+
"""Get combined statistics from all components."""
|
|
346
|
+
stats = {
|
|
347
|
+
"backend": self.name,
|
|
348
|
+
"underlying": self.underlying.name,
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
if self._rate_limiter:
|
|
352
|
+
stats["rate_limiter"] = self._rate_limiter.get_stats()
|
|
353
|
+
|
|
354
|
+
if self._circuit_breaker:
|
|
355
|
+
stats["circuit_breaker"] = self._circuit_breaker.get_stats()
|
|
356
|
+
|
|
357
|
+
if self._retry_handler:
|
|
358
|
+
stats["retry"] = self._retry_handler.get_stats()
|
|
359
|
+
|
|
360
|
+
if self._health_checker:
|
|
361
|
+
stats["health"] = self._health_checker.get_stats()
|
|
362
|
+
|
|
363
|
+
return stats
|
|
364
|
+
|
|
365
|
+
def start_health_checks(self) -> None:
|
|
366
|
+
"""Start periodic health checking."""
|
|
367
|
+
if self._health_checker:
|
|
368
|
+
self._health_checker.start()
|
|
369
|
+
|
|
370
|
+
def stop_health_checks(self) -> None:
|
|
371
|
+
"""Stop periodic health checking."""
|
|
372
|
+
if self._health_checker:
|
|
373
|
+
self._health_checker.stop()
|
|
374
|
+
|
|
375
|
+
def reset(self) -> None:
|
|
376
|
+
"""Reset all resilience components."""
|
|
377
|
+
if self._rate_limiter:
|
|
378
|
+
self._rate_limiter.reset()
|
|
379
|
+
|
|
380
|
+
if self._circuit_breaker:
|
|
381
|
+
self._circuit_breaker.reset()
|
|
382
|
+
|
|
383
|
+
if self._health_checker:
|
|
384
|
+
self._health_checker.reset()
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def wrap_backend(
|
|
388
|
+
backend: Backend,
|
|
389
|
+
circuit_breaker: Optional[CircuitBreakerConfig] = None,
|
|
390
|
+
rate_limit: Optional[RateLimitConfig] = None,
|
|
391
|
+
retry: Optional[RetryConfig] = None,
|
|
392
|
+
degradation: Optional[DegradationConfig] = None,
|
|
393
|
+
health_check: Optional[HealthCheckConfig] = None,
|
|
394
|
+
fallback_backend: Optional[Backend] = None,
|
|
395
|
+
event_callback: Optional[EventCallback] = None,
|
|
396
|
+
) -> ResilientBackend:
|
|
397
|
+
"""
|
|
398
|
+
Convenience function to wrap a backend with resilience.
|
|
399
|
+
|
|
400
|
+
Example:
|
|
401
|
+
backend = wrap_backend(
|
|
402
|
+
ThreadPoolBackend(),
|
|
403
|
+
circuit_breaker=CircuitBreakerConfig(failure_threshold=5),
|
|
404
|
+
rate_limit=RateLimitConfig(requests_per_second=10),
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
Args:
|
|
408
|
+
backend: The backend to wrap
|
|
409
|
+
circuit_breaker: Circuit breaker configuration
|
|
410
|
+
rate_limit: Rate limit configuration
|
|
411
|
+
retry: Retry configuration
|
|
412
|
+
degradation: Degradation configuration
|
|
413
|
+
health_check: Health check configuration
|
|
414
|
+
fallback_backend: Optional fallback backend
|
|
415
|
+
event_callback: Callback for events
|
|
416
|
+
|
|
417
|
+
Returns:
|
|
418
|
+
ResilientBackend wrapping the original
|
|
419
|
+
"""
|
|
420
|
+
config = ResilienceConfig(
|
|
421
|
+
circuit_breaker=circuit_breaker,
|
|
422
|
+
rate_limit=rate_limit,
|
|
423
|
+
retry=retry,
|
|
424
|
+
degradation=degradation,
|
|
425
|
+
health_check=health_check,
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
return ResilientBackend(
|
|
429
|
+
underlying=backend,
|
|
430
|
+
config=config,
|
|
431
|
+
fallback_backend=fallback_backend,
|
|
432
|
+
event_callback=event_callback,
|
|
433
|
+
)
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Core types for the evaluation framework.
|
|
3
|
+
|
|
4
|
+
This module defines the foundational types used throughout the framework:
|
|
5
|
+
- ExecutionMode: How evaluations are executed (blocking, non-blocking, distributed)
|
|
6
|
+
- EvalStatus: Status of an evaluation execution
|
|
7
|
+
- EvalResult: Result from a single evaluation
|
|
8
|
+
- BatchEvalResult: Aggregated results from batch evaluation
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import TypeVar, Generic, Any, Dict, Optional, List, Union
|
|
14
|
+
from datetime import datetime, timezone
|
|
15
|
+
|
|
16
|
+
T = TypeVar("T")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ExecutionMode(Enum):
|
|
20
|
+
"""
|
|
21
|
+
How an evaluation should be executed.
|
|
22
|
+
|
|
23
|
+
- BLOCKING: Synchronous execution, waits for result. Introduces latency.
|
|
24
|
+
- NON_BLOCKING: Asynchronous execution, returns immediately. Zero latency impact.
|
|
25
|
+
- DISTRIBUTED: Distributed across workers for batch processing.
|
|
26
|
+
"""
|
|
27
|
+
BLOCKING = "blocking"
|
|
28
|
+
NON_BLOCKING = "non_blocking"
|
|
29
|
+
DISTRIBUTED = "distributed"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class EvalStatus(Enum):
|
|
33
|
+
"""Status of an evaluation execution."""
|
|
34
|
+
PENDING = "pending"
|
|
35
|
+
RUNNING = "running"
|
|
36
|
+
COMPLETED = "completed"
|
|
37
|
+
FAILED = "failed"
|
|
38
|
+
CANCELLED = "cancelled"
|
|
39
|
+
TIMEOUT = "timeout"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class FrameworkEvalResult(Generic[T]):
|
|
44
|
+
"""
|
|
45
|
+
Result from any evaluation.
|
|
46
|
+
|
|
47
|
+
Attributes:
|
|
48
|
+
value: The evaluation result value (typed)
|
|
49
|
+
eval_name: Name of the evaluation that produced this result
|
|
50
|
+
eval_version: Version of the evaluation
|
|
51
|
+
latency_ms: Time taken to run the evaluation in milliseconds
|
|
52
|
+
status: Current status of the evaluation
|
|
53
|
+
error: Error message if status is FAILED
|
|
54
|
+
metadata: Additional metadata about the evaluation run
|
|
55
|
+
timestamp: When the evaluation completed
|
|
56
|
+
|
|
57
|
+
Example:
|
|
58
|
+
result = EvalResult(
|
|
59
|
+
value={"score": 0.95, "passed": True},
|
|
60
|
+
eval_name="faithfulness",
|
|
61
|
+
eval_version="1.0.0",
|
|
62
|
+
latency_ms=150.5,
|
|
63
|
+
)
|
|
64
|
+
"""
|
|
65
|
+
value: T
|
|
66
|
+
eval_name: str
|
|
67
|
+
eval_version: str
|
|
68
|
+
latency_ms: float
|
|
69
|
+
status: EvalStatus = EvalStatus.COMPLETED
|
|
70
|
+
error: Optional[str] = None
|
|
71
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
72
|
+
timestamp: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def succeeded(self) -> bool:
|
|
76
|
+
"""Whether the evaluation completed successfully."""
|
|
77
|
+
return self.status == EvalStatus.COMPLETED
|
|
78
|
+
|
|
79
|
+
@property
|
|
80
|
+
def failed(self) -> bool:
|
|
81
|
+
"""Whether the evaluation failed."""
|
|
82
|
+
return self.status == EvalStatus.FAILED
|
|
83
|
+
|
|
84
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
85
|
+
"""Convert to dictionary for serialization."""
|
|
86
|
+
return {
|
|
87
|
+
"value": self.value,
|
|
88
|
+
"eval_name": self.eval_name,
|
|
89
|
+
"eval_version": self.eval_version,
|
|
90
|
+
"latency_ms": self.latency_ms,
|
|
91
|
+
"status": self.status.value,
|
|
92
|
+
"error": self.error,
|
|
93
|
+
"metadata": self.metadata,
|
|
94
|
+
"timestamp": self.timestamp.isoformat(),
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
@classmethod
|
|
98
|
+
def from_dict(cls, data: Dict[str, Any]) -> "FrameworkEvalResult":
|
|
99
|
+
"""Create from dictionary."""
|
|
100
|
+
return cls(
|
|
101
|
+
value=data["value"],
|
|
102
|
+
eval_name=data["eval_name"],
|
|
103
|
+
eval_version=data["eval_version"],
|
|
104
|
+
latency_ms=data["latency_ms"],
|
|
105
|
+
status=EvalStatus(data["status"]),
|
|
106
|
+
error=data.get("error"),
|
|
107
|
+
metadata=data.get("metadata", {}),
|
|
108
|
+
timestamp=datetime.fromisoformat(data["timestamp"]) if data.get("timestamp") else datetime.utcnow(),
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
def to_span_attributes(self) -> Dict[str, Any]:
|
|
112
|
+
"""
|
|
113
|
+
Convert to flat dict for span attributes.
|
|
114
|
+
|
|
115
|
+
Returns attributes suitable for OTEL span.set_attribute().
|
|
116
|
+
"""
|
|
117
|
+
attrs = {
|
|
118
|
+
"eval_name": self.eval_name,
|
|
119
|
+
"eval_version": self.eval_version,
|
|
120
|
+
"latency_ms": self.latency_ms,
|
|
121
|
+
"status": self.status.value,
|
|
122
|
+
"timestamp": self.timestamp.isoformat(),
|
|
123
|
+
}
|
|
124
|
+
if self.error:
|
|
125
|
+
attrs["error"] = self.error
|
|
126
|
+
return attrs
|
|
127
|
+
|
|
128
|
+
@classmethod
|
|
129
|
+
def failure(
|
|
130
|
+
cls,
|
|
131
|
+
eval_name: str,
|
|
132
|
+
eval_version: str,
|
|
133
|
+
error: str,
|
|
134
|
+
latency_ms: float = 0.0,
|
|
135
|
+
) -> "FrameworkEvalResult":
|
|
136
|
+
"""Create a failed result."""
|
|
137
|
+
return cls(
|
|
138
|
+
value=None,
|
|
139
|
+
eval_name=eval_name,
|
|
140
|
+
eval_version=eval_version,
|
|
141
|
+
latency_ms=latency_ms,
|
|
142
|
+
status=EvalStatus.FAILED,
|
|
143
|
+
error=error,
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
@dataclass
|
|
148
|
+
class BatchEvalResult:
|
|
149
|
+
"""
|
|
150
|
+
Aggregated result from batch evaluation.
|
|
151
|
+
|
|
152
|
+
Attributes:
|
|
153
|
+
results: List of individual EvalResult objects
|
|
154
|
+
total_count: Total number of evaluations attempted
|
|
155
|
+
success_count: Number of successful evaluations
|
|
156
|
+
failure_count: Number of failed evaluations
|
|
157
|
+
total_latency_ms: Sum of all evaluation latencies
|
|
158
|
+
|
|
159
|
+
Example:
|
|
160
|
+
batch = BatchEvalResult.from_results(results)
|
|
161
|
+
print(f"Success rate: {batch.success_rate:.1%}")
|
|
162
|
+
"""
|
|
163
|
+
results: List[FrameworkEvalResult]
|
|
164
|
+
total_count: int
|
|
165
|
+
success_count: int
|
|
166
|
+
failure_count: int
|
|
167
|
+
total_latency_ms: float
|
|
168
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
169
|
+
|
|
170
|
+
@property
|
|
171
|
+
def success_rate(self) -> float:
|
|
172
|
+
"""Percentage of successful evaluations (0.0 to 1.0)."""
|
|
173
|
+
if self.total_count == 0:
|
|
174
|
+
return 0.0
|
|
175
|
+
return self.success_count / self.total_count
|
|
176
|
+
|
|
177
|
+
@property
|
|
178
|
+
def avg_latency_ms(self) -> float:
|
|
179
|
+
"""Average latency per evaluation in milliseconds."""
|
|
180
|
+
if self.total_count == 0:
|
|
181
|
+
return 0.0
|
|
182
|
+
return self.total_latency_ms / self.total_count
|
|
183
|
+
|
|
184
|
+
@classmethod
|
|
185
|
+
def from_results(cls, results: List[FrameworkEvalResult], **metadata) -> "BatchEvalResult":
|
|
186
|
+
"""Create from a list of EvalResult objects."""
|
|
187
|
+
return cls(
|
|
188
|
+
results=results,
|
|
189
|
+
total_count=len(results),
|
|
190
|
+
success_count=sum(1 for r in results if r.status == EvalStatus.COMPLETED),
|
|
191
|
+
failure_count=sum(1 for r in results if r.status == EvalStatus.FAILED),
|
|
192
|
+
total_latency_ms=sum(r.latency_ms for r in results),
|
|
193
|
+
metadata=metadata,
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
def get_by_name(self, eval_name: str) -> List[FrameworkEvalResult]:
|
|
197
|
+
"""Get all results for a specific evaluation name."""
|
|
198
|
+
return [r for r in self.results if r.eval_name == eval_name]
|
|
199
|
+
|
|
200
|
+
def get_failures(self) -> List[FrameworkEvalResult]:
|
|
201
|
+
"""Get all failed results."""
|
|
202
|
+
return [r for r in self.results if r.status == EvalStatus.FAILED]
|
|
203
|
+
|
|
204
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
205
|
+
"""Convert to dictionary for serialization."""
|
|
206
|
+
return {
|
|
207
|
+
"results": [r.to_dict() for r in self.results],
|
|
208
|
+
"total_count": self.total_count,
|
|
209
|
+
"success_count": self.success_count,
|
|
210
|
+
"failure_count": self.failure_count,
|
|
211
|
+
"total_latency_ms": self.total_latency_ms,
|
|
212
|
+
"metadata": self.metadata,
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
# Type aliases for common patterns
|
|
217
|
+
EvalInputs = Dict[str, Any]
|
|
218
|
+
SpanAttributes = Dict[str, Union[str, int, float, bool]]
|