agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Thread pool backend for local evaluation execution.
|
|
3
|
+
|
|
4
|
+
Provides a simple, lightweight backend for running evaluations
|
|
5
|
+
in a local thread pool. Suitable for development and single-machine
|
|
6
|
+
production deployments.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from concurrent.futures import ThreadPoolExecutor, Future, TimeoutError as FuturesTimeout
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Dict, Any, Optional, Callable, TypeVar, List
|
|
12
|
+
from datetime import datetime, timezone
|
|
13
|
+
import threading
|
|
14
|
+
import uuid
|
|
15
|
+
|
|
16
|
+
from .base import Backend, BackendConfig, TaskHandle, TaskStatus
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
T = TypeVar("T")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class ThreadPoolConfig(BackendConfig):
|
|
24
|
+
"""Configuration for thread pool backend."""
|
|
25
|
+
thread_name_prefix: str = "eval_"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ThreadPoolBackend(Backend):
|
|
29
|
+
"""
|
|
30
|
+
Thread pool backend for local execution.
|
|
31
|
+
|
|
32
|
+
Runs evaluation tasks in a local thread pool. Best for:
|
|
33
|
+
- Development and testing
|
|
34
|
+
- Single-machine deployments
|
|
35
|
+
- Low-latency requirements
|
|
36
|
+
|
|
37
|
+
Example:
|
|
38
|
+
config = ThreadPoolConfig(max_workers=8)
|
|
39
|
+
backend = ThreadPoolBackend(config)
|
|
40
|
+
|
|
41
|
+
handle = backend.submit(my_eval_fn, args=(inputs,))
|
|
42
|
+
result = backend.get_result(handle, timeout=30.0)
|
|
43
|
+
|
|
44
|
+
Thread Safety:
|
|
45
|
+
This class is thread-safe.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
name = "thread_pool"
|
|
49
|
+
|
|
50
|
+
def __init__(self, config: Optional[ThreadPoolConfig] = None):
|
|
51
|
+
"""
|
|
52
|
+
Initialize the thread pool backend.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
config: Configuration options (uses defaults if None)
|
|
56
|
+
"""
|
|
57
|
+
self.config = config or ThreadPoolConfig()
|
|
58
|
+
self._executor: Optional[ThreadPoolExecutor] = None
|
|
59
|
+
self._futures: Dict[str, Future] = {}
|
|
60
|
+
self._handles: Dict[str, TaskHandle] = {}
|
|
61
|
+
self._lock = threading.Lock()
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def executor(self) -> ThreadPoolExecutor:
|
|
65
|
+
"""Get or create the thread pool executor."""
|
|
66
|
+
if self._executor is None:
|
|
67
|
+
with self._lock:
|
|
68
|
+
if self._executor is None:
|
|
69
|
+
self._executor = ThreadPoolExecutor(
|
|
70
|
+
max_workers=self.config.max_workers,
|
|
71
|
+
thread_name_prefix=self.config.thread_name_prefix,
|
|
72
|
+
)
|
|
73
|
+
return self._executor
|
|
74
|
+
|
|
75
|
+
def submit(
|
|
76
|
+
self,
|
|
77
|
+
fn: Callable[..., T],
|
|
78
|
+
args: tuple = (),
|
|
79
|
+
kwargs: Optional[Dict[str, Any]] = None,
|
|
80
|
+
context: Optional[Dict[str, Any]] = None,
|
|
81
|
+
) -> TaskHandle[T]:
|
|
82
|
+
"""
|
|
83
|
+
Submit a task to the thread pool.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
fn: Function to execute
|
|
87
|
+
args: Positional arguments
|
|
88
|
+
kwargs: Keyword arguments
|
|
89
|
+
context: Trace context (stored in handle.metadata)
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
TaskHandle to track the task
|
|
93
|
+
"""
|
|
94
|
+
task_id = uuid.uuid4().hex[:16]
|
|
95
|
+
kwargs = kwargs or {}
|
|
96
|
+
|
|
97
|
+
# Create handle
|
|
98
|
+
handle: TaskHandle[T] = TaskHandle(
|
|
99
|
+
task_id=task_id,
|
|
100
|
+
backend_name=self.name,
|
|
101
|
+
metadata={"context": context} if context else {},
|
|
102
|
+
)
|
|
103
|
+
handle._status = TaskStatus.PENDING
|
|
104
|
+
|
|
105
|
+
# Submit to executor
|
|
106
|
+
future = self.executor.submit(fn, *args, **kwargs)
|
|
107
|
+
|
|
108
|
+
# Store mappings
|
|
109
|
+
with self._lock:
|
|
110
|
+
self._futures[task_id] = future
|
|
111
|
+
self._handles[task_id] = handle
|
|
112
|
+
|
|
113
|
+
# Update status when running
|
|
114
|
+
def on_start():
|
|
115
|
+
handle._status = TaskStatus.RUNNING
|
|
116
|
+
|
|
117
|
+
# Update status on completion
|
|
118
|
+
def on_done(f: Future):
|
|
119
|
+
with self._lock:
|
|
120
|
+
handle._completed_at = datetime.now(timezone.utc)
|
|
121
|
+
try:
|
|
122
|
+
result = f.result(timeout=0) # Don't block
|
|
123
|
+
handle._result = result
|
|
124
|
+
handle._status = TaskStatus.COMPLETED
|
|
125
|
+
except FuturesTimeout:
|
|
126
|
+
handle._status = TaskStatus.TIMEOUT
|
|
127
|
+
handle._error = "Task timed out"
|
|
128
|
+
except Exception as e:
|
|
129
|
+
handle._status = TaskStatus.FAILED
|
|
130
|
+
handle._error = str(e)
|
|
131
|
+
|
|
132
|
+
future.add_done_callback(on_done)
|
|
133
|
+
|
|
134
|
+
return handle
|
|
135
|
+
|
|
136
|
+
def get_result(
|
|
137
|
+
self,
|
|
138
|
+
handle: TaskHandle[T],
|
|
139
|
+
timeout: Optional[float] = None,
|
|
140
|
+
) -> T:
|
|
141
|
+
"""
|
|
142
|
+
Get result from a submitted task.
|
|
143
|
+
|
|
144
|
+
Args:
|
|
145
|
+
handle: The task handle
|
|
146
|
+
timeout: Maximum seconds to wait
|
|
147
|
+
|
|
148
|
+
Returns:
|
|
149
|
+
The task result
|
|
150
|
+
|
|
151
|
+
Raises:
|
|
152
|
+
TimeoutError: If timeout exceeded
|
|
153
|
+
KeyError: If handle not found
|
|
154
|
+
Exception: If task raised an exception
|
|
155
|
+
"""
|
|
156
|
+
with self._lock:
|
|
157
|
+
future = self._futures.get(handle.task_id)
|
|
158
|
+
if future is None:
|
|
159
|
+
raise KeyError(f"Task not found: {handle.task_id}")
|
|
160
|
+
|
|
161
|
+
effective_timeout = timeout or self.config.timeout_seconds
|
|
162
|
+
return future.result(timeout=effective_timeout)
|
|
163
|
+
|
|
164
|
+
def get_status(self, handle: TaskHandle) -> TaskStatus:
|
|
165
|
+
"""
|
|
166
|
+
Get current status of a task.
|
|
167
|
+
|
|
168
|
+
Args:
|
|
169
|
+
handle: The task handle
|
|
170
|
+
|
|
171
|
+
Returns:
|
|
172
|
+
Current TaskStatus
|
|
173
|
+
"""
|
|
174
|
+
with self._lock:
|
|
175
|
+
future = self._futures.get(handle.task_id)
|
|
176
|
+
if future is None:
|
|
177
|
+
return TaskStatus.FAILED # Not found
|
|
178
|
+
|
|
179
|
+
if future.cancelled():
|
|
180
|
+
return TaskStatus.CANCELLED
|
|
181
|
+
elif future.done():
|
|
182
|
+
try:
|
|
183
|
+
future.result(timeout=0)
|
|
184
|
+
return TaskStatus.COMPLETED
|
|
185
|
+
except Exception:
|
|
186
|
+
return TaskStatus.FAILED
|
|
187
|
+
elif future.running():
|
|
188
|
+
return TaskStatus.RUNNING
|
|
189
|
+
else:
|
|
190
|
+
return TaskStatus.PENDING
|
|
191
|
+
|
|
192
|
+
def cancel(self, handle: TaskHandle) -> bool:
|
|
193
|
+
"""
|
|
194
|
+
Attempt to cancel a task.
|
|
195
|
+
|
|
196
|
+
Args:
|
|
197
|
+
handle: The task handle
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
True if cancelled, False otherwise
|
|
201
|
+
"""
|
|
202
|
+
with self._lock:
|
|
203
|
+
future = self._futures.get(handle.task_id)
|
|
204
|
+
if future is None:
|
|
205
|
+
return False
|
|
206
|
+
|
|
207
|
+
cancelled = future.cancel()
|
|
208
|
+
if cancelled:
|
|
209
|
+
handle._status = TaskStatus.CANCELLED
|
|
210
|
+
return cancelled
|
|
211
|
+
|
|
212
|
+
def submit_batch(
|
|
213
|
+
self,
|
|
214
|
+
tasks: List[tuple],
|
|
215
|
+
) -> List[TaskHandle]:
|
|
216
|
+
"""
|
|
217
|
+
Submit multiple tasks to the thread pool.
|
|
218
|
+
|
|
219
|
+
Args:
|
|
220
|
+
tasks: List of (fn, args, kwargs, context) tuples
|
|
221
|
+
|
|
222
|
+
Returns:
|
|
223
|
+
List of TaskHandles
|
|
224
|
+
"""
|
|
225
|
+
handles = []
|
|
226
|
+
for task in tasks:
|
|
227
|
+
fn = task[0]
|
|
228
|
+
args = task[1] if len(task) > 1 else ()
|
|
229
|
+
kwargs = task[2] if len(task) > 2 else {}
|
|
230
|
+
context = task[3] if len(task) > 3 else None
|
|
231
|
+
handle = self.submit(fn, args, kwargs, context)
|
|
232
|
+
handles.append(handle)
|
|
233
|
+
return handles
|
|
234
|
+
|
|
235
|
+
def wait_all(
|
|
236
|
+
self,
|
|
237
|
+
handles: List[TaskHandle],
|
|
238
|
+
timeout: Optional[float] = None,
|
|
239
|
+
) -> List[Any]:
|
|
240
|
+
"""
|
|
241
|
+
Wait for all tasks and return results.
|
|
242
|
+
|
|
243
|
+
Args:
|
|
244
|
+
handles: List of task handles
|
|
245
|
+
timeout: Maximum total wait time
|
|
246
|
+
|
|
247
|
+
Returns:
|
|
248
|
+
List of results (in same order as handles)
|
|
249
|
+
"""
|
|
250
|
+
results = []
|
|
251
|
+
for handle in handles:
|
|
252
|
+
try:
|
|
253
|
+
result = self.get_result(handle, timeout=timeout)
|
|
254
|
+
results.append(result)
|
|
255
|
+
except Exception as e:
|
|
256
|
+
results.append(e)
|
|
257
|
+
return results
|
|
258
|
+
|
|
259
|
+
def pending_count(self) -> int:
|
|
260
|
+
"""Get count of pending/running tasks."""
|
|
261
|
+
with self._lock:
|
|
262
|
+
return sum(
|
|
263
|
+
1 for f in self._futures.values()
|
|
264
|
+
if not f.done()
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
def shutdown(self, wait: bool = True) -> None:
|
|
268
|
+
"""
|
|
269
|
+
Shutdown the thread pool.
|
|
270
|
+
|
|
271
|
+
Args:
|
|
272
|
+
wait: Whether to wait for pending tasks
|
|
273
|
+
"""
|
|
274
|
+
if self._executor:
|
|
275
|
+
self._executor.shutdown(wait=wait)
|
|
276
|
+
self._executor = None
|
|
277
|
+
|
|
278
|
+
with self._lock:
|
|
279
|
+
self._futures.clear()
|
|
280
|
+
self._handles.clear()
|
|
281
|
+
|
|
282
|
+
def __enter__(self) -> "ThreadPoolBackend":
|
|
283
|
+
return self
|
|
284
|
+
|
|
285
|
+
def __exit__(self, exc_type, exc_val, exc_tb) -> None:
|
|
286
|
+
self.shutdown(wait=True)
|
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Evaluation context for trace propagation.
|
|
3
|
+
|
|
4
|
+
This module provides EvalContext, which captures and propagates trace context
|
|
5
|
+
across thread and process boundaries. This is essential for:
|
|
6
|
+
- Background threads adding attributes to original spans
|
|
7
|
+
- Distributed workers maintaining trace continuity
|
|
8
|
+
- Async evaluations enriching parent spans
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Dict, Optional, Any
|
|
13
|
+
import uuid
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class EvalContext:
|
|
18
|
+
"""
|
|
19
|
+
Context passed to all evaluations.
|
|
20
|
+
|
|
21
|
+
Captures trace/span IDs for propagation across threads/processes.
|
|
22
|
+
Follows W3C Trace Context specification for interoperability.
|
|
23
|
+
|
|
24
|
+
Attributes:
|
|
25
|
+
trace_id: 32-character hex trace ID
|
|
26
|
+
span_id: 16-character hex span ID
|
|
27
|
+
parent_span_id: Optional parent span ID
|
|
28
|
+
baggage: Key-value pairs propagated with the trace
|
|
29
|
+
eval_run_id: Unique ID for this evaluation run
|
|
30
|
+
|
|
31
|
+
Example:
|
|
32
|
+
# Capture from current span
|
|
33
|
+
ctx = EvalContext.from_current_span()
|
|
34
|
+
|
|
35
|
+
# Serialize for propagation
|
|
36
|
+
headers = ctx.to_headers()
|
|
37
|
+
|
|
38
|
+
# Reconstruct in worker
|
|
39
|
+
ctx = EvalContext.from_headers(headers)
|
|
40
|
+
"""
|
|
41
|
+
trace_id: str
|
|
42
|
+
span_id: str
|
|
43
|
+
parent_span_id: Optional[str] = None
|
|
44
|
+
baggage: Dict[str, str] = field(default_factory=dict)
|
|
45
|
+
eval_run_id: str = field(default_factory=lambda: uuid.uuid4().hex[:16])
|
|
46
|
+
|
|
47
|
+
def __post_init__(self):
|
|
48
|
+
"""Validate context fields."""
|
|
49
|
+
if not self.trace_id:
|
|
50
|
+
self.trace_id = uuid.uuid4().hex
|
|
51
|
+
if not self.span_id:
|
|
52
|
+
self.span_id = uuid.uuid4().hex[:16]
|
|
53
|
+
|
|
54
|
+
@classmethod
|
|
55
|
+
def from_current_span(cls) -> "EvalContext":
|
|
56
|
+
"""
|
|
57
|
+
Capture context from current OTEL span.
|
|
58
|
+
|
|
59
|
+
If OTEL is not available or no span is active, creates a standalone context.
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
EvalContext captured from current span or newly created
|
|
63
|
+
"""
|
|
64
|
+
try:
|
|
65
|
+
from opentelemetry import trace
|
|
66
|
+
from opentelemetry import baggage as otel_baggage
|
|
67
|
+
|
|
68
|
+
span = trace.get_current_span()
|
|
69
|
+
ctx = span.get_span_context()
|
|
70
|
+
|
|
71
|
+
if ctx.is_valid:
|
|
72
|
+
return cls(
|
|
73
|
+
trace_id=format(ctx.trace_id, '032x'),
|
|
74
|
+
span_id=format(ctx.span_id, '016x'),
|
|
75
|
+
parent_span_id=None,
|
|
76
|
+
baggage=dict(otel_baggage.get_all()),
|
|
77
|
+
)
|
|
78
|
+
except ImportError:
|
|
79
|
+
pass
|
|
80
|
+
except Exception:
|
|
81
|
+
pass
|
|
82
|
+
|
|
83
|
+
# No OTEL or invalid context - create standalone
|
|
84
|
+
return cls(
|
|
85
|
+
trace_id=uuid.uuid4().hex,
|
|
86
|
+
span_id=uuid.uuid4().hex[:16],
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
@classmethod
|
|
90
|
+
def from_headers(cls, headers: Dict[str, str]) -> "EvalContext":
|
|
91
|
+
"""
|
|
92
|
+
Extract context from W3C Trace Context headers.
|
|
93
|
+
|
|
94
|
+
Parses traceparent and baggage headers according to the W3C spec:
|
|
95
|
+
https://www.w3.org/TR/trace-context/
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
headers: Dict containing traceparent and optionally baggage headers
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
EvalContext reconstructed from headers
|
|
102
|
+
"""
|
|
103
|
+
traceparent = headers.get("traceparent", "")
|
|
104
|
+
baggage_str = headers.get("baggage", "")
|
|
105
|
+
eval_run_id = headers.get("x-eval-run-id", uuid.uuid4().hex[:16])
|
|
106
|
+
|
|
107
|
+
# Parse traceparent: version-trace_id-span_id-flags
|
|
108
|
+
# Example: 00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01
|
|
109
|
+
trace_id = uuid.uuid4().hex
|
|
110
|
+
span_id = uuid.uuid4().hex[:16]
|
|
111
|
+
|
|
112
|
+
parts = traceparent.split("-")
|
|
113
|
+
if len(parts) >= 3:
|
|
114
|
+
# Validate version (should be "00")
|
|
115
|
+
if len(parts[1]) == 32:
|
|
116
|
+
trace_id = parts[1]
|
|
117
|
+
if len(parts[2]) == 16:
|
|
118
|
+
span_id = parts[2]
|
|
119
|
+
|
|
120
|
+
# Parse baggage: key1=value1,key2=value2
|
|
121
|
+
baggage = {}
|
|
122
|
+
if baggage_str:
|
|
123
|
+
for item in baggage_str.split(","):
|
|
124
|
+
item = item.strip()
|
|
125
|
+
if "=" in item:
|
|
126
|
+
k, v = item.split("=", 1)
|
|
127
|
+
baggage[k.strip()] = v.strip()
|
|
128
|
+
|
|
129
|
+
return cls(
|
|
130
|
+
trace_id=trace_id,
|
|
131
|
+
span_id=span_id,
|
|
132
|
+
baggage=baggage,
|
|
133
|
+
eval_run_id=eval_run_id,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
def to_headers(self) -> Dict[str, str]:
|
|
137
|
+
"""
|
|
138
|
+
Convert to W3C Trace Context headers for propagation.
|
|
139
|
+
|
|
140
|
+
Returns:
|
|
141
|
+
Dict with traceparent, baggage, and x-eval-run-id headers
|
|
142
|
+
"""
|
|
143
|
+
headers = {
|
|
144
|
+
"traceparent": f"00-{self.trace_id}-{self.span_id}-01",
|
|
145
|
+
"x-eval-run-id": self.eval_run_id,
|
|
146
|
+
}
|
|
147
|
+
if self.baggage:
|
|
148
|
+
headers["baggage"] = ",".join(f"{k}={v}" for k, v in self.baggage.items())
|
|
149
|
+
return headers
|
|
150
|
+
|
|
151
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
152
|
+
"""Serialize to dict for storage/transmission."""
|
|
153
|
+
return {
|
|
154
|
+
"trace_id": self.trace_id,
|
|
155
|
+
"span_id": self.span_id,
|
|
156
|
+
"parent_span_id": self.parent_span_id,
|
|
157
|
+
"baggage": self.baggage,
|
|
158
|
+
"eval_run_id": self.eval_run_id,
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
@classmethod
|
|
162
|
+
def from_dict(cls, data: Dict[str, Any]) -> "EvalContext":
|
|
163
|
+
"""Deserialize from dict."""
|
|
164
|
+
return cls(
|
|
165
|
+
trace_id=data.get("trace_id", uuid.uuid4().hex),
|
|
166
|
+
span_id=data.get("span_id", uuid.uuid4().hex[:16]),
|
|
167
|
+
parent_span_id=data.get("parent_span_id"),
|
|
168
|
+
baggage=data.get("baggage", {}),
|
|
169
|
+
eval_run_id=data.get("eval_run_id", uuid.uuid4().hex[:16]),
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
def with_baggage(self, key: str, value: str) -> "EvalContext":
|
|
173
|
+
"""
|
|
174
|
+
Create a new context with additional baggage.
|
|
175
|
+
|
|
176
|
+
Args:
|
|
177
|
+
key: Baggage key
|
|
178
|
+
value: Baggage value
|
|
179
|
+
|
|
180
|
+
Returns:
|
|
181
|
+
New EvalContext with the added baggage
|
|
182
|
+
"""
|
|
183
|
+
new_baggage = dict(self.baggage)
|
|
184
|
+
new_baggage[key] = value
|
|
185
|
+
return EvalContext(
|
|
186
|
+
trace_id=self.trace_id,
|
|
187
|
+
span_id=self.span_id,
|
|
188
|
+
parent_span_id=self.parent_span_id,
|
|
189
|
+
baggage=new_baggage,
|
|
190
|
+
eval_run_id=self.eval_run_id,
|
|
191
|
+
)
|
|
192
|
+
|
|
193
|
+
def child_context(self) -> "EvalContext":
|
|
194
|
+
"""
|
|
195
|
+
Create a child context with new span_id.
|
|
196
|
+
|
|
197
|
+
The current span becomes the parent span.
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
New EvalContext representing a child span
|
|
201
|
+
"""
|
|
202
|
+
return EvalContext(
|
|
203
|
+
trace_id=self.trace_id,
|
|
204
|
+
span_id=uuid.uuid4().hex[:16],
|
|
205
|
+
parent_span_id=self.span_id,
|
|
206
|
+
baggage=dict(self.baggage),
|
|
207
|
+
eval_run_id=self.eval_run_id,
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
@property
|
|
211
|
+
def is_valid(self) -> bool:
|
|
212
|
+
"""Check if context has valid trace and span IDs."""
|
|
213
|
+
return (
|
|
214
|
+
len(self.trace_id) == 32 and
|
|
215
|
+
len(self.span_id) == 16 and
|
|
216
|
+
self.trace_id != "0" * 32 and
|
|
217
|
+
self.span_id != "0" * 16
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
def __str__(self) -> str:
|
|
221
|
+
return f"EvalContext(trace={self.trace_id[:8]}..., span={self.span_id})"
|
|
222
|
+
|
|
223
|
+
def __repr__(self) -> str:
|
|
224
|
+
return (
|
|
225
|
+
f"EvalContext(trace_id='{self.trace_id}', span_id='{self.span_id}', "
|
|
226
|
+
f"eval_run_id='{self.eval_run_id}')"
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def get_current_context() -> EvalContext:
|
|
231
|
+
"""
|
|
232
|
+
Get EvalContext from current span.
|
|
233
|
+
|
|
234
|
+
Convenience function that wraps EvalContext.from_current_span().
|
|
235
|
+
|
|
236
|
+
Returns:
|
|
237
|
+
EvalContext captured from current span or newly created
|
|
238
|
+
"""
|
|
239
|
+
return EvalContext.from_current_span()
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def create_standalone_context(**baggage) -> EvalContext:
|
|
243
|
+
"""
|
|
244
|
+
Create a standalone context not linked to any span.
|
|
245
|
+
|
|
246
|
+
Useful for testing or when running outside of a traced context.
|
|
247
|
+
|
|
248
|
+
Args:
|
|
249
|
+
**baggage: Key-value pairs to include as baggage
|
|
250
|
+
|
|
251
|
+
Returns:
|
|
252
|
+
New standalone EvalContext
|
|
253
|
+
"""
|
|
254
|
+
return EvalContext(
|
|
255
|
+
trace_id=uuid.uuid4().hex,
|
|
256
|
+
span_id=uuid.uuid4().hex[:16],
|
|
257
|
+
baggage=baggage,
|
|
258
|
+
)
|