agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,371 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Automatic Span Enrichment.
|
|
3
|
+
|
|
4
|
+
Automatically adds evaluation results to active OTEL spans
|
|
5
|
+
when evaluations are run through fi.evals.
|
|
6
|
+
|
|
7
|
+
This enables the "evals data automatically goes into spans" workflow.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from typing import Any, Dict, Optional, Union
|
|
11
|
+
import logging
|
|
12
|
+
import time
|
|
13
|
+
|
|
14
|
+
from .processors import OTEL_AVAILABLE
|
|
15
|
+
from .conventions import EvaluationAttributes, GenAIAttributes
|
|
16
|
+
|
|
17
|
+
if OTEL_AVAILABLE:
|
|
18
|
+
from opentelemetry import trace
|
|
19
|
+
from opentelemetry.trace import Status, StatusCode
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
# Global flag to enable/disable automatic enrichment
|
|
24
|
+
_auto_enrichment_enabled = True
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def enable_auto_enrichment():
|
|
28
|
+
"""Enable automatic span enrichment for evaluations."""
|
|
29
|
+
global _auto_enrichment_enabled
|
|
30
|
+
_auto_enrichment_enabled = True
|
|
31
|
+
logger.info("Automatic span enrichment enabled")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def disable_auto_enrichment():
|
|
35
|
+
"""Disable automatic span enrichment for evaluations."""
|
|
36
|
+
global _auto_enrichment_enabled
|
|
37
|
+
_auto_enrichment_enabled = False
|
|
38
|
+
logger.info("Automatic span enrichment disabled")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def is_auto_enrichment_enabled() -> bool:
|
|
42
|
+
"""Check if automatic span enrichment is enabled."""
|
|
43
|
+
return _auto_enrichment_enabled
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def get_current_span() -> Optional[Any]:
|
|
47
|
+
"""
|
|
48
|
+
Get the current active span.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
Current span or None if no active span or OTEL not available
|
|
52
|
+
"""
|
|
53
|
+
if not OTEL_AVAILABLE:
|
|
54
|
+
return None
|
|
55
|
+
|
|
56
|
+
try:
|
|
57
|
+
span = trace.get_current_span()
|
|
58
|
+
# Check if it's a valid, recording span
|
|
59
|
+
if span and span.is_recording():
|
|
60
|
+
return span
|
|
61
|
+
return None
|
|
62
|
+
except Exception as e:
|
|
63
|
+
logger.debug(f"Failed to get current span: {e}")
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def enrich_span_with_evaluation(
|
|
68
|
+
metric_name: str,
|
|
69
|
+
score: Union[float, int, bool],
|
|
70
|
+
reason: Optional[str] = None,
|
|
71
|
+
latency_ms: Optional[float] = None,
|
|
72
|
+
span: Optional[Any] = None,
|
|
73
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
74
|
+
) -> bool:
|
|
75
|
+
"""
|
|
76
|
+
Enrich a span with evaluation results.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
metric_name: Name of the evaluation metric
|
|
80
|
+
score: Evaluation score (float 0-1, int, or bool)
|
|
81
|
+
reason: Explanation for the score
|
|
82
|
+
latency_ms: Time taken to evaluate
|
|
83
|
+
span: Specific span to enrich, or None for current span
|
|
84
|
+
metadata: Additional metadata to add
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
True if enrichment succeeded, False otherwise
|
|
88
|
+
"""
|
|
89
|
+
if not OTEL_AVAILABLE:
|
|
90
|
+
return False
|
|
91
|
+
|
|
92
|
+
if not _auto_enrichment_enabled:
|
|
93
|
+
return False
|
|
94
|
+
|
|
95
|
+
# Get span to enrich
|
|
96
|
+
target_span = span or get_current_span()
|
|
97
|
+
if target_span is None:
|
|
98
|
+
logger.debug(f"No active span to enrich with {metric_name} evaluation")
|
|
99
|
+
return False
|
|
100
|
+
|
|
101
|
+
try:
|
|
102
|
+
# Normalize score to float
|
|
103
|
+
if isinstance(score, bool):
|
|
104
|
+
normalized_score = 1.0 if score else 0.0
|
|
105
|
+
elif isinstance(score, (int, float)):
|
|
106
|
+
normalized_score = float(score)
|
|
107
|
+
elif isinstance(score, str):
|
|
108
|
+
try:
|
|
109
|
+
normalized_score = float(score)
|
|
110
|
+
except ValueError:
|
|
111
|
+
low = score.strip().lower()
|
|
112
|
+
if low in ("true", "yes", "pass", "passed"):
|
|
113
|
+
normalized_score = 1.0
|
|
114
|
+
elif low in ("false", "no", "fail", "failed"):
|
|
115
|
+
normalized_score = 0.0
|
|
116
|
+
else:
|
|
117
|
+
logger.debug(f"Cannot normalize score string '{score}' to float")
|
|
118
|
+
return False
|
|
119
|
+
else:
|
|
120
|
+
normalized_score = float(score)
|
|
121
|
+
|
|
122
|
+
target_span.set_attribute(EvaluationAttributes.NAME, metric_name)
|
|
123
|
+
target_span.set_attribute(EvaluationAttributes.SCORE_VALUE, normalized_score)
|
|
124
|
+
|
|
125
|
+
if reason:
|
|
126
|
+
reason_text = reason[:1000] if len(reason) > 1000 else reason
|
|
127
|
+
target_span.set_attribute(EvaluationAttributes.EXPLANATION, reason_text)
|
|
128
|
+
|
|
129
|
+
if latency_ms is not None:
|
|
130
|
+
target_span.set_attribute(
|
|
131
|
+
EvaluationAttributes.latency(metric_name),
|
|
132
|
+
latency_ms
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
logger.debug(f"Enriched span with {metric_name}={normalized_score}")
|
|
136
|
+
return True
|
|
137
|
+
|
|
138
|
+
except Exception as e:
|
|
139
|
+
logger.warning(f"Failed to enrich span with evaluation: {e}")
|
|
140
|
+
return False
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def enrich_span_with_eval_result(
|
|
144
|
+
eval_result: Any,
|
|
145
|
+
span: Optional[Any] = None,
|
|
146
|
+
) -> bool:
|
|
147
|
+
"""
|
|
148
|
+
Enrich a span with an EvalResult object.
|
|
149
|
+
|
|
150
|
+
Args:
|
|
151
|
+
eval_result: EvalResult from fi.evals
|
|
152
|
+
span: Specific span to enrich, or None for current span
|
|
153
|
+
|
|
154
|
+
Returns:
|
|
155
|
+
True if enrichment succeeded
|
|
156
|
+
"""
|
|
157
|
+
if eval_result is None:
|
|
158
|
+
return False
|
|
159
|
+
|
|
160
|
+
try:
|
|
161
|
+
metric_name = getattr(eval_result, 'name', 'unknown')
|
|
162
|
+
score = getattr(eval_result, 'output', None)
|
|
163
|
+
reason = getattr(eval_result, 'reason', None)
|
|
164
|
+
runtime = getattr(eval_result, 'runtime', None)
|
|
165
|
+
|
|
166
|
+
if score is None:
|
|
167
|
+
return False
|
|
168
|
+
|
|
169
|
+
return enrich_span_with_evaluation(
|
|
170
|
+
metric_name=metric_name,
|
|
171
|
+
score=score,
|
|
172
|
+
reason=reason,
|
|
173
|
+
latency_ms=float(runtime) if runtime else None,
|
|
174
|
+
span=span,
|
|
175
|
+
)
|
|
176
|
+
except Exception as e:
|
|
177
|
+
logger.warning(f"Failed to enrich span with EvalResult: {e}")
|
|
178
|
+
return False
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def enrich_span_with_batch_result(
|
|
182
|
+
batch_result: Any,
|
|
183
|
+
span: Optional[Any] = None,
|
|
184
|
+
) -> int:
|
|
185
|
+
"""
|
|
186
|
+
Enrich a span with a BatchRunResult.
|
|
187
|
+
|
|
188
|
+
Args:
|
|
189
|
+
batch_result: BatchRunResult from fi.evals
|
|
190
|
+
span: Specific span to enrich, or None for current span
|
|
191
|
+
|
|
192
|
+
Returns:
|
|
193
|
+
Number of evaluations successfully added
|
|
194
|
+
"""
|
|
195
|
+
if batch_result is None:
|
|
196
|
+
return 0
|
|
197
|
+
|
|
198
|
+
count = 0
|
|
199
|
+
try:
|
|
200
|
+
eval_results = getattr(batch_result, 'eval_results', [])
|
|
201
|
+
for result in eval_results:
|
|
202
|
+
if result is not None and enrich_span_with_eval_result(result, span):
|
|
203
|
+
count += 1
|
|
204
|
+
except Exception as e:
|
|
205
|
+
logger.warning(f"Failed to enrich span with BatchRunResult: {e}")
|
|
206
|
+
|
|
207
|
+
return count
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def create_evaluation_span(
|
|
211
|
+
metric_name: str,
|
|
212
|
+
parent_span: Optional[Any] = None,
|
|
213
|
+
) -> Any:
|
|
214
|
+
"""
|
|
215
|
+
Create a child span for an evaluation operation.
|
|
216
|
+
|
|
217
|
+
Args:
|
|
218
|
+
metric_name: Name of the metric being evaluated
|
|
219
|
+
parent_span: Parent span, or None for current span
|
|
220
|
+
|
|
221
|
+
Returns:
|
|
222
|
+
Context manager for the evaluation span
|
|
223
|
+
"""
|
|
224
|
+
if not OTEL_AVAILABLE:
|
|
225
|
+
from .tracer import _NoOpContextManager
|
|
226
|
+
return _NoOpContextManager()
|
|
227
|
+
|
|
228
|
+
try:
|
|
229
|
+
tracer = trace.get_tracer("fi.evals.evaluation")
|
|
230
|
+
return tracer.start_as_current_span(
|
|
231
|
+
f"eval.{metric_name}",
|
|
232
|
+
attributes={
|
|
233
|
+
GenAIAttributes.SPAN_KIND: "EVALUATOR",
|
|
234
|
+
EvaluationAttributes.NAME: metric_name,
|
|
235
|
+
EvaluationAttributes.EVALUATED_AT: time.time(),
|
|
236
|
+
}
|
|
237
|
+
)
|
|
238
|
+
except Exception as e:
|
|
239
|
+
logger.debug(f"Failed to create evaluation span: {e}")
|
|
240
|
+
from .tracer import _NoOpContextManager
|
|
241
|
+
return _NoOpContextManager()
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
class EvaluationSpanContext:
|
|
245
|
+
"""
|
|
246
|
+
Context manager for evaluation operations with automatic span enrichment.
|
|
247
|
+
|
|
248
|
+
Example:
|
|
249
|
+
with EvaluationSpanContext("relevance") as ctx:
|
|
250
|
+
result = run_evaluation()
|
|
251
|
+
ctx.record_result(score=0.85, reason="Good match")
|
|
252
|
+
"""
|
|
253
|
+
|
|
254
|
+
def __init__(self, metric_name: str, create_child_span: bool = True):
|
|
255
|
+
"""
|
|
256
|
+
Initialize evaluation context.
|
|
257
|
+
|
|
258
|
+
Args:
|
|
259
|
+
metric_name: Name of the metric
|
|
260
|
+
create_child_span: Whether to create a child span
|
|
261
|
+
"""
|
|
262
|
+
self.metric_name = metric_name
|
|
263
|
+
self.create_child_span = create_child_span
|
|
264
|
+
self._span = None
|
|
265
|
+
self._parent_span = None
|
|
266
|
+
self._start_time = None
|
|
267
|
+
self._result_recorded = False
|
|
268
|
+
|
|
269
|
+
def __enter__(self):
|
|
270
|
+
self._start_time = time.time()
|
|
271
|
+
|
|
272
|
+
if OTEL_AVAILABLE and self.create_child_span:
|
|
273
|
+
try:
|
|
274
|
+
self._parent_span = get_current_span()
|
|
275
|
+
tracer = trace.get_tracer("fi.evals.evaluation")
|
|
276
|
+
self._span = tracer.start_span(
|
|
277
|
+
f"eval.{self.metric_name}",
|
|
278
|
+
attributes={
|
|
279
|
+
GenAIAttributes.SPAN_KIND: "EVALUATOR",
|
|
280
|
+
EvaluationAttributes.NAME: self.metric_name,
|
|
281
|
+
}
|
|
282
|
+
)
|
|
283
|
+
# Make it the current span
|
|
284
|
+
self._token = trace.use_span(self._span, end_on_exit=False)
|
|
285
|
+
self._token.__enter__()
|
|
286
|
+
except Exception as e:
|
|
287
|
+
logger.debug(f"Failed to create evaluation span: {e}")
|
|
288
|
+
|
|
289
|
+
return self
|
|
290
|
+
|
|
291
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
292
|
+
latency_ms = (time.time() - self._start_time) * 1000
|
|
293
|
+
|
|
294
|
+
if self._span:
|
|
295
|
+
try:
|
|
296
|
+
if exc_type:
|
|
297
|
+
self._span.set_status(Status(StatusCode.ERROR, str(exc_val)))
|
|
298
|
+
self._span.record_exception(exc_val)
|
|
299
|
+
else:
|
|
300
|
+
self._span.set_status(Status(StatusCode.OK))
|
|
301
|
+
|
|
302
|
+
self._span.set_attribute(
|
|
303
|
+
EvaluationAttributes.latency(self.metric_name),
|
|
304
|
+
latency_ms
|
|
305
|
+
)
|
|
306
|
+
self._token.__exit__(exc_type, exc_val, exc_tb)
|
|
307
|
+
self._span.end()
|
|
308
|
+
except Exception as e:
|
|
309
|
+
logger.debug(f"Error closing evaluation span: {e}")
|
|
310
|
+
|
|
311
|
+
return False
|
|
312
|
+
|
|
313
|
+
def record_result(
|
|
314
|
+
self,
|
|
315
|
+
score: Union[float, int, bool],
|
|
316
|
+
reason: Optional[str] = None,
|
|
317
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
318
|
+
) -> bool:
|
|
319
|
+
"""
|
|
320
|
+
Record the evaluation result.
|
|
321
|
+
|
|
322
|
+
Args:
|
|
323
|
+
score: Evaluation score
|
|
324
|
+
reason: Explanation
|
|
325
|
+
metadata: Additional metadata
|
|
326
|
+
|
|
327
|
+
Returns:
|
|
328
|
+
True if recorded successfully
|
|
329
|
+
"""
|
|
330
|
+
if self._result_recorded:
|
|
331
|
+
return False
|
|
332
|
+
|
|
333
|
+
latency_ms = (time.time() - self._start_time) * 1000 if self._start_time else None
|
|
334
|
+
|
|
335
|
+
# Enrich the child span if created
|
|
336
|
+
if self._span:
|
|
337
|
+
enrich_span_with_evaluation(
|
|
338
|
+
metric_name=self.metric_name,
|
|
339
|
+
score=score,
|
|
340
|
+
reason=reason,
|
|
341
|
+
latency_ms=latency_ms,
|
|
342
|
+
span=self._span,
|
|
343
|
+
metadata=metadata,
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
# Also enrich the parent span
|
|
347
|
+
if self._parent_span:
|
|
348
|
+
enrich_span_with_evaluation(
|
|
349
|
+
metric_name=self.metric_name,
|
|
350
|
+
score=score,
|
|
351
|
+
reason=reason,
|
|
352
|
+
latency_ms=latency_ms,
|
|
353
|
+
span=self._parent_span,
|
|
354
|
+
metadata=metadata,
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
self._result_recorded = True
|
|
358
|
+
return True
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
__all__ = [
|
|
362
|
+
"enable_auto_enrichment",
|
|
363
|
+
"disable_auto_enrichment",
|
|
364
|
+
"is_auto_enrichment_enabled",
|
|
365
|
+
"get_current_span",
|
|
366
|
+
"enrich_span_with_evaluation",
|
|
367
|
+
"enrich_span_with_eval_result",
|
|
368
|
+
"enrich_span_with_batch_result",
|
|
369
|
+
"create_evaluation_span",
|
|
370
|
+
"EvaluationSpanContext",
|
|
371
|
+
]
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""
|
|
2
|
+
LLM Client Instrumentors.
|
|
3
|
+
|
|
4
|
+
Automatic instrumentation for popular LLM client libraries.
|
|
5
|
+
Each instrumentor patches the library to automatically create
|
|
6
|
+
spans with standardized attributes.
|
|
7
|
+
|
|
8
|
+
Example:
|
|
9
|
+
# Instrument individual libraries
|
|
10
|
+
from fi.evals.otel.instrumentors import OpenAIInstrumentor, AnthropicInstrumentor
|
|
11
|
+
|
|
12
|
+
OpenAIInstrumentor().instrument()
|
|
13
|
+
AnthropicInstrumentor().instrument()
|
|
14
|
+
|
|
15
|
+
# Or use the convenience function to instrument all available
|
|
16
|
+
from fi.evals.otel.instrumentors import instrument_all
|
|
17
|
+
|
|
18
|
+
instrumented = instrument_all()
|
|
19
|
+
print(f"Instrumented: {instrumented}")
|
|
20
|
+
|
|
21
|
+
# Clean up
|
|
22
|
+
from fi.evals.otel.instrumentors import uninstrument_all
|
|
23
|
+
uninstrument_all()
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from typing import List, Optional
|
|
27
|
+
|
|
28
|
+
from .base import BaseInstrumentor, InstrumentorManager
|
|
29
|
+
from .openai import OpenAIInstrumentor
|
|
30
|
+
from .anthropic import AnthropicInstrumentor
|
|
31
|
+
|
|
32
|
+
# Global manager instance
|
|
33
|
+
_manager: Optional[InstrumentorManager] = None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def get_manager() -> InstrumentorManager:
|
|
37
|
+
"""Get the global instrumentor manager."""
|
|
38
|
+
global _manager
|
|
39
|
+
if _manager is None:
|
|
40
|
+
_manager = InstrumentorManager()
|
|
41
|
+
# Register available instrumentors
|
|
42
|
+
_manager.add(OpenAIInstrumentor())
|
|
43
|
+
_manager.add(AnthropicInstrumentor())
|
|
44
|
+
return _manager
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def instrument_all(**kwargs) -> List[str]:
|
|
48
|
+
"""
|
|
49
|
+
Instrument all available LLM libraries.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
List of library names that were instrumented
|
|
53
|
+
|
|
54
|
+
Example:
|
|
55
|
+
instrumented = instrument_all()
|
|
56
|
+
# ['openai', 'anthropic']
|
|
57
|
+
"""
|
|
58
|
+
return get_manager().instrument_all(**kwargs)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def uninstrument_all(**kwargs) -> List[str]:
|
|
62
|
+
"""
|
|
63
|
+
Remove instrumentation from all libraries.
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
List of library names that were uninstrumented
|
|
67
|
+
"""
|
|
68
|
+
return get_manager().uninstrument_all(**kwargs)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def instrument(library: str, **kwargs) -> bool:
|
|
72
|
+
"""
|
|
73
|
+
Instrument a specific library.
|
|
74
|
+
|
|
75
|
+
Args:
|
|
76
|
+
library: Library name ('openai', 'anthropic', etc.)
|
|
77
|
+
**kwargs: Options passed to the instrumentor
|
|
78
|
+
|
|
79
|
+
Returns:
|
|
80
|
+
True if instrumented successfully
|
|
81
|
+
"""
|
|
82
|
+
instrumentor = get_manager().get(library)
|
|
83
|
+
if instrumentor is None:
|
|
84
|
+
return False
|
|
85
|
+
try:
|
|
86
|
+
instrumentor.instrument(**kwargs)
|
|
87
|
+
return instrumentor.is_instrumented
|
|
88
|
+
except Exception:
|
|
89
|
+
return False
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def uninstrument(library: str, **kwargs) -> bool:
|
|
93
|
+
"""
|
|
94
|
+
Remove instrumentation from a specific library.
|
|
95
|
+
|
|
96
|
+
Args:
|
|
97
|
+
library: Library name
|
|
98
|
+
|
|
99
|
+
Returns:
|
|
100
|
+
True if uninstrumented successfully
|
|
101
|
+
"""
|
|
102
|
+
instrumentor = get_manager().get(library)
|
|
103
|
+
if instrumentor is None:
|
|
104
|
+
return False
|
|
105
|
+
try:
|
|
106
|
+
instrumentor.uninstrument(**kwargs)
|
|
107
|
+
return not instrumentor.is_instrumented
|
|
108
|
+
except Exception:
|
|
109
|
+
return False
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def is_instrumented(library: str) -> bool:
|
|
113
|
+
"""Check if a library is currently instrumented."""
|
|
114
|
+
instrumentor = get_manager().get(library)
|
|
115
|
+
return instrumentor.is_instrumented if instrumentor else False
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def get_instrumented_libraries() -> List[str]:
|
|
119
|
+
"""Get list of currently instrumented libraries."""
|
|
120
|
+
return list(get_manager().instrumented_libraries)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
__all__ = [
|
|
124
|
+
# Base classes
|
|
125
|
+
"BaseInstrumentor",
|
|
126
|
+
"InstrumentorManager",
|
|
127
|
+
|
|
128
|
+
# Specific instrumentors
|
|
129
|
+
"OpenAIInstrumentor",
|
|
130
|
+
"AnthropicInstrumentor",
|
|
131
|
+
|
|
132
|
+
# Convenience functions
|
|
133
|
+
"instrument_all",
|
|
134
|
+
"uninstrument_all",
|
|
135
|
+
"instrument",
|
|
136
|
+
"uninstrument",
|
|
137
|
+
"is_instrumented",
|
|
138
|
+
"get_instrumented_libraries",
|
|
139
|
+
"get_manager",
|
|
140
|
+
]
|