agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/evals/execution.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Execution handles for async eval and composite runs.
|
|
3
|
+
|
|
4
|
+
An ``Execution`` is a lightweight view into a (possibly still-running) eval
|
|
5
|
+
on the backend (for single evals) or a background thread (for composite
|
|
6
|
+
evals, which the backend runs synchronously).
|
|
7
|
+
|
|
8
|
+
Typical usage::
|
|
9
|
+
|
|
10
|
+
from fi.evals import Evaluator, EvalTemplateManager
|
|
11
|
+
|
|
12
|
+
ev = Evaluator()
|
|
13
|
+
mgr = EvalTemplateManager()
|
|
14
|
+
|
|
15
|
+
# --- Single eval: real backend async via is_async=True ---
|
|
16
|
+
handle = ev.submit("tone", {"output": "I love this!"})
|
|
17
|
+
handle.wait() # polls until completion
|
|
18
|
+
print(handle.result.output) # -> "love"
|
|
19
|
+
|
|
20
|
+
# --- Composite eval: SDK-side threaded execution ---
|
|
21
|
+
handle = mgr.submit_composite(composite_id, mapping={"output": "Hi!"})
|
|
22
|
+
handle.wait()
|
|
23
|
+
print(handle.result["aggregate_score"])
|
|
24
|
+
|
|
25
|
+
# --- Resumable by ID (single eval only) ---
|
|
26
|
+
other_handle = ev.get_execution(handle.id)
|
|
27
|
+
other_handle.wait()
|
|
28
|
+
|
|
29
|
+
Note on composite executions: the handle lives in a background thread
|
|
30
|
+
inside the calling process. If the process dies or the handle is dropped,
|
|
31
|
+
the in-flight work is lost. Use single-eval async submissions when you
|
|
32
|
+
need cross-process resumability.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import time
|
|
38
|
+
from dataclasses import dataclass, field
|
|
39
|
+
from typing import Any, Callable, Dict, Optional
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ExecutionError(Exception):
|
|
43
|
+
"""Raised when an Execution that finished in the ``failed`` state is awaited."""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# Backend eval_status → SDK-normalized status.
|
|
47
|
+
_STATUS_MAP = {
|
|
48
|
+
"pending": "pending",
|
|
49
|
+
"PENDING": "pending",
|
|
50
|
+
"processing": "processing",
|
|
51
|
+
"PROCESSING": "processing",
|
|
52
|
+
"running": "processing",
|
|
53
|
+
"completed": "completed",
|
|
54
|
+
"COMPLETED": "completed",
|
|
55
|
+
"failed": "failed",
|
|
56
|
+
"FAILED": "failed",
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _normalize_status(raw: Optional[str]) -> str:
|
|
61
|
+
if not raw:
|
|
62
|
+
return "pending"
|
|
63
|
+
return _STATUS_MAP.get(raw, raw.lower())
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass
|
|
67
|
+
class Execution:
|
|
68
|
+
"""
|
|
69
|
+
Handle to an in-flight or completed eval execution.
|
|
70
|
+
|
|
71
|
+
Attributes:
|
|
72
|
+
id: Execution identifier. For single evals this is the server-side
|
|
73
|
+
``eval_id`` (a UUID) and is resumable from any process via
|
|
74
|
+
:py:meth:`fi.evals.Evaluator.get_execution`. For composite
|
|
75
|
+
evals this is a client-side UUID — see the module docstring
|
|
76
|
+
for the caveat.
|
|
77
|
+
kind: ``"eval"`` for a single eval execution, ``"composite"`` for
|
|
78
|
+
a composite one.
|
|
79
|
+
status: ``"pending"`` | ``"processing"`` | ``"completed"`` |
|
|
80
|
+
``"failed"``.
|
|
81
|
+
result: Populated once ``status == "completed"``. An ``EvalResult``
|
|
82
|
+
for single evals; a dict matching the composite execute
|
|
83
|
+
response for composite ones.
|
|
84
|
+
error_message: Populated when the execution failed.
|
|
85
|
+
error_localizer: Populated for single evals when error
|
|
86
|
+
localization was enabled and the analysis is available.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
id: str
|
|
90
|
+
kind: str
|
|
91
|
+
status: str = "pending"
|
|
92
|
+
result: Any = None
|
|
93
|
+
error_message: Optional[str] = None
|
|
94
|
+
error_localizer: Optional[Dict[str, Any]] = None
|
|
95
|
+
|
|
96
|
+
# Closure that (re)fetches the latest state. Set by the factory
|
|
97
|
+
# method on Evaluator / EvalTemplateManager. Excluded from repr.
|
|
98
|
+
_refresher: Optional[Callable[[], "Execution"]] = field(
|
|
99
|
+
default=None, repr=False, compare=False
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
def is_done(self) -> bool:
|
|
103
|
+
"""Return True if status is terminal (``completed`` or ``failed``)."""
|
|
104
|
+
return self.status in ("completed", "failed")
|
|
105
|
+
|
|
106
|
+
def refresh(self) -> "Execution":
|
|
107
|
+
"""
|
|
108
|
+
Re-fetch the latest state from the source (backend for single
|
|
109
|
+
evals, background thread for composites). Returns ``self``.
|
|
110
|
+
"""
|
|
111
|
+
if self._refresher is None:
|
|
112
|
+
return self
|
|
113
|
+
updated = self._refresher()
|
|
114
|
+
self.status = updated.status
|
|
115
|
+
self.result = updated.result
|
|
116
|
+
self.error_message = updated.error_message
|
|
117
|
+
self.error_localizer = updated.error_localizer
|
|
118
|
+
return self
|
|
119
|
+
|
|
120
|
+
def wait(
|
|
121
|
+
self,
|
|
122
|
+
*,
|
|
123
|
+
timeout: float = 300.0,
|
|
124
|
+
poll_interval: float = 2.0,
|
|
125
|
+
raise_on_failure: bool = True,
|
|
126
|
+
) -> "Execution":
|
|
127
|
+
"""
|
|
128
|
+
Block until the execution reaches a terminal state, refreshing
|
|
129
|
+
every ``poll_interval`` seconds.
|
|
130
|
+
|
|
131
|
+
Args:
|
|
132
|
+
timeout: Maximum number of seconds to wait before giving up.
|
|
133
|
+
poll_interval: Seconds between refreshes.
|
|
134
|
+
raise_on_failure: If True (default) raise
|
|
135
|
+
:class:`ExecutionError` when the run finished in
|
|
136
|
+
``"failed"`` state. If False, return the handle so the
|
|
137
|
+
caller can inspect ``error_message`` themselves.
|
|
138
|
+
|
|
139
|
+
Raises:
|
|
140
|
+
TimeoutError: If the execution did not reach a terminal
|
|
141
|
+
state within ``timeout`` seconds.
|
|
142
|
+
ExecutionError: If the execution failed and
|
|
143
|
+
``raise_on_failure=True``.
|
|
144
|
+
"""
|
|
145
|
+
if self.is_done():
|
|
146
|
+
if self.status == "failed" and raise_on_failure:
|
|
147
|
+
raise ExecutionError(
|
|
148
|
+
f"Execution {self.id} failed: {self.error_message}"
|
|
149
|
+
)
|
|
150
|
+
return self
|
|
151
|
+
|
|
152
|
+
deadline = time.monotonic() + float(timeout)
|
|
153
|
+
while True:
|
|
154
|
+
time.sleep(poll_interval)
|
|
155
|
+
self.refresh()
|
|
156
|
+
if self.is_done():
|
|
157
|
+
break
|
|
158
|
+
if time.monotonic() > deadline:
|
|
159
|
+
raise TimeoutError(
|
|
160
|
+
f"Execution {self.id} did not complete within {timeout}s "
|
|
161
|
+
f"(last status: {self.status})"
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
if self.status == "failed" and raise_on_failure:
|
|
165
|
+
raise ExecutionError(
|
|
166
|
+
f"Execution {self.id} failed: {self.error_message}"
|
|
167
|
+
)
|
|
168
|
+
return self
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Feedback Loop system for improving evaluations over time.
|
|
2
|
+
|
|
3
|
+
Store developer feedback on metric results, retrieve similar past feedback
|
|
4
|
+
as few-shot examples for LLM judges, and calibrate thresholds statistically.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .types import FeedbackEntry, CalibrationProfile, FeedbackStats
|
|
8
|
+
from .store import FeedbackStore, InMemoryFeedbackStore
|
|
9
|
+
from .collector import FeedbackCollector
|
|
10
|
+
from .retriever import FeedbackRetriever
|
|
11
|
+
from .calibrator import ThresholdCalibrator
|
|
12
|
+
from .hooks import configure_feedback, get_default_store
|
|
13
|
+
|
|
14
|
+
# ChromaFeedbackStore requires chromadb — import conditionally
|
|
15
|
+
try:
|
|
16
|
+
from .store import ChromaFeedbackStore
|
|
17
|
+
except Exception:
|
|
18
|
+
pass
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"FeedbackEntry",
|
|
22
|
+
"CalibrationProfile",
|
|
23
|
+
"FeedbackStats",
|
|
24
|
+
"FeedbackStore",
|
|
25
|
+
"InMemoryFeedbackStore",
|
|
26
|
+
"ChromaFeedbackStore",
|
|
27
|
+
"FeedbackCollector",
|
|
28
|
+
"FeedbackRetriever",
|
|
29
|
+
"ThresholdCalibrator",
|
|
30
|
+
"configure_feedback",
|
|
31
|
+
"get_default_store",
|
|
32
|
+
]
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Statistical threshold calibration based on feedback.
|
|
2
|
+
|
|
3
|
+
Optimizes pass/fail thresholds by computing confusion matrices against
|
|
4
|
+
developer-provided correct labels across a range of threshold values.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
import math
|
|
9
|
+
from typing import List, Tuple
|
|
10
|
+
|
|
11
|
+
from .store import FeedbackStore
|
|
12
|
+
from .types import CalibrationProfile, FeedbackEntry
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ThresholdCalibrator:
|
|
18
|
+
"""Optimizes pass/fail thresholds based on accumulated feedback.
|
|
19
|
+
|
|
20
|
+
For each candidate threshold, computes a confusion matrix against
|
|
21
|
+
the developer's correct_passed labels, and selects the threshold
|
|
22
|
+
that maximizes agreement (accuracy) or F1 score.
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
store: FeedbackStore containing feedback entries.
|
|
26
|
+
optimize_for: Metric to optimize. "accuracy" (default) or "f1".
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self,
|
|
31
|
+
store: FeedbackStore,
|
|
32
|
+
optimize_for: str = "accuracy",
|
|
33
|
+
):
|
|
34
|
+
self.store = store
|
|
35
|
+
self.optimize_for = optimize_for
|
|
36
|
+
|
|
37
|
+
def calibrate(
|
|
38
|
+
self,
|
|
39
|
+
metric_name: str,
|
|
40
|
+
threshold_range: Tuple[float, float] = (0.3, 0.9),
|
|
41
|
+
steps: int = 13,
|
|
42
|
+
) -> CalibrationProfile:
|
|
43
|
+
"""Find the optimal pass/fail threshold for a metric.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
metric_name: The metric to calibrate.
|
|
47
|
+
threshold_range: (min_threshold, max_threshold) to search.
|
|
48
|
+
steps: Number of evenly-spaced thresholds to try.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
CalibrationProfile with the optimal threshold and stats.
|
|
52
|
+
|
|
53
|
+
Raises:
|
|
54
|
+
ValueError: If insufficient feedback (< 5 entries with corrections).
|
|
55
|
+
"""
|
|
56
|
+
entries = self.store.get_by_metric(metric_name)
|
|
57
|
+
|
|
58
|
+
# Filter to entries with both a correct label and a score
|
|
59
|
+
usable = [
|
|
60
|
+
e for e in entries
|
|
61
|
+
if e.correct_score is not None
|
|
62
|
+
and e.original_score is not None
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
if len(usable) < 5:
|
|
66
|
+
raise ValueError(
|
|
67
|
+
f"Need at least 5 feedback entries with corrections to calibrate "
|
|
68
|
+
f"'{metric_name}', but only {len(usable)} found. Submit more feedback first."
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
# Derive correct_passed if not explicitly set
|
|
72
|
+
for e in usable:
|
|
73
|
+
if e.correct_passed is None:
|
|
74
|
+
e.correct_passed = e.correct_score >= 0.5
|
|
75
|
+
|
|
76
|
+
# Search over threshold space
|
|
77
|
+
min_t, max_t = threshold_range
|
|
78
|
+
best_score = -1.0
|
|
79
|
+
best_threshold = 0.5
|
|
80
|
+
best_matrix = (0, 0, 0, 0)
|
|
81
|
+
|
|
82
|
+
for i in range(steps):
|
|
83
|
+
t = min_t + (max_t - min_t) * i / (steps - 1) if steps > 1 else (min_t + max_t) / 2
|
|
84
|
+
tp, fp, tn, fn = self._confusion_matrix(usable, t)
|
|
85
|
+
|
|
86
|
+
if self.optimize_for == "f1":
|
|
87
|
+
score = self._f1(tp, fp, fn)
|
|
88
|
+
else:
|
|
89
|
+
total = tp + fp + tn + fn
|
|
90
|
+
score = (tp + tn) / total if total > 0 else 0.0
|
|
91
|
+
|
|
92
|
+
if score > best_score:
|
|
93
|
+
best_score = score
|
|
94
|
+
best_threshold = t
|
|
95
|
+
best_matrix = (tp, fp, tn, fn)
|
|
96
|
+
|
|
97
|
+
# Compute score statistics
|
|
98
|
+
scores = [e.correct_score for e in usable if e.correct_score is not None]
|
|
99
|
+
mean = sum(scores) / len(scores) if scores else 0.0
|
|
100
|
+
variance = sum((s - mean) ** 2 for s in scores) / len(scores) if scores else 0.0
|
|
101
|
+
|
|
102
|
+
tp, fp, tn, fn = best_matrix
|
|
103
|
+
|
|
104
|
+
profile = CalibrationProfile(
|
|
105
|
+
eval_name=metric_name,
|
|
106
|
+
optimal_threshold=round(best_threshold, 3),
|
|
107
|
+
sample_size=len(usable),
|
|
108
|
+
accuracy_at_threshold=best_score,
|
|
109
|
+
score_mean=round(mean, 4),
|
|
110
|
+
score_std=round(math.sqrt(variance), 4),
|
|
111
|
+
true_positives=tp,
|
|
112
|
+
false_positives=fp,
|
|
113
|
+
true_negatives=tn,
|
|
114
|
+
false_negatives=fn,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
logger.info(
|
|
118
|
+
f"Calibrated '{metric_name}': threshold={profile.optimal_threshold} "
|
|
119
|
+
f"accuracy={profile.accuracy_at_threshold:.1%} (n={profile.sample_size})"
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
return profile
|
|
123
|
+
|
|
124
|
+
@staticmethod
|
|
125
|
+
def _confusion_matrix(
|
|
126
|
+
entries: List[FeedbackEntry],
|
|
127
|
+
threshold: float,
|
|
128
|
+
) -> Tuple[int, int, int, int]:
|
|
129
|
+
"""Compute confusion matrix at a given threshold.
|
|
130
|
+
|
|
131
|
+
Predicted positive = original_score >= threshold
|
|
132
|
+
Actual positive = correct_passed is True
|
|
133
|
+
|
|
134
|
+
Returns:
|
|
135
|
+
(true_positives, false_positives, true_negatives, false_negatives)
|
|
136
|
+
"""
|
|
137
|
+
tp = fp = tn = fn = 0
|
|
138
|
+
for e in entries:
|
|
139
|
+
predicted_pass = e.original_score >= threshold
|
|
140
|
+
actual_pass = e.correct_passed
|
|
141
|
+
|
|
142
|
+
if predicted_pass and actual_pass:
|
|
143
|
+
tp += 1
|
|
144
|
+
elif predicted_pass and not actual_pass:
|
|
145
|
+
fp += 1
|
|
146
|
+
elif not predicted_pass and not actual_pass:
|
|
147
|
+
tn += 1
|
|
148
|
+
else:
|
|
149
|
+
fn += 1
|
|
150
|
+
|
|
151
|
+
return tp, fp, tn, fn
|
|
152
|
+
|
|
153
|
+
@staticmethod
|
|
154
|
+
def _f1(tp: int, fp: int, fn: int) -> float:
|
|
155
|
+
"""Compute F1 score from confusion matrix components."""
|
|
156
|
+
precision = tp / (tp + fp) if (tp + fp) > 0 else 0.0
|
|
157
|
+
recall = tp / (tp + fn) if (tp + fn) > 0 else 0.0
|
|
158
|
+
if precision + recall == 0:
|
|
159
|
+
return 0.0
|
|
160
|
+
return 2 * precision * recall / (precision + recall)
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""User-facing API for the feedback loop system.
|
|
2
|
+
|
|
3
|
+
Provides a clean interface for submitting feedback, querying statistics,
|
|
4
|
+
calibrating thresholds, and creating retrievers.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
from typing import Any, Dict, List, Optional
|
|
9
|
+
|
|
10
|
+
from ..core.result import EvalResult
|
|
11
|
+
from .store import FeedbackStore
|
|
12
|
+
from .types import FeedbackEntry, FeedbackStats
|
|
13
|
+
from .retriever import FeedbackRetriever
|
|
14
|
+
from .calibrator import ThresholdCalibrator
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class FeedbackCollector:
|
|
20
|
+
"""Main user-facing class for the feedback loop system.
|
|
21
|
+
|
|
22
|
+
Provides a clean API for:
|
|
23
|
+
- Submitting feedback on metric results
|
|
24
|
+
- Retrieving statistics on accumulated feedback
|
|
25
|
+
- Calibrating thresholds based on feedback
|
|
26
|
+
- Creating a retriever for pipeline integration
|
|
27
|
+
|
|
28
|
+
Usage:
|
|
29
|
+
from fi.evals.feedback import FeedbackCollector, InMemoryFeedbackStore
|
|
30
|
+
|
|
31
|
+
store = InMemoryFeedbackStore() # or ChromaFeedbackStore()
|
|
32
|
+
feedback = FeedbackCollector(store)
|
|
33
|
+
|
|
34
|
+
# After running a metric that gave wrong results:
|
|
35
|
+
result = run_metric("faithfulness", output="...", context="...")
|
|
36
|
+
|
|
37
|
+
# Submit correction
|
|
38
|
+
feedback.submit(
|
|
39
|
+
result,
|
|
40
|
+
inputs={"output": "...", "context": "..."},
|
|
41
|
+
correct_score=0.9,
|
|
42
|
+
correct_reason="The response IS faithful via semantic equivalence.",
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
# Later, get a retriever for pipeline integration
|
|
46
|
+
retriever = feedback.get_retriever()
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
store: The FeedbackStore backend.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(self, store: FeedbackStore):
|
|
53
|
+
self.store = store
|
|
54
|
+
|
|
55
|
+
def submit(
|
|
56
|
+
self,
|
|
57
|
+
result: EvalResult,
|
|
58
|
+
*,
|
|
59
|
+
inputs: Dict[str, Any],
|
|
60
|
+
correct_score: Optional[float] = None,
|
|
61
|
+
correct_passed: Optional[bool] = None,
|
|
62
|
+
correct_reason: str = "",
|
|
63
|
+
tags: Optional[List[str]] = None,
|
|
64
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
65
|
+
) -> FeedbackEntry:
|
|
66
|
+
"""Submit feedback on a metric result.
|
|
67
|
+
|
|
68
|
+
Args:
|
|
69
|
+
result: The EvalResult that needs correction.
|
|
70
|
+
inputs: The original inputs that were used.
|
|
71
|
+
correct_score: What the score SHOULD have been (0.0-1.0).
|
|
72
|
+
correct_passed: What the pass/fail SHOULD have been.
|
|
73
|
+
correct_reason: Why the original result was wrong.
|
|
74
|
+
tags: Optional tags for organizing feedback.
|
|
75
|
+
metadata: Optional metadata dict.
|
|
76
|
+
|
|
77
|
+
Returns:
|
|
78
|
+
The stored FeedbackEntry.
|
|
79
|
+
|
|
80
|
+
Raises:
|
|
81
|
+
ValueError: If neither correct_score nor correct_reason is provided.
|
|
82
|
+
"""
|
|
83
|
+
if correct_score is None and not correct_reason:
|
|
84
|
+
raise ValueError(
|
|
85
|
+
"Feedback must include at least one of: correct_score, correct_reason. "
|
|
86
|
+
"If the result was correct, use confirm() instead."
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
entry = FeedbackEntry(
|
|
90
|
+
eval_name=result.eval_name,
|
|
91
|
+
inputs=inputs,
|
|
92
|
+
original_score=result.score,
|
|
93
|
+
original_reason=result.reason,
|
|
94
|
+
original_passed=result.passed,
|
|
95
|
+
correct_score=correct_score,
|
|
96
|
+
correct_passed=correct_passed,
|
|
97
|
+
correct_reason=correct_reason,
|
|
98
|
+
tags=tags or [],
|
|
99
|
+
metadata=metadata or {},
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
self.store.add(entry)
|
|
103
|
+
logger.info(
|
|
104
|
+
f"Feedback submitted for '{result.eval_name}': "
|
|
105
|
+
f"original={result.score} -> corrected={correct_score}"
|
|
106
|
+
)
|
|
107
|
+
return entry
|
|
108
|
+
|
|
109
|
+
def confirm(
|
|
110
|
+
self,
|
|
111
|
+
result: EvalResult,
|
|
112
|
+
*,
|
|
113
|
+
inputs: Dict[str, Any],
|
|
114
|
+
reason: str = "",
|
|
115
|
+
) -> FeedbackEntry:
|
|
116
|
+
"""Confirm that a metric result was correct.
|
|
117
|
+
|
|
118
|
+
Records that the system got it right, which helps calibration
|
|
119
|
+
accuracy calculations. These entries are stored but NOT injected
|
|
120
|
+
as few-shot examples (since they don't correct anything).
|
|
121
|
+
|
|
122
|
+
Args:
|
|
123
|
+
result: The correct EvalResult.
|
|
124
|
+
inputs: The original inputs.
|
|
125
|
+
reason: Optional note on why this was correct.
|
|
126
|
+
|
|
127
|
+
Returns:
|
|
128
|
+
The stored FeedbackEntry.
|
|
129
|
+
"""
|
|
130
|
+
entry = FeedbackEntry(
|
|
131
|
+
eval_name=result.eval_name,
|
|
132
|
+
inputs=inputs,
|
|
133
|
+
original_score=result.score,
|
|
134
|
+
original_reason=result.reason,
|
|
135
|
+
original_passed=result.passed,
|
|
136
|
+
correct_score=result.score, # Same as original = confirmed correct
|
|
137
|
+
correct_passed=result.passed,
|
|
138
|
+
correct_reason=reason or "Confirmed correct by developer.",
|
|
139
|
+
tags=["confirmed"],
|
|
140
|
+
)
|
|
141
|
+
self.store.add(entry)
|
|
142
|
+
return entry
|
|
143
|
+
|
|
144
|
+
def stats(self, metric_name: str) -> FeedbackStats:
|
|
145
|
+
"""Get aggregate statistics for feedback on a metric.
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
metric_name: The metric to get stats for.
|
|
149
|
+
|
|
150
|
+
Returns:
|
|
151
|
+
FeedbackStats with counts and agreement rates.
|
|
152
|
+
"""
|
|
153
|
+
entries = self.store.get_by_metric(metric_name)
|
|
154
|
+
|
|
155
|
+
if not entries:
|
|
156
|
+
return FeedbackStats(eval_name=metric_name)
|
|
157
|
+
|
|
158
|
+
total = len(entries)
|
|
159
|
+
agreements = 0
|
|
160
|
+
score_deltas = []
|
|
161
|
+
|
|
162
|
+
for e in entries:
|
|
163
|
+
if e.correct_score is not None and e.original_score is not None:
|
|
164
|
+
delta = e.correct_score - e.original_score
|
|
165
|
+
score_deltas.append(delta)
|
|
166
|
+
# Agreement = within 0.1 of each other
|
|
167
|
+
if abs(delta) < 0.1:
|
|
168
|
+
agreements += 1
|
|
169
|
+
|
|
170
|
+
# Score distribution in 0.1 buckets
|
|
171
|
+
distribution: Dict[str, int] = {}
|
|
172
|
+
for e in entries:
|
|
173
|
+
score = e.correct_score if e.correct_score is not None else e.original_score
|
|
174
|
+
if score is not None:
|
|
175
|
+
bucket = f"{int(score * 10) / 10:.1f}"
|
|
176
|
+
distribution[bucket] = distribution.get(bucket, 0) + 1
|
|
177
|
+
|
|
178
|
+
return FeedbackStats(
|
|
179
|
+
eval_name=metric_name,
|
|
180
|
+
total_entries=total,
|
|
181
|
+
agreement_rate=agreements / total if total > 0 else 0.0,
|
|
182
|
+
avg_score_delta=sum(score_deltas) / len(score_deltas) if score_deltas else 0.0,
|
|
183
|
+
score_distribution=distribution,
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
def get_retriever(self, max_examples: int = 3) -> FeedbackRetriever:
|
|
187
|
+
"""Create a FeedbackRetriever wired to this collector's store.
|
|
188
|
+
|
|
189
|
+
Args:
|
|
190
|
+
max_examples: Max few-shot examples to retrieve per query.
|
|
191
|
+
|
|
192
|
+
Returns:
|
|
193
|
+
A FeedbackRetriever instance.
|
|
194
|
+
"""
|
|
195
|
+
return FeedbackRetriever(store=self.store, max_examples=max_examples)
|
|
196
|
+
|
|
197
|
+
def calibrate(
|
|
198
|
+
self,
|
|
199
|
+
metric_name: str,
|
|
200
|
+
threshold_range: tuple = (0.3, 0.9),
|
|
201
|
+
steps: int = 13,
|
|
202
|
+
):
|
|
203
|
+
"""Run threshold calibration for a metric.
|
|
204
|
+
|
|
205
|
+
Args:
|
|
206
|
+
metric_name: Metric to calibrate.
|
|
207
|
+
threshold_range: (min, max) thresholds to search.
|
|
208
|
+
steps: Number of threshold steps to try.
|
|
209
|
+
|
|
210
|
+
Returns:
|
|
211
|
+
CalibrationProfile with optimal threshold.
|
|
212
|
+
"""
|
|
213
|
+
calibrator = ThresholdCalibrator(self.store)
|
|
214
|
+
return calibrator.calibrate(metric_name, threshold_range, steps)
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Integration hooks for wiring feedback into the pipeline.
|
|
2
|
+
|
|
3
|
+
These functions are called from the augmentation flow when a feedback_store
|
|
4
|
+
is provided. Kept in a separate module to avoid circular imports.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import logging
|
|
8
|
+
from typing import Any, Dict, Optional
|
|
9
|
+
|
|
10
|
+
from .store import FeedbackStore
|
|
11
|
+
from .retriever import FeedbackRetriever
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
# Module-level default store (set via configure_feedback)
|
|
16
|
+
_default_store: Optional[FeedbackStore] = None
|
|
17
|
+
_default_max_examples: int = 3
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def configure_feedback(
|
|
21
|
+
store: FeedbackStore,
|
|
22
|
+
max_examples: int = 3,
|
|
23
|
+
) -> None:
|
|
24
|
+
"""Set a global default feedback store for all augmented metric runs.
|
|
25
|
+
|
|
26
|
+
After calling this, all augmented runs will automatically retrieve
|
|
27
|
+
feedback examples -- no need to pass feedback_store= every time.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
store: The FeedbackStore to use globally.
|
|
31
|
+
max_examples: Max few-shot examples per query.
|
|
32
|
+
|
|
33
|
+
Usage:
|
|
34
|
+
from fi.evals.feedback import ChromaFeedbackStore, configure_feedback
|
|
35
|
+
|
|
36
|
+
store = ChromaFeedbackStore()
|
|
37
|
+
configure_feedback(store)
|
|
38
|
+
|
|
39
|
+
# Now all augmented runs automatically use feedback
|
|
40
|
+
result = run_metric("faithfulness", ..., augment=True, model="gemini/...")
|
|
41
|
+
"""
|
|
42
|
+
global _default_store, _default_max_examples
|
|
43
|
+
_default_store = store
|
|
44
|
+
_default_max_examples = max_examples
|
|
45
|
+
logger.info(f"Feedback configured globally (max_examples={max_examples})")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def get_default_store() -> Optional[FeedbackStore]:
|
|
49
|
+
"""Get the globally configured feedback store, if any."""
|
|
50
|
+
return _default_store
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def retrieve_feedback_config(
|
|
54
|
+
metric_name: str,
|
|
55
|
+
inputs: Dict[str, Any],
|
|
56
|
+
store: Optional[FeedbackStore] = None,
|
|
57
|
+
config: Optional[Dict[str, Any]] = None,
|
|
58
|
+
max_examples: Optional[int] = None,
|
|
59
|
+
) -> Dict[str, Any]:
|
|
60
|
+
"""Retrieve feedback examples and inject into config dict.
|
|
61
|
+
|
|
62
|
+
Called from the augmentation flow. Can also be called directly.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
metric_name: Metric being run.
|
|
66
|
+
inputs: Current inputs.
|
|
67
|
+
store: Explicit store override. Falls back to global default.
|
|
68
|
+
config: Existing config dict to merge into.
|
|
69
|
+
max_examples: Override for max examples.
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
Config dict with few_shot_examples populated (or unchanged if
|
|
73
|
+
no store is configured / no feedback found).
|
|
74
|
+
"""
|
|
75
|
+
effective_store = store or _default_store
|
|
76
|
+
if effective_store is None:
|
|
77
|
+
return dict(config or {})
|
|
78
|
+
|
|
79
|
+
n = max_examples or _default_max_examples
|
|
80
|
+
retriever = FeedbackRetriever(store=effective_store, max_examples=n)
|
|
81
|
+
return retriever.inject_into_config(metric_name, inputs, config)
|