agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Feedback retrieval and few-shot formatting.
|
|
2
|
+
|
|
3
|
+
Retrieves semantically similar feedback from the store and formats it
|
|
4
|
+
as few-shot examples for the LLM judge pipeline.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import logging
|
|
9
|
+
from typing import Any, Dict, List, Optional
|
|
10
|
+
|
|
11
|
+
from .store import FeedbackStore
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class FeedbackRetriever:
|
|
17
|
+
"""Retrieves semantically similar feedback and formats as few-shot examples.
|
|
18
|
+
|
|
19
|
+
This is the bridge between the feedback store and the LLM judge pipeline.
|
|
20
|
+
When a metric is run with a feedback store, the retriever:
|
|
21
|
+
|
|
22
|
+
1. Builds an embedding query from the current inputs
|
|
23
|
+
2. Searches the store for similar past feedback entries
|
|
24
|
+
3. Converts matching entries to the few_shot_examples format
|
|
25
|
+
expected by CustomLLMJudge's Jinja2 template
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
store: The FeedbackStore to search.
|
|
29
|
+
max_examples: Maximum number of few-shot examples to inject. Default 3.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
store: FeedbackStore,
|
|
35
|
+
max_examples: int = 3,
|
|
36
|
+
):
|
|
37
|
+
self.store = store
|
|
38
|
+
self.max_examples = max_examples
|
|
39
|
+
|
|
40
|
+
def build_query_text(self, metric_name: str, inputs: Dict[str, Any]) -> str:
|
|
41
|
+
"""Build a query string from inputs for semantic search.
|
|
42
|
+
|
|
43
|
+
Uses the same concatenation strategy as FeedbackEntry.to_embedding_text()
|
|
44
|
+
to ensure query-document alignment.
|
|
45
|
+
"""
|
|
46
|
+
parts = [f"metric: {metric_name}"]
|
|
47
|
+
for key in ("output", "response", "context", "input", "query"):
|
|
48
|
+
val = inputs.get(key)
|
|
49
|
+
if val:
|
|
50
|
+
text = val if isinstance(val, str) else json.dumps(val, default=str)
|
|
51
|
+
parts.append(f"{key}: {text[:500]}")
|
|
52
|
+
return "\n".join(parts)
|
|
53
|
+
|
|
54
|
+
def retrieve_few_shot_examples(
|
|
55
|
+
self,
|
|
56
|
+
metric_name: str,
|
|
57
|
+
inputs: Dict[str, Any],
|
|
58
|
+
) -> List[Dict[str, Any]]:
|
|
59
|
+
"""Retrieve few-shot examples from feedback store.
|
|
60
|
+
|
|
61
|
+
Returns a list of dicts in the format expected by
|
|
62
|
+
CustomLLMJudge's config["few_shot_examples"]:
|
|
63
|
+
|
|
64
|
+
[{"inputs": {...}, "output": '{"score": 0.8, "reason": "..."}'}]
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
metric_name: The metric being run.
|
|
68
|
+
inputs: The current inputs.
|
|
69
|
+
|
|
70
|
+
Returns:
|
|
71
|
+
List of few-shot example dicts, possibly empty if no feedback exists.
|
|
72
|
+
"""
|
|
73
|
+
if self.store.count(metric_name) == 0:
|
|
74
|
+
return []
|
|
75
|
+
|
|
76
|
+
query_text = self.build_query_text(metric_name, inputs)
|
|
77
|
+
|
|
78
|
+
similar_entries = self.store.query_similar(
|
|
79
|
+
metric_name=metric_name,
|
|
80
|
+
text=query_text,
|
|
81
|
+
n_results=self.max_examples,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
if not similar_entries:
|
|
85
|
+
return []
|
|
86
|
+
|
|
87
|
+
examples = []
|
|
88
|
+
for entry in similar_entries:
|
|
89
|
+
# Only include entries where the developer provided a correction
|
|
90
|
+
if entry.correct_score is None and not entry.correct_reason:
|
|
91
|
+
continue
|
|
92
|
+
examples.append(entry.to_few_shot())
|
|
93
|
+
|
|
94
|
+
if examples:
|
|
95
|
+
logger.debug(
|
|
96
|
+
f"Retrieved {len(examples)} feedback examples for '{metric_name}'"
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
return examples
|
|
100
|
+
|
|
101
|
+
def inject_into_config(
|
|
102
|
+
self,
|
|
103
|
+
metric_name: str,
|
|
104
|
+
inputs: Dict[str, Any],
|
|
105
|
+
config: Optional[Dict[str, Any]] = None,
|
|
106
|
+
) -> Dict[str, Any]:
|
|
107
|
+
"""Retrieve feedback and merge into a config dict for LLMEngine/CustomLLMJudge.
|
|
108
|
+
|
|
109
|
+
This is the primary integration point. Call this before passing config
|
|
110
|
+
to LLMEngine.run() to inject few-shot examples.
|
|
111
|
+
|
|
112
|
+
Args:
|
|
113
|
+
metric_name: Metric name.
|
|
114
|
+
inputs: Current inputs.
|
|
115
|
+
config: Existing config dict (will not be mutated).
|
|
116
|
+
|
|
117
|
+
Returns:
|
|
118
|
+
New config dict with few_shot_examples populated.
|
|
119
|
+
"""
|
|
120
|
+
config = dict(config or {})
|
|
121
|
+
|
|
122
|
+
examples = self.retrieve_few_shot_examples(metric_name, inputs)
|
|
123
|
+
if examples:
|
|
124
|
+
# Merge with any existing few-shot examples
|
|
125
|
+
existing = config.get("few_shot_examples", [])
|
|
126
|
+
config["few_shot_examples"] = existing + examples
|
|
127
|
+
|
|
128
|
+
return config
|
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""Feedback storage backends.
|
|
2
|
+
|
|
3
|
+
Provides abstract FeedbackStore and two implementations:
|
|
4
|
+
- InMemoryFeedbackStore: for testing and small-scale usage
|
|
5
|
+
- ChromaFeedbackStore: for production with semantic vector search
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import logging
|
|
10
|
+
from abc import ABC, abstractmethod
|
|
11
|
+
from typing import Any, Dict, List, Optional
|
|
12
|
+
|
|
13
|
+
from .types import FeedbackEntry
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class FeedbackStore(ABC):
|
|
19
|
+
"""Abstract base class for feedback persistence."""
|
|
20
|
+
|
|
21
|
+
@abstractmethod
|
|
22
|
+
def add(self, entry: FeedbackEntry) -> str:
|
|
23
|
+
"""Store a feedback entry. Returns the entry ID."""
|
|
24
|
+
...
|
|
25
|
+
|
|
26
|
+
@abstractmethod
|
|
27
|
+
def query_similar(
|
|
28
|
+
self,
|
|
29
|
+
metric_name: str,
|
|
30
|
+
text: str,
|
|
31
|
+
n_results: int = 5,
|
|
32
|
+
) -> List[FeedbackEntry]:
|
|
33
|
+
"""Find feedback entries semantically similar to the given text,
|
|
34
|
+
filtered by metric_name."""
|
|
35
|
+
...
|
|
36
|
+
|
|
37
|
+
@abstractmethod
|
|
38
|
+
def get_by_metric(self, metric_name: str, limit: int = 100) -> List[FeedbackEntry]:
|
|
39
|
+
"""Get all feedback entries for a specific metric."""
|
|
40
|
+
...
|
|
41
|
+
|
|
42
|
+
@abstractmethod
|
|
43
|
+
def count(self, metric_name: Optional[str] = None) -> int:
|
|
44
|
+
"""Count entries, optionally filtered by metric_name."""
|
|
45
|
+
...
|
|
46
|
+
|
|
47
|
+
@abstractmethod
|
|
48
|
+
def delete(self, entry_id: str) -> bool:
|
|
49
|
+
"""Delete a feedback entry by ID."""
|
|
50
|
+
...
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class InMemoryFeedbackStore(FeedbackStore):
|
|
54
|
+
"""In-memory feedback store for testing and small-scale usage.
|
|
55
|
+
|
|
56
|
+
No vector search -- falls back to recency-based retrieval.
|
|
57
|
+
Suitable for unit tests and quick experimentation.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
def __init__(self):
|
|
61
|
+
self._entries: Dict[str, FeedbackEntry] = {}
|
|
62
|
+
|
|
63
|
+
def add(self, entry: FeedbackEntry) -> str:
|
|
64
|
+
self._entries[entry.id] = entry
|
|
65
|
+
return entry.id
|
|
66
|
+
|
|
67
|
+
def query_similar(
|
|
68
|
+
self,
|
|
69
|
+
metric_name: str,
|
|
70
|
+
text: str,
|
|
71
|
+
n_results: int = 5,
|
|
72
|
+
) -> List[FeedbackEntry]:
|
|
73
|
+
# No semantic search -- return most recent entries for this metric
|
|
74
|
+
filtered = [
|
|
75
|
+
e for e in self._entries.values()
|
|
76
|
+
if e.eval_name == metric_name
|
|
77
|
+
]
|
|
78
|
+
filtered.sort(key=lambda e: e.created_at, reverse=True)
|
|
79
|
+
return filtered[:n_results]
|
|
80
|
+
|
|
81
|
+
def get_by_metric(self, metric_name: str, limit: int = 100) -> List[FeedbackEntry]:
|
|
82
|
+
return [
|
|
83
|
+
e for e in self._entries.values()
|
|
84
|
+
if e.eval_name == metric_name
|
|
85
|
+
][:limit]
|
|
86
|
+
|
|
87
|
+
def count(self, metric_name: Optional[str] = None) -> int:
|
|
88
|
+
if metric_name:
|
|
89
|
+
return sum(1 for e in self._entries.values() if e.eval_name == metric_name)
|
|
90
|
+
return len(self._entries)
|
|
91
|
+
|
|
92
|
+
def delete(self, entry_id: str) -> bool:
|
|
93
|
+
return self._entries.pop(entry_id, None) is not None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class ChromaFeedbackStore(FeedbackStore):
|
|
97
|
+
"""ChromaDB-backed feedback store with semantic vector search.
|
|
98
|
+
|
|
99
|
+
Supports two modes:
|
|
100
|
+
- Local: in-process persistent ChromaDB (default)
|
|
101
|
+
- Service: connects to a remote ChromaDB server (e.g. Docker container)
|
|
102
|
+
|
|
103
|
+
Embedding is handled by ChromaDB's built-in default embedding function
|
|
104
|
+
(all-MiniLM-L6-v2 via sentence-transformers) OR via a LiteLLM embedding
|
|
105
|
+
function for API-based embeddings.
|
|
106
|
+
|
|
107
|
+
Args:
|
|
108
|
+
host: ChromaDB server host. None = local persistent mode.
|
|
109
|
+
port: ChromaDB server port. Default 8000.
|
|
110
|
+
persist_directory: Local storage path (local mode only).
|
|
111
|
+
Default "~/.fi/feedback/chroma".
|
|
112
|
+
collection_prefix: Prefix for ChromaDB collection names.
|
|
113
|
+
embedding_model: LiteLLM model string for embeddings.
|
|
114
|
+
None = use ChromaDB's default (sentence-transformers).
|
|
115
|
+
"""
|
|
116
|
+
|
|
117
|
+
def __init__(
|
|
118
|
+
self,
|
|
119
|
+
host: Optional[str] = None,
|
|
120
|
+
port: int = 8000,
|
|
121
|
+
persist_directory: Optional[str] = None,
|
|
122
|
+
collection_prefix: str = "fi_feedback",
|
|
123
|
+
embedding_model: Optional[str] = None,
|
|
124
|
+
):
|
|
125
|
+
try:
|
|
126
|
+
import chromadb
|
|
127
|
+
except ImportError:
|
|
128
|
+
raise ImportError(
|
|
129
|
+
"chromadb is required for ChromaFeedbackStore. "
|
|
130
|
+
"Install it with: pip install ai-evaluation[feedback]"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
self._collection_prefix = collection_prefix
|
|
134
|
+
self._embedding_model = embedding_model
|
|
135
|
+
|
|
136
|
+
# Initialize ChromaDB client
|
|
137
|
+
if host:
|
|
138
|
+
self._client = chromadb.HttpClient(host=host, port=port)
|
|
139
|
+
logger.info(f"Connected to ChromaDB server at {host}:{port}")
|
|
140
|
+
else:
|
|
141
|
+
import os
|
|
142
|
+
path = persist_directory or os.path.expanduser("~/.fi/feedback/chroma")
|
|
143
|
+
os.makedirs(path, exist_ok=True)
|
|
144
|
+
self._client = chromadb.PersistentClient(path=path)
|
|
145
|
+
logger.info(f"Using local ChromaDB at {path}")
|
|
146
|
+
|
|
147
|
+
# Set up embedding function
|
|
148
|
+
self._embedding_fn = None
|
|
149
|
+
if embedding_model:
|
|
150
|
+
self._embedding_fn = self._make_litellm_embedding_fn(embedding_model)
|
|
151
|
+
|
|
152
|
+
@staticmethod
|
|
153
|
+
def _make_litellm_embedding_fn(model: str):
|
|
154
|
+
"""Create a ChromaDB-compatible embedding function using LiteLLM."""
|
|
155
|
+
from chromadb.api.types import EmbeddingFunction, Documents, Embeddings
|
|
156
|
+
import litellm
|
|
157
|
+
|
|
158
|
+
class LiteLLMEmbedding(EmbeddingFunction):
|
|
159
|
+
def __call__(self, input: Documents) -> Embeddings:
|
|
160
|
+
response = litellm.embedding(model=model, input=input)
|
|
161
|
+
return [item["embedding"] for item in response.data]
|
|
162
|
+
|
|
163
|
+
return LiteLLMEmbedding()
|
|
164
|
+
|
|
165
|
+
def _get_collection(self, metric_name: str):
|
|
166
|
+
"""Get or create a ChromaDB collection for a specific metric."""
|
|
167
|
+
name = f"{self._collection_prefix}_{metric_name}".replace(".", "_")
|
|
168
|
+
kwargs: Dict[str, Any] = {"name": name}
|
|
169
|
+
if self._embedding_fn:
|
|
170
|
+
kwargs["embedding_function"] = self._embedding_fn
|
|
171
|
+
return self._client.get_or_create_collection(**kwargs)
|
|
172
|
+
|
|
173
|
+
def add(self, entry: FeedbackEntry) -> str:
|
|
174
|
+
collection = self._get_collection(entry.eval_name)
|
|
175
|
+
collection.add(
|
|
176
|
+
ids=[entry.id],
|
|
177
|
+
documents=[entry.to_embedding_text()],
|
|
178
|
+
metadatas=[{
|
|
179
|
+
"metric_name": entry.eval_name,
|
|
180
|
+
"original_score": entry.original_score or 0.0,
|
|
181
|
+
"correct_score": entry.correct_score if entry.correct_score is not None else -1.0,
|
|
182
|
+
"correct_reason": entry.correct_reason[:1000],
|
|
183
|
+
"inputs_json": json.dumps(entry.inputs, default=str)[:4000],
|
|
184
|
+
"original_reason": entry.original_reason[:1000],
|
|
185
|
+
"created_at": entry.created_at.isoformat(),
|
|
186
|
+
}],
|
|
187
|
+
)
|
|
188
|
+
return entry.id
|
|
189
|
+
|
|
190
|
+
def query_similar(
|
|
191
|
+
self,
|
|
192
|
+
metric_name: str,
|
|
193
|
+
text: str,
|
|
194
|
+
n_results: int = 5,
|
|
195
|
+
) -> List[FeedbackEntry]:
|
|
196
|
+
collection = self._get_collection(metric_name)
|
|
197
|
+
|
|
198
|
+
if collection.count() == 0:
|
|
199
|
+
return []
|
|
200
|
+
|
|
201
|
+
actual_n = min(n_results, collection.count())
|
|
202
|
+
|
|
203
|
+
results = collection.query(
|
|
204
|
+
query_texts=[text],
|
|
205
|
+
n_results=actual_n,
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
entries = []
|
|
209
|
+
for i, meta in enumerate(results["metadatas"][0]):
|
|
210
|
+
inputs = {}
|
|
211
|
+
try:
|
|
212
|
+
inputs = json.loads(meta.get("inputs_json", "{}"))
|
|
213
|
+
except (json.JSONDecodeError, TypeError):
|
|
214
|
+
pass
|
|
215
|
+
|
|
216
|
+
correct_score_val = meta.get("correct_score", -1.0)
|
|
217
|
+
entry = FeedbackEntry(
|
|
218
|
+
id=results["ids"][0][i],
|
|
219
|
+
eval_name=meta.get("metric_name", metric_name),
|
|
220
|
+
inputs=inputs,
|
|
221
|
+
original_score=meta.get("original_score"),
|
|
222
|
+
original_reason=meta.get("original_reason", ""),
|
|
223
|
+
correct_score=correct_score_val if correct_score_val >= 0 else None,
|
|
224
|
+
correct_reason=meta.get("correct_reason", ""),
|
|
225
|
+
)
|
|
226
|
+
entries.append(entry)
|
|
227
|
+
|
|
228
|
+
return entries
|
|
229
|
+
|
|
230
|
+
def get_by_metric(self, metric_name: str, limit: int = 100) -> List[FeedbackEntry]:
|
|
231
|
+
collection = self._get_collection(metric_name)
|
|
232
|
+
if collection.count() == 0:
|
|
233
|
+
return []
|
|
234
|
+
results = collection.get(limit=limit)
|
|
235
|
+
entries = []
|
|
236
|
+
for i, meta in enumerate(results["metadatas"]):
|
|
237
|
+
inputs = {}
|
|
238
|
+
try:
|
|
239
|
+
inputs = json.loads(meta.get("inputs_json", "{}"))
|
|
240
|
+
except (json.JSONDecodeError, TypeError):
|
|
241
|
+
pass
|
|
242
|
+
correct_score_val = meta.get("correct_score", -1.0)
|
|
243
|
+
entry = FeedbackEntry(
|
|
244
|
+
id=results["ids"][i],
|
|
245
|
+
eval_name=meta.get("metric_name", metric_name),
|
|
246
|
+
inputs=inputs,
|
|
247
|
+
original_score=meta.get("original_score"),
|
|
248
|
+
original_reason=meta.get("original_reason", ""),
|
|
249
|
+
correct_score=correct_score_val if correct_score_val >= 0 else None,
|
|
250
|
+
correct_reason=meta.get("correct_reason", ""),
|
|
251
|
+
)
|
|
252
|
+
entries.append(entry)
|
|
253
|
+
return entries
|
|
254
|
+
|
|
255
|
+
def count(self, metric_name: Optional[str] = None) -> int:
|
|
256
|
+
if metric_name:
|
|
257
|
+
return self._get_collection(metric_name).count()
|
|
258
|
+
total = 0
|
|
259
|
+
for col in self._client.list_collections():
|
|
260
|
+
if col.name.startswith(self._collection_prefix):
|
|
261
|
+
total += col.count()
|
|
262
|
+
return total
|
|
263
|
+
|
|
264
|
+
def delete(self, entry_id: str) -> bool:
|
|
265
|
+
for col in self._client.list_collections():
|
|
266
|
+
if col.name.startswith(self._collection_prefix):
|
|
267
|
+
try:
|
|
268
|
+
col.delete(ids=[entry_id])
|
|
269
|
+
return True
|
|
270
|
+
except Exception:
|
|
271
|
+
continue
|
|
272
|
+
return False
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Type definitions for the Feedback Loop system.
|
|
2
|
+
|
|
3
|
+
Provides dataclasses for storing developer feedback on evaluation results,
|
|
4
|
+
calibration profiles, and aggregate statistics.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import uuid
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
from typing import Any, Dict, List, Optional
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class FeedbackEntry:
|
|
16
|
+
"""A single piece of developer feedback on an evaluation result."""
|
|
17
|
+
|
|
18
|
+
id: str = field(default_factory=lambda: str(uuid.uuid4()))
|
|
19
|
+
|
|
20
|
+
# What was evaluated
|
|
21
|
+
eval_name: str = ""
|
|
22
|
+
inputs: Dict[str, Any] = field(default_factory=dict)
|
|
23
|
+
|
|
24
|
+
# What the system produced
|
|
25
|
+
original_score: Optional[float] = None
|
|
26
|
+
original_reason: str = ""
|
|
27
|
+
original_passed: Optional[bool] = None
|
|
28
|
+
|
|
29
|
+
# What the developer says is correct
|
|
30
|
+
correct_score: Optional[float] = None
|
|
31
|
+
correct_passed: Optional[bool] = None
|
|
32
|
+
correct_reason: str = ""
|
|
33
|
+
|
|
34
|
+
# Metadata
|
|
35
|
+
tags: List[str] = field(default_factory=list)
|
|
36
|
+
created_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
|
37
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
38
|
+
|
|
39
|
+
def to_few_shot(self) -> Dict[str, Any]:
|
|
40
|
+
"""Convert to the format expected by CustomLLMJudge few_shot_examples.
|
|
41
|
+
|
|
42
|
+
Returns dict matching the template schema:
|
|
43
|
+
{"inputs": {...}, "output": "<json with score/reason>"}
|
|
44
|
+
"""
|
|
45
|
+
output_dict = {
|
|
46
|
+
"score": self.correct_score if self.correct_score is not None else self.original_score,
|
|
47
|
+
"reason": self.correct_reason or self.original_reason,
|
|
48
|
+
}
|
|
49
|
+
return {
|
|
50
|
+
"inputs": self.inputs,
|
|
51
|
+
"output": json.dumps(output_dict),
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
def to_embedding_text(self) -> str:
|
|
55
|
+
"""Create a text representation for embedding.
|
|
56
|
+
|
|
57
|
+
Concatenates key input fields into a string suitable for
|
|
58
|
+
semantic embedding. Prioritizes output, context, and input fields.
|
|
59
|
+
"""
|
|
60
|
+
parts = [f"metric: {self.eval_name}"]
|
|
61
|
+
for key in ("output", "response", "context", "input", "query"):
|
|
62
|
+
val = self.inputs.get(key)
|
|
63
|
+
if val:
|
|
64
|
+
text = val if isinstance(val, str) else json.dumps(val, default=str)
|
|
65
|
+
parts.append(f"{key}: {text[:500]}")
|
|
66
|
+
return "\n".join(parts)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class CalibrationProfile:
|
|
71
|
+
"""Optimized threshold settings for a metric based on feedback."""
|
|
72
|
+
|
|
73
|
+
eval_name: str
|
|
74
|
+
optimal_threshold: float
|
|
75
|
+
sample_size: int
|
|
76
|
+
accuracy_at_threshold: float # % of feedback entries that agree at this threshold
|
|
77
|
+
|
|
78
|
+
# Distribution stats
|
|
79
|
+
score_mean: float = 0.0
|
|
80
|
+
score_std: float = 0.0
|
|
81
|
+
|
|
82
|
+
# Confusion matrix at the optimal threshold
|
|
83
|
+
true_positives: int = 0
|
|
84
|
+
false_positives: int = 0
|
|
85
|
+
true_negatives: int = 0
|
|
86
|
+
false_negatives: int = 0
|
|
87
|
+
|
|
88
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass
|
|
92
|
+
class FeedbackStats:
|
|
93
|
+
"""Aggregate statistics for feedback on a given metric."""
|
|
94
|
+
|
|
95
|
+
eval_name: str
|
|
96
|
+
total_entries: int = 0
|
|
97
|
+
agreement_rate: float = 0.0 # How often original == correct
|
|
98
|
+
avg_score_delta: float = 0.0 # avg(correct_score - original_score)
|
|
99
|
+
score_distribution: Dict[str, int] = field(default_factory=dict) # buckets
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Evaluation Framework
|
|
2
|
+
|
|
3
|
+
A scalable evaluation infrastructure for AI systems with support for blocking, non-blocking, and distributed execution modes.
|
|
4
|
+
|
|
5
|
+
## Quick Start
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
from fi.evals import FrameworkEvaluator, ExecutionMode
|
|
9
|
+
from fi.evals.framework.evals import CoherenceEval, ActionSafetyEval
|
|
10
|
+
|
|
11
|
+
# Create an evaluator with multiple evaluations
|
|
12
|
+
evaluator = FrameworkEvaluator(
|
|
13
|
+
evaluations=[
|
|
14
|
+
CoherenceEval(),
|
|
15
|
+
ActionSafetyEval(),
|
|
16
|
+
],
|
|
17
|
+
mode=ExecutionMode.BLOCKING,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
# Run evaluations
|
|
21
|
+
result = evaluator.run({
|
|
22
|
+
"response": "Paris is the capital of France. It is in Western Europe.",
|
|
23
|
+
"trajectory": [
|
|
24
|
+
{"type": "tool_call", "tool": "search", "args": "Paris facts"},
|
|
25
|
+
],
|
|
26
|
+
})
|
|
27
|
+
|
|
28
|
+
# Check results
|
|
29
|
+
for r in result.results:
|
|
30
|
+
print(f"{r.eval_name}: score={r.value.score:.2f}, passed={r.value.passed}")
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Execution Modes
|
|
34
|
+
|
|
35
|
+
| Mode | Use Case | Latency Impact |
|
|
36
|
+
|------|----------|----------------|
|
|
37
|
+
| `BLOCKING` | Development, testing, sync workflows | Full evaluation time |
|
|
38
|
+
| `NON_BLOCKING` | Production, real-time applications | Zero (async) |
|
|
39
|
+
| `DISTRIBUTED` | Batch processing, high throughput | Zero (remote) |
|
|
40
|
+
|
|
41
|
+
## Available Evaluations
|
|
42
|
+
|
|
43
|
+
### Semantic Evaluations
|
|
44
|
+
- `CoherenceEval` - Check text coherence
|
|
45
|
+
|
|
46
|
+
### Agentic Evaluations
|
|
47
|
+
- `ActionSafetyEval` - Safety scanning
|
|
48
|
+
- `ReasoningQualityEval` - Reasoning quality
|
|
49
|
+
|
|
50
|
+
### Custom Evaluation Builders
|
|
51
|
+
- `EvalBuilder` - Fluent builder pattern
|
|
52
|
+
- `@custom_eval` - Decorator for functions
|
|
53
|
+
- `simple_eval()` - Score-based evaluation
|
|
54
|
+
- `comparison_eval()` - Compare two fields
|
|
55
|
+
- `threshold_eval()` - Min/max thresholds
|
|
56
|
+
- `pattern_match_eval()` - Regex patterns
|
|
57
|
+
|
|
58
|
+
## OpenTelemetry Integration
|
|
59
|
+
|
|
60
|
+
All evaluations automatically generate span attributes compatible with OpenTelemetry:
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
from fi.evals.framework import register_current_span, async_evaluator
|
|
64
|
+
|
|
65
|
+
# Register span for cross-thread enrichment
|
|
66
|
+
with tracer.start_as_current_span("llm_call") as span:
|
|
67
|
+
register_current_span()
|
|
68
|
+
|
|
69
|
+
response = llm.complete(prompt)
|
|
70
|
+
evaluator.run({"response": response}) # Enriches span automatically
|
|
71
|
+
|
|
72
|
+
return response
|
|
73
|
+
|
|
74
|
+
# Span attributes include:
|
|
75
|
+
# eval.coherence.score
|
|
76
|
+
# eval.coherence.passed
|
|
77
|
+
# eval.action_safety.score
|
|
78
|
+
# etc.
|
|
79
|
+
```
|