agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Guardrails Gateway - High-level API for content screening.
|
|
3
|
+
|
|
4
|
+
Provides a simple, ergonomic interface for screening content
|
|
5
|
+
with automatic backend management and context managers.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from contextlib import asynccontextmanager, contextmanager
|
|
9
|
+
from typing import AsyncIterator, Iterator, List, Optional
|
|
10
|
+
|
|
11
|
+
from fi.evals.guardrails.base import Guardrails
|
|
12
|
+
from fi.evals.guardrails.config import GuardrailModel, GuardrailsConfig, AggregationStrategy
|
|
13
|
+
from fi.evals.guardrails.types import GuardrailsResponse
|
|
14
|
+
from fi.evals.guardrails.discovery import discover_backends, get_backend_details
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ScreeningSession:
|
|
18
|
+
"""A screening session for synchronous operations."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, guardrails: Guardrails):
|
|
21
|
+
self._guardrails = guardrails
|
|
22
|
+
self._history: List[GuardrailsResponse] = []
|
|
23
|
+
|
|
24
|
+
def input(self, content: str, metadata: Optional[dict] = None) -> GuardrailsResponse:
|
|
25
|
+
"""Screen user input.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
content: The user input to screen.
|
|
29
|
+
metadata: Optional metadata for the request.
|
|
30
|
+
|
|
31
|
+
Returns:
|
|
32
|
+
GuardrailsResponse with screening results.
|
|
33
|
+
"""
|
|
34
|
+
result = self._guardrails.screen_input(content, metadata=metadata)
|
|
35
|
+
self._history.append(result)
|
|
36
|
+
return result
|
|
37
|
+
|
|
38
|
+
def output(
|
|
39
|
+
self,
|
|
40
|
+
content: str,
|
|
41
|
+
context: Optional[str] = None,
|
|
42
|
+
metadata: Optional[dict] = None,
|
|
43
|
+
) -> GuardrailsResponse:
|
|
44
|
+
"""Screen LLM output.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
content: The LLM response to screen.
|
|
48
|
+
context: Optional original user query for context.
|
|
49
|
+
metadata: Optional metadata for the request.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
GuardrailsResponse with screening results.
|
|
53
|
+
"""
|
|
54
|
+
result = self._guardrails.screen_output(content, context=context, metadata=metadata)
|
|
55
|
+
self._history.append(result)
|
|
56
|
+
return result
|
|
57
|
+
|
|
58
|
+
def retrieval(
|
|
59
|
+
self,
|
|
60
|
+
chunks: List[str],
|
|
61
|
+
query: Optional[str] = None,
|
|
62
|
+
metadata: Optional[dict] = None,
|
|
63
|
+
) -> List[GuardrailsResponse]:
|
|
64
|
+
"""Screen retrieval chunks.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
chunks: List of document chunks to screen.
|
|
68
|
+
query: Optional user query for context.
|
|
69
|
+
metadata: Optional metadata for the request.
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
List of GuardrailsResponse, one per chunk.
|
|
73
|
+
"""
|
|
74
|
+
results = self._guardrails.screen_retrieval(chunks, query=query, metadata=metadata)
|
|
75
|
+
self._history.extend(results)
|
|
76
|
+
return results
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def history(self) -> List[GuardrailsResponse]:
|
|
80
|
+
"""Get the history of all screening results in this session."""
|
|
81
|
+
return self._history.copy()
|
|
82
|
+
|
|
83
|
+
@property
|
|
84
|
+
def all_passed(self) -> bool:
|
|
85
|
+
"""Check if all screenings in this session passed."""
|
|
86
|
+
return all(r.passed for r in self._history)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class AsyncScreeningSession:
|
|
90
|
+
"""A screening session for async operations."""
|
|
91
|
+
|
|
92
|
+
def __init__(self, guardrails: Guardrails):
|
|
93
|
+
self._guardrails = guardrails
|
|
94
|
+
self._history: List[GuardrailsResponse] = []
|
|
95
|
+
|
|
96
|
+
async def input(self, content: str, metadata: Optional[dict] = None) -> GuardrailsResponse:
|
|
97
|
+
"""Screen user input asynchronously.
|
|
98
|
+
|
|
99
|
+
Args:
|
|
100
|
+
content: The user input to screen.
|
|
101
|
+
metadata: Optional metadata for the request.
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
GuardrailsResponse with screening results.
|
|
105
|
+
"""
|
|
106
|
+
result = await self._guardrails.screen_input_async(content, metadata=metadata)
|
|
107
|
+
self._history.append(result)
|
|
108
|
+
return result
|
|
109
|
+
|
|
110
|
+
async def output(
|
|
111
|
+
self,
|
|
112
|
+
content: str,
|
|
113
|
+
context: Optional[str] = None,
|
|
114
|
+
metadata: Optional[dict] = None,
|
|
115
|
+
) -> GuardrailsResponse:
|
|
116
|
+
"""Screen LLM output asynchronously.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
content: The LLM response to screen.
|
|
120
|
+
context: Optional original user query for context.
|
|
121
|
+
metadata: Optional metadata for the request.
|
|
122
|
+
|
|
123
|
+
Returns:
|
|
124
|
+
GuardrailsResponse with screening results.
|
|
125
|
+
"""
|
|
126
|
+
result = await self._guardrails.screen_output_async(content, context=context, metadata=metadata)
|
|
127
|
+
self._history.append(result)
|
|
128
|
+
return result
|
|
129
|
+
|
|
130
|
+
async def retrieval(
|
|
131
|
+
self,
|
|
132
|
+
chunks: List[str],
|
|
133
|
+
query: Optional[str] = None,
|
|
134
|
+
metadata: Optional[dict] = None,
|
|
135
|
+
) -> List[GuardrailsResponse]:
|
|
136
|
+
"""Screen retrieval chunks asynchronously.
|
|
137
|
+
|
|
138
|
+
Args:
|
|
139
|
+
chunks: List of document chunks to screen.
|
|
140
|
+
query: Optional user query for context.
|
|
141
|
+
metadata: Optional metadata for the request.
|
|
142
|
+
|
|
143
|
+
Returns:
|
|
144
|
+
List of GuardrailsResponse, one per chunk.
|
|
145
|
+
"""
|
|
146
|
+
results = await self._guardrails.screen_retrieval_async(chunks, query=query, metadata=metadata)
|
|
147
|
+
self._history.extend(results)
|
|
148
|
+
return results
|
|
149
|
+
|
|
150
|
+
async def batch(
|
|
151
|
+
self,
|
|
152
|
+
contents: List[str],
|
|
153
|
+
metadata: Optional[dict] = None,
|
|
154
|
+
) -> List[GuardrailsResponse]:
|
|
155
|
+
"""Screen multiple contents in batch asynchronously.
|
|
156
|
+
|
|
157
|
+
Args:
|
|
158
|
+
contents: List of contents to screen.
|
|
159
|
+
metadata: Optional metadata for the request.
|
|
160
|
+
|
|
161
|
+
Returns:
|
|
162
|
+
List of GuardrailsResponse, one per content.
|
|
163
|
+
"""
|
|
164
|
+
results = await self._guardrails.screen_batch_async(contents, metadata=metadata)
|
|
165
|
+
self._history.extend(results)
|
|
166
|
+
return results
|
|
167
|
+
|
|
168
|
+
@property
|
|
169
|
+
def history(self) -> List[GuardrailsResponse]:
|
|
170
|
+
"""Get the history of all screening results in this session."""
|
|
171
|
+
return self._history.copy()
|
|
172
|
+
|
|
173
|
+
@property
|
|
174
|
+
def all_passed(self) -> bool:
|
|
175
|
+
"""Check if all screenings in this session passed."""
|
|
176
|
+
return all(r.passed for r in self._history)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class GuardrailsGateway:
|
|
180
|
+
"""High-level gateway for content screening.
|
|
181
|
+
|
|
182
|
+
Provides convenient methods for creating screening sessions
|
|
183
|
+
and managing guardrails configuration.
|
|
184
|
+
|
|
185
|
+
Example:
|
|
186
|
+
# Simple usage
|
|
187
|
+
gateway = GuardrailsGateway()
|
|
188
|
+
result = gateway.screen("Hello world")
|
|
189
|
+
|
|
190
|
+
# With context manager
|
|
191
|
+
with gateway.screening() as session:
|
|
192
|
+
input_result = session.input("user message")
|
|
193
|
+
if input_result.passed:
|
|
194
|
+
response = call_llm("user message")
|
|
195
|
+
output_result = session.output(response)
|
|
196
|
+
|
|
197
|
+
# Async context manager
|
|
198
|
+
async with gateway.screening_async() as session:
|
|
199
|
+
input_result = await session.input("user message")
|
|
200
|
+
output_result = await session.output(response)
|
|
201
|
+
"""
|
|
202
|
+
|
|
203
|
+
def __init__(
|
|
204
|
+
self,
|
|
205
|
+
models: Optional[List[GuardrailModel]] = None,
|
|
206
|
+
config: Optional[GuardrailsConfig] = None,
|
|
207
|
+
auto_discover: bool = False,
|
|
208
|
+
):
|
|
209
|
+
"""Initialize the gateway.
|
|
210
|
+
|
|
211
|
+
Args:
|
|
212
|
+
models: List of models to use. If None, uses default.
|
|
213
|
+
config: Full configuration object. Takes precedence over models.
|
|
214
|
+
auto_discover: If True and no models specified, auto-discover available backends.
|
|
215
|
+
"""
|
|
216
|
+
if config:
|
|
217
|
+
self._config = config
|
|
218
|
+
elif models:
|
|
219
|
+
self._config = GuardrailsConfig(models=models)
|
|
220
|
+
elif auto_discover:
|
|
221
|
+
available = discover_backends()
|
|
222
|
+
if not available:
|
|
223
|
+
raise ValueError("No backends available. Set API keys or start VLLM server.")
|
|
224
|
+
self._config = GuardrailsConfig(models=available[:1]) # Use first available
|
|
225
|
+
else:
|
|
226
|
+
self._config = GuardrailsConfig()
|
|
227
|
+
|
|
228
|
+
self._guardrails = Guardrails(config=self._config)
|
|
229
|
+
|
|
230
|
+
@classmethod
|
|
231
|
+
def with_openai(cls, api_key: Optional[str] = None) -> "GuardrailsGateway":
|
|
232
|
+
"""Create a gateway using OpenAI Moderation (FREE).
|
|
233
|
+
|
|
234
|
+
Args:
|
|
235
|
+
api_key: OpenAI API key. If None, uses OPENAI_API_KEY env var.
|
|
236
|
+
|
|
237
|
+
Returns:
|
|
238
|
+
GuardrailsGateway configured for OpenAI.
|
|
239
|
+
"""
|
|
240
|
+
import os
|
|
241
|
+
if api_key:
|
|
242
|
+
os.environ["OPENAI_API_KEY"] = api_key
|
|
243
|
+
|
|
244
|
+
config = GuardrailsConfig(
|
|
245
|
+
models=[GuardrailModel.OPENAI_MODERATION],
|
|
246
|
+
timeout_ms=30000,
|
|
247
|
+
)
|
|
248
|
+
return cls(config=config)
|
|
249
|
+
|
|
250
|
+
@classmethod
|
|
251
|
+
def with_azure(
|
|
252
|
+
cls,
|
|
253
|
+
endpoint: Optional[str] = None,
|
|
254
|
+
api_key: Optional[str] = None,
|
|
255
|
+
) -> "GuardrailsGateway":
|
|
256
|
+
"""Create a gateway using Azure Content Safety.
|
|
257
|
+
|
|
258
|
+
Args:
|
|
259
|
+
endpoint: Azure endpoint URL. If None, uses env var.
|
|
260
|
+
api_key: Azure API key. If None, uses env var.
|
|
261
|
+
|
|
262
|
+
Returns:
|
|
263
|
+
GuardrailsGateway configured for Azure.
|
|
264
|
+
"""
|
|
265
|
+
import os
|
|
266
|
+
if endpoint:
|
|
267
|
+
os.environ["AZURE_CONTENT_SAFETY_ENDPOINT"] = endpoint
|
|
268
|
+
if api_key:
|
|
269
|
+
os.environ["AZURE_CONTENT_SAFETY_KEY"] = api_key
|
|
270
|
+
|
|
271
|
+
config = GuardrailsConfig(
|
|
272
|
+
models=[GuardrailModel.AZURE_CONTENT_SAFETY],
|
|
273
|
+
timeout_ms=30000,
|
|
274
|
+
)
|
|
275
|
+
return cls(config=config)
|
|
276
|
+
|
|
277
|
+
@classmethod
|
|
278
|
+
def with_local_model(
|
|
279
|
+
cls,
|
|
280
|
+
model: GuardrailModel,
|
|
281
|
+
vllm_url: Optional[str] = None,
|
|
282
|
+
) -> "GuardrailsGateway":
|
|
283
|
+
"""Create a gateway using a local model via VLLM.
|
|
284
|
+
|
|
285
|
+
Args:
|
|
286
|
+
model: The local model to use (e.g., GuardrailModel.WILDGUARD_7B).
|
|
287
|
+
vllm_url: VLLM server URL. If None, uses env var.
|
|
288
|
+
|
|
289
|
+
Returns:
|
|
290
|
+
GuardrailsGateway configured for the local model.
|
|
291
|
+
"""
|
|
292
|
+
import os
|
|
293
|
+
if vllm_url:
|
|
294
|
+
os.environ["VLLM_SERVER_URL"] = vllm_url
|
|
295
|
+
|
|
296
|
+
config = GuardrailsConfig(
|
|
297
|
+
models=[model],
|
|
298
|
+
timeout_ms=60000, # Local models may be slower
|
|
299
|
+
)
|
|
300
|
+
return cls(config=config)
|
|
301
|
+
|
|
302
|
+
@classmethod
|
|
303
|
+
def with_ensemble(
|
|
304
|
+
cls,
|
|
305
|
+
models: List[GuardrailModel],
|
|
306
|
+
aggregation: AggregationStrategy = AggregationStrategy.ANY,
|
|
307
|
+
parallel: bool = True,
|
|
308
|
+
) -> "GuardrailsGateway":
|
|
309
|
+
"""Create a gateway with ensemble of multiple backends.
|
|
310
|
+
|
|
311
|
+
Args:
|
|
312
|
+
models: List of models to use.
|
|
313
|
+
aggregation: How to combine results (ANY, ALL, MAJORITY).
|
|
314
|
+
parallel: Whether to run backends in parallel.
|
|
315
|
+
|
|
316
|
+
Returns:
|
|
317
|
+
GuardrailsGateway configured for ensemble mode.
|
|
318
|
+
"""
|
|
319
|
+
config = GuardrailsConfig(
|
|
320
|
+
models=models,
|
|
321
|
+
aggregation=aggregation,
|
|
322
|
+
parallel=parallel,
|
|
323
|
+
timeout_ms=60000,
|
|
324
|
+
)
|
|
325
|
+
return cls(config=config)
|
|
326
|
+
|
|
327
|
+
@classmethod
|
|
328
|
+
def auto(cls) -> "GuardrailsGateway":
|
|
329
|
+
"""Create a gateway that auto-discovers available backends.
|
|
330
|
+
|
|
331
|
+
Returns:
|
|
332
|
+
GuardrailsGateway with the best available backend.
|
|
333
|
+
|
|
334
|
+
Raises:
|
|
335
|
+
ValueError: If no backends are available.
|
|
336
|
+
"""
|
|
337
|
+
return cls(auto_discover=True)
|
|
338
|
+
|
|
339
|
+
def screen(self, content: str, metadata: Optional[dict] = None) -> GuardrailsResponse:
|
|
340
|
+
"""Quick screen content (alias for screen_input).
|
|
341
|
+
|
|
342
|
+
Args:
|
|
343
|
+
content: Content to screen.
|
|
344
|
+
metadata: Optional metadata.
|
|
345
|
+
|
|
346
|
+
Returns:
|
|
347
|
+
GuardrailsResponse with results.
|
|
348
|
+
"""
|
|
349
|
+
return self._guardrails.screen_input(content, metadata=metadata)
|
|
350
|
+
|
|
351
|
+
async def screen_async(self, content: str, metadata: Optional[dict] = None) -> GuardrailsResponse:
|
|
352
|
+
"""Quick screen content asynchronously.
|
|
353
|
+
|
|
354
|
+
Args:
|
|
355
|
+
content: Content to screen.
|
|
356
|
+
metadata: Optional metadata.
|
|
357
|
+
|
|
358
|
+
Returns:
|
|
359
|
+
GuardrailsResponse with results.
|
|
360
|
+
"""
|
|
361
|
+
return await self._guardrails.screen_input_async(content, metadata=metadata)
|
|
362
|
+
|
|
363
|
+
@contextmanager
|
|
364
|
+
def screening(self) -> Iterator[ScreeningSession]:
|
|
365
|
+
"""Create a synchronous screening session.
|
|
366
|
+
|
|
367
|
+
Yields:
|
|
368
|
+
ScreeningSession for screening operations.
|
|
369
|
+
|
|
370
|
+
Example:
|
|
371
|
+
with gateway.screening() as session:
|
|
372
|
+
input_result = session.input("user message")
|
|
373
|
+
if input_result.passed:
|
|
374
|
+
response = generate_response()
|
|
375
|
+
output_result = session.output(response)
|
|
376
|
+
"""
|
|
377
|
+
session = ScreeningSession(self._guardrails)
|
|
378
|
+
yield session
|
|
379
|
+
|
|
380
|
+
@asynccontextmanager
|
|
381
|
+
async def screening_async(self) -> AsyncIterator[AsyncScreeningSession]:
|
|
382
|
+
"""Create an async screening session.
|
|
383
|
+
|
|
384
|
+
Yields:
|
|
385
|
+
AsyncScreeningSession for async screening operations.
|
|
386
|
+
|
|
387
|
+
Example:
|
|
388
|
+
async with gateway.screening_async() as session:
|
|
389
|
+
input_result = await session.input("user message")
|
|
390
|
+
if input_result.passed:
|
|
391
|
+
response = await generate_response()
|
|
392
|
+
output_result = await session.output(response)
|
|
393
|
+
"""
|
|
394
|
+
session = AsyncScreeningSession(self._guardrails)
|
|
395
|
+
yield session
|
|
396
|
+
|
|
397
|
+
@property
|
|
398
|
+
def available_backends(self) -> List[GuardrailModel]:
|
|
399
|
+
"""Get list of available backends."""
|
|
400
|
+
return discover_backends()
|
|
401
|
+
|
|
402
|
+
@property
|
|
403
|
+
def configured_models(self) -> List[GuardrailModel]:
|
|
404
|
+
"""Get list of configured models."""
|
|
405
|
+
return self._config.models
|
|
406
|
+
|
|
407
|
+
@staticmethod
|
|
408
|
+
def discover() -> List[GuardrailModel]:
|
|
409
|
+
"""Discover available backends.
|
|
410
|
+
|
|
411
|
+
Returns:
|
|
412
|
+
List of available GuardrailModel values.
|
|
413
|
+
"""
|
|
414
|
+
return discover_backends()
|
|
415
|
+
|
|
416
|
+
@staticmethod
|
|
417
|
+
def get_details() -> dict:
|
|
418
|
+
"""Get detailed backend availability info.
|
|
419
|
+
|
|
420
|
+
Returns:
|
|
421
|
+
Dict mapping model names to availability details.
|
|
422
|
+
"""
|
|
423
|
+
return get_backend_details()
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
# Convenience alias
|
|
427
|
+
Gateway = GuardrailsGateway
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
__all__ = [
|
|
431
|
+
"GuardrailsGateway",
|
|
432
|
+
"Gateway",
|
|
433
|
+
"ScreeningSession",
|
|
434
|
+
"AsyncScreeningSession",
|
|
435
|
+
"discover_backends",
|
|
436
|
+
"get_backend_details",
|
|
437
|
+
]
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Model Registry for Guardrails.
|
|
3
|
+
|
|
4
|
+
Central registry for all supported guardrail models with metadata
|
|
5
|
+
about backends, model types, and requirements.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Dict, List, Optional, Type
|
|
10
|
+
|
|
11
|
+
from fi.evals.guardrails.config import GuardrailModel
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class ModelInfo:
|
|
16
|
+
"""Information about a guardrail model."""
|
|
17
|
+
model: GuardrailModel
|
|
18
|
+
backend_class: str # String reference to avoid circular imports
|
|
19
|
+
backend_module: str # Module containing the backend class
|
|
20
|
+
model_type: str # "api", "local", or "vllm"
|
|
21
|
+
hf_model_name: Optional[str] = None # HuggingFace model name
|
|
22
|
+
vram_required_gb: Optional[float] = None # Minimum VRAM in GB
|
|
23
|
+
description: str = ""
|
|
24
|
+
is_gated: bool = False # Requires HF token
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# Central registry of all supported models
|
|
28
|
+
MODEL_REGISTRY: Dict[str, ModelInfo] = {
|
|
29
|
+
# Turing (FutureAGI API)
|
|
30
|
+
"turing_flash": ModelInfo(
|
|
31
|
+
model=GuardrailModel.TURING_FLASH,
|
|
32
|
+
backend_class="TuringBackend",
|
|
33
|
+
backend_module="fi.evals.guardrails.backends.turing",
|
|
34
|
+
model_type="api",
|
|
35
|
+
description="Fast binary classification via FutureAGI API",
|
|
36
|
+
),
|
|
37
|
+
"turing_safety": ModelInfo(
|
|
38
|
+
model=GuardrailModel.TURING_SAFETY,
|
|
39
|
+
backend_class="TuringBackend",
|
|
40
|
+
backend_module="fi.evals.guardrails.backends.turing",
|
|
41
|
+
model_type="api",
|
|
42
|
+
description="Detailed safety analysis via FutureAGI API",
|
|
43
|
+
),
|
|
44
|
+
|
|
45
|
+
# OpenAI (Free API)
|
|
46
|
+
"openai-moderation": ModelInfo(
|
|
47
|
+
model=GuardrailModel.OPENAI_MODERATION,
|
|
48
|
+
backend_class="OpenAIBackend",
|
|
49
|
+
backend_module="fi.evals.guardrails.backends.openai",
|
|
50
|
+
model_type="api",
|
|
51
|
+
description="OpenAI Moderation API (FREE, 13 categories)",
|
|
52
|
+
),
|
|
53
|
+
|
|
54
|
+
# Azure (Paid API)
|
|
55
|
+
"azure-content-safety": ModelInfo(
|
|
56
|
+
model=GuardrailModel.AZURE_CONTENT_SAFETY,
|
|
57
|
+
backend_class="AzureBackend",
|
|
58
|
+
backend_module="fi.evals.guardrails.backends.azure",
|
|
59
|
+
model_type="api",
|
|
60
|
+
description="Azure Content Safety API (4 categories)",
|
|
61
|
+
),
|
|
62
|
+
|
|
63
|
+
# WildGuard (Local)
|
|
64
|
+
"wildguard-7b": ModelInfo(
|
|
65
|
+
model=GuardrailModel.WILDGUARD_7B,
|
|
66
|
+
backend_class="WildGuardBackend",
|
|
67
|
+
backend_module="fi.evals.guardrails.backends.wildguard",
|
|
68
|
+
model_type="local",
|
|
69
|
+
hf_model_name="allenai/wildguard",
|
|
70
|
+
vram_required_gb=8.0,
|
|
71
|
+
description="AllenAI WildGuard safety classifier",
|
|
72
|
+
is_gated=True,
|
|
73
|
+
),
|
|
74
|
+
|
|
75
|
+
# LlamaGuard (Local)
|
|
76
|
+
"llamaguard-3-8b": ModelInfo(
|
|
77
|
+
model=GuardrailModel.LLAMAGUARD_3_8B,
|
|
78
|
+
backend_class="LlamaGuardBackend",
|
|
79
|
+
backend_module="fi.evals.guardrails.backends.llamaguard",
|
|
80
|
+
model_type="local",
|
|
81
|
+
hf_model_name="meta-llama/Llama-Guard-3-8B",
|
|
82
|
+
vram_required_gb=16.0,
|
|
83
|
+
description="Meta LlamaGuard 3 8B safety classifier",
|
|
84
|
+
is_gated=True,
|
|
85
|
+
),
|
|
86
|
+
"llamaguard-3-1b": ModelInfo(
|
|
87
|
+
model=GuardrailModel.LLAMAGUARD_3_1B,
|
|
88
|
+
backend_class="LlamaGuardBackend",
|
|
89
|
+
backend_module="fi.evals.guardrails.backends.llamaguard",
|
|
90
|
+
model_type="local",
|
|
91
|
+
hf_model_name="meta-llama/Llama-Guard-3-1B",
|
|
92
|
+
vram_required_gb=4.0,
|
|
93
|
+
description="Meta LlamaGuard 3 1B safety classifier (lightweight)",
|
|
94
|
+
is_gated=True,
|
|
95
|
+
),
|
|
96
|
+
|
|
97
|
+
# Granite Guardian (Local)
|
|
98
|
+
"granite-guardian-3.3-8b": ModelInfo(
|
|
99
|
+
model=GuardrailModel.GRANITE_GUARDIAN_8B,
|
|
100
|
+
backend_class="GraniteGuardianBackend",
|
|
101
|
+
backend_module="fi.evals.guardrails.backends.granite",
|
|
102
|
+
model_type="local",
|
|
103
|
+
hf_model_name="ibm-granite/granite-guardian-3.3-8b",
|
|
104
|
+
vram_required_gb=16.0,
|
|
105
|
+
description="IBM Granite Guardian 3.3 8B",
|
|
106
|
+
),
|
|
107
|
+
"granite-guardian-3.2-5b": ModelInfo(
|
|
108
|
+
model=GuardrailModel.GRANITE_GUARDIAN_5B,
|
|
109
|
+
backend_class="GraniteGuardianBackend",
|
|
110
|
+
backend_module="fi.evals.guardrails.backends.granite",
|
|
111
|
+
model_type="local",
|
|
112
|
+
hf_model_name="ibm-granite/granite-guardian-3.2-5b",
|
|
113
|
+
vram_required_gb=10.0,
|
|
114
|
+
description="IBM Granite Guardian 3.2 5B (lightweight)",
|
|
115
|
+
),
|
|
116
|
+
|
|
117
|
+
# Qwen3Guard (Local)
|
|
118
|
+
"qwen3guard-8b": ModelInfo(
|
|
119
|
+
model=GuardrailModel.QWEN3GUARD_8B,
|
|
120
|
+
backend_class="Qwen3GuardBackend",
|
|
121
|
+
backend_module="fi.evals.guardrails.backends.qwen",
|
|
122
|
+
model_type="local",
|
|
123
|
+
hf_model_name="Qwen/Qwen3Guard-8B",
|
|
124
|
+
vram_required_gb=16.0,
|
|
125
|
+
description="Alibaba Qwen3Guard 8B (119 languages)",
|
|
126
|
+
),
|
|
127
|
+
"qwen3guard-4b": ModelInfo(
|
|
128
|
+
model=GuardrailModel.QWEN3GUARD_4B,
|
|
129
|
+
backend_class="Qwen3GuardBackend",
|
|
130
|
+
backend_module="fi.evals.guardrails.backends.qwen",
|
|
131
|
+
model_type="local",
|
|
132
|
+
hf_model_name="Qwen/Qwen3Guard-4B",
|
|
133
|
+
vram_required_gb=8.0,
|
|
134
|
+
description="Alibaba Qwen3Guard 4B (lightweight, 119 languages)",
|
|
135
|
+
),
|
|
136
|
+
|
|
137
|
+
"qwen3guard-0.6b": ModelInfo(
|
|
138
|
+
model=GuardrailModel.QWEN3GUARD_0_6B,
|
|
139
|
+
backend_class="Qwen3GuardBackend",
|
|
140
|
+
backend_module="fi.evals.guardrails.backends.qwen",
|
|
141
|
+
model_type="local",
|
|
142
|
+
hf_model_name="Qwen/Qwen3Guard-0.6B",
|
|
143
|
+
vram_required_gb=1.0,
|
|
144
|
+
description="Alibaba Qwen3Guard 0.6B (ultra-lightweight, 119 languages)",
|
|
145
|
+
),
|
|
146
|
+
|
|
147
|
+
# Generic LLM as guard (prompted for safety classification)
|
|
148
|
+
"llama3.2-3b": ModelInfo(
|
|
149
|
+
model=GuardrailModel.LLAMA_3_2_3B,
|
|
150
|
+
backend_class="GenericLLMGuardBackend",
|
|
151
|
+
backend_module="fi.evals.guardrails.backends.generic_llm",
|
|
152
|
+
model_type="local",
|
|
153
|
+
hf_model_name="meta-llama/Llama-3.2-3B-Instruct",
|
|
154
|
+
vram_required_gb=4.0,
|
|
155
|
+
description="Llama 3.2 3B as prompted safety classifier",
|
|
156
|
+
),
|
|
157
|
+
|
|
158
|
+
# ShieldGemma (Local)
|
|
159
|
+
"shieldgemma-2b": ModelInfo(
|
|
160
|
+
model=GuardrailModel.SHIELDGEMMA_2B,
|
|
161
|
+
backend_class="ShieldGemmaBackend",
|
|
162
|
+
backend_module="fi.evals.guardrails.backends.shieldgemma",
|
|
163
|
+
model_type="local",
|
|
164
|
+
hf_model_name="google/shieldgemma-2b",
|
|
165
|
+
vram_required_gb=4.0,
|
|
166
|
+
description="Google ShieldGemma 2B (lightweight, fast)",
|
|
167
|
+
),
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def get_model_info(model: GuardrailModel) -> Optional[ModelInfo]:
|
|
172
|
+
"""
|
|
173
|
+
Get model information from registry.
|
|
174
|
+
|
|
175
|
+
Args:
|
|
176
|
+
model: GuardrailModel enum value
|
|
177
|
+
|
|
178
|
+
Returns:
|
|
179
|
+
ModelInfo or None if not found
|
|
180
|
+
"""
|
|
181
|
+
return MODEL_REGISTRY.get(model.value)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def get_backend_class(model: GuardrailModel) -> Type:
|
|
185
|
+
"""
|
|
186
|
+
Get the backend class for a model.
|
|
187
|
+
|
|
188
|
+
Args:
|
|
189
|
+
model: GuardrailModel enum value
|
|
190
|
+
|
|
191
|
+
Returns:
|
|
192
|
+
Backend class
|
|
193
|
+
|
|
194
|
+
Raises:
|
|
195
|
+
ValueError: If model not found in registry
|
|
196
|
+
ImportError: If backend module cannot be imported
|
|
197
|
+
"""
|
|
198
|
+
info = get_model_info(model)
|
|
199
|
+
if not info:
|
|
200
|
+
raise ValueError(f"Model {model.value} not found in registry")
|
|
201
|
+
|
|
202
|
+
# Dynamic import to avoid circular dependencies
|
|
203
|
+
import importlib
|
|
204
|
+
module = importlib.import_module(info.backend_module)
|
|
205
|
+
return getattr(module, info.backend_class)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def list_models(model_type: Optional[str] = None) -> List[ModelInfo]:
|
|
209
|
+
"""
|
|
210
|
+
List all models in registry.
|
|
211
|
+
|
|
212
|
+
Args:
|
|
213
|
+
model_type: Filter by type ("api", "local", "vllm")
|
|
214
|
+
|
|
215
|
+
Returns:
|
|
216
|
+
List of ModelInfo objects
|
|
217
|
+
"""
|
|
218
|
+
models = list(MODEL_REGISTRY.values())
|
|
219
|
+
if model_type:
|
|
220
|
+
models = [m for m in models if m.model_type == model_type]
|
|
221
|
+
return models
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def list_api_models() -> List[ModelInfo]:
|
|
225
|
+
"""List all API-based models."""
|
|
226
|
+
return list_models(model_type="api")
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def list_local_models() -> List[ModelInfo]:
|
|
230
|
+
"""List all local models."""
|
|
231
|
+
return list_models(model_type="local")
|