agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Guardrails Configuration Module.
|
|
3
|
+
|
|
4
|
+
Defines configuration classes for the guardrails system including:
|
|
5
|
+
- GuardrailModel: Enum of supported models
|
|
6
|
+
- RailType: Types of rails (input, output, retrieval)
|
|
7
|
+
- AggregationStrategy: How to combine results from multiple models
|
|
8
|
+
- SafetyCategory: Per-category configuration
|
|
9
|
+
- GuardrailsConfig: Main configuration class
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Dict, List, Optional, Literal, Set
|
|
14
|
+
from enum import Enum
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class GuardrailModel(Enum):
|
|
18
|
+
"""Supported guardrail models."""
|
|
19
|
+
|
|
20
|
+
# Turing Models (FutureAGI API)
|
|
21
|
+
TURING_FLASH = "turing_flash"
|
|
22
|
+
TURING_SAFETY = "turing_safety"
|
|
23
|
+
|
|
24
|
+
# Local Models
|
|
25
|
+
QWEN3GUARD_8B = "qwen3guard-8b"
|
|
26
|
+
QWEN3GUARD_4B = "qwen3guard-4b"
|
|
27
|
+
QWEN3GUARD_0_6B = "qwen3guard-0.6b"
|
|
28
|
+
GRANITE_GUARDIAN_8B = "granite-guardian-3.3-8b"
|
|
29
|
+
GRANITE_GUARDIAN_5B = "granite-guardian-3.2-5b"
|
|
30
|
+
WILDGUARD_7B = "wildguard-7b"
|
|
31
|
+
LLAMAGUARD_3_8B = "llamaguard-3-8b"
|
|
32
|
+
LLAMAGUARD_3_1B = "llamaguard-3-1b"
|
|
33
|
+
SHIELDGEMMA_2B = "shieldgemma-2b"
|
|
34
|
+
|
|
35
|
+
# Generic LLM as guard (any chat model prompted for safety)
|
|
36
|
+
LLAMA_3_2_3B = "llama3.2-3b"
|
|
37
|
+
|
|
38
|
+
# Third-party API Models
|
|
39
|
+
OPENAI_MODERATION = "openai-moderation"
|
|
40
|
+
AZURE_CONTENT_SAFETY = "azure-content-safety"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class RailType(Enum):
|
|
44
|
+
"""Types of rails for screening content."""
|
|
45
|
+
INPUT = "input" # Screen user input before LLM
|
|
46
|
+
OUTPUT = "output" # Screen LLM response before user
|
|
47
|
+
RETRIEVAL = "retrieval" # Screen RAG chunks
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class AggregationStrategy(Enum):
|
|
51
|
+
"""Strategy for combining results from multiple models."""
|
|
52
|
+
ANY = "any" # Fail if ANY model flags
|
|
53
|
+
ALL = "all" # Fail if ALL models flag
|
|
54
|
+
MAJORITY = "majority" # Fail if majority flags
|
|
55
|
+
WEIGHTED = "weighted" # Weighted voting (uses model_weights)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class SafetyCategory:
|
|
60
|
+
"""Configuration for a specific safety category."""
|
|
61
|
+
name: str
|
|
62
|
+
enabled: bool = True
|
|
63
|
+
threshold: float = 0.7
|
|
64
|
+
action: Literal["block", "flag", "redact", "warn"] = "block"
|
|
65
|
+
models: List[GuardrailModel] = field(default_factory=list)
|
|
66
|
+
|
|
67
|
+
def __post_init__(self):
|
|
68
|
+
"""Validate configuration."""
|
|
69
|
+
if not 0.0 <= self.threshold <= 1.0:
|
|
70
|
+
raise ValueError(f"threshold must be between 0.0 and 1.0, got {self.threshold}")
|
|
71
|
+
if self.action not in ("block", "flag", "redact", "warn"):
|
|
72
|
+
raise ValueError(f"action must be one of block, flag, redact, warn, got {self.action}")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class TopicConfig:
|
|
77
|
+
"""Configuration for topic restriction scanner."""
|
|
78
|
+
allowed: List[str] = field(default_factory=list)
|
|
79
|
+
denied: List[str] = field(default_factory=list)
|
|
80
|
+
custom_topics: Dict[str, Set[str]] = field(default_factory=dict)
|
|
81
|
+
min_keyword_matches: int = 2
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@dataclass
|
|
85
|
+
class LanguageConfig:
|
|
86
|
+
"""Configuration for language detection scanner."""
|
|
87
|
+
allowed: List[str] = field(default_factory=list) # e.g., ["en", "es", "fr"]
|
|
88
|
+
blocked: List[str] = field(default_factory=list)
|
|
89
|
+
allowed_scripts: List[str] = field(default_factory=lambda: ["Latin", "Common"])
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass
|
|
93
|
+
class RegexPatternConfig:
|
|
94
|
+
"""Configuration for a custom regex pattern."""
|
|
95
|
+
name: str
|
|
96
|
+
pattern: str
|
|
97
|
+
confidence: float = 0.8
|
|
98
|
+
action: Literal["block", "flag", "redact", "warn"] = "block"
|
|
99
|
+
description: str = ""
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
@dataclass
|
|
103
|
+
class ScannerConfig:
|
|
104
|
+
"""
|
|
105
|
+
Configuration for content scanners.
|
|
106
|
+
|
|
107
|
+
Scanners are lightweight, fast detectors that run before model-based backends.
|
|
108
|
+
They provide quick detection of specific threats like jailbreaks, code injection, etc.
|
|
109
|
+
|
|
110
|
+
Attributes:
|
|
111
|
+
enabled: Master switch for all scanners
|
|
112
|
+
jailbreak: Enable jailbreak detection
|
|
113
|
+
code_injection: Enable SQL/shell injection detection
|
|
114
|
+
secrets: Enable secrets/credential detection
|
|
115
|
+
urls: Enable malicious URL detection
|
|
116
|
+
invisible_chars: Enable invisible character detection
|
|
117
|
+
language: Language restriction config
|
|
118
|
+
topics: Topic restriction config
|
|
119
|
+
regex_patterns: Custom regex patterns
|
|
120
|
+
parallel: Run scanners in parallel
|
|
121
|
+
fail_fast: Stop on first failure
|
|
122
|
+
"""
|
|
123
|
+
enabled: bool = True
|
|
124
|
+
|
|
125
|
+
# Individual scanner toggles
|
|
126
|
+
jailbreak: bool = True
|
|
127
|
+
code_injection: bool = True
|
|
128
|
+
secrets: bool = True
|
|
129
|
+
urls: bool = False # Disabled by default (can be noisy)
|
|
130
|
+
invisible_chars: bool = False # Disabled by default
|
|
131
|
+
language: Optional[LanguageConfig] = None
|
|
132
|
+
topics: Optional[TopicConfig] = None
|
|
133
|
+
|
|
134
|
+
# Custom regex patterns
|
|
135
|
+
regex_patterns: List[RegexPatternConfig] = field(default_factory=list)
|
|
136
|
+
predefined_patterns: List[str] = field(default_factory=list) # e.g., ["credit_card", "ssn"]
|
|
137
|
+
|
|
138
|
+
# Performance settings
|
|
139
|
+
parallel: bool = True
|
|
140
|
+
fail_fast: bool = True
|
|
141
|
+
|
|
142
|
+
# Thresholds
|
|
143
|
+
jailbreak_threshold: float = 0.7
|
|
144
|
+
code_injection_threshold: float = 0.7
|
|
145
|
+
secrets_threshold: float = 0.7
|
|
146
|
+
urls_threshold: float = 0.7
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@dataclass
|
|
150
|
+
class GuardrailsConfig:
|
|
151
|
+
"""
|
|
152
|
+
Main configuration for the guardrails system.
|
|
153
|
+
|
|
154
|
+
Attributes:
|
|
155
|
+
models: List of models to use for screening
|
|
156
|
+
rails: Types of rails to enable
|
|
157
|
+
aggregation: How to combine results from multiple models
|
|
158
|
+
categories: Per-category configuration
|
|
159
|
+
timeout_ms: Timeout for each model in milliseconds
|
|
160
|
+
parallel: Whether to run models in parallel
|
|
161
|
+
max_workers: Maximum parallel workers
|
|
162
|
+
fail_open: If True, allow content when guardrails fail
|
|
163
|
+
fallback_model: Model to use if primary fails
|
|
164
|
+
"""
|
|
165
|
+
|
|
166
|
+
# Model selection
|
|
167
|
+
models: List[GuardrailModel] = field(default_factory=lambda: [
|
|
168
|
+
GuardrailModel.TURING_FLASH,
|
|
169
|
+
])
|
|
170
|
+
|
|
171
|
+
# Rail types to enable
|
|
172
|
+
rails: List[RailType] = field(default_factory=lambda: [
|
|
173
|
+
RailType.INPUT,
|
|
174
|
+
RailType.OUTPUT,
|
|
175
|
+
])
|
|
176
|
+
|
|
177
|
+
# Aggregation strategy for ensemble
|
|
178
|
+
aggregation: AggregationStrategy = AggregationStrategy.ANY
|
|
179
|
+
|
|
180
|
+
# Category-specific configurations
|
|
181
|
+
categories: Dict[str, SafetyCategory] = field(default_factory=lambda: {
|
|
182
|
+
"toxicity": SafetyCategory(name="toxicity", threshold=0.7),
|
|
183
|
+
"hate_speech": SafetyCategory(name="hate_speech", threshold=0.7),
|
|
184
|
+
"violence": SafetyCategory(name="violence", threshold=0.8),
|
|
185
|
+
"sexual_content": SafetyCategory(name="sexual_content", threshold=0.8),
|
|
186
|
+
"self_harm": SafetyCategory(name="self_harm", threshold=0.6, action="block"),
|
|
187
|
+
"prompt_injection": SafetyCategory(name="prompt_injection", threshold=0.8),
|
|
188
|
+
"jailbreak": SafetyCategory(name="jailbreak", threshold=0.7),
|
|
189
|
+
"pii": SafetyCategory(name="pii", action="redact"),
|
|
190
|
+
"harmful_content": SafetyCategory(name="harmful_content", threshold=0.7),
|
|
191
|
+
"harassment": SafetyCategory(name="harassment", threshold=0.7),
|
|
192
|
+
"fraud": SafetyCategory(name="fraud", threshold=0.8),
|
|
193
|
+
"illegal_activity": SafetyCategory(name="illegal_activity", threshold=0.8),
|
|
194
|
+
})
|
|
195
|
+
|
|
196
|
+
# Performance settings
|
|
197
|
+
timeout_ms: int = 1000
|
|
198
|
+
parallel: bool = True
|
|
199
|
+
max_workers: int = 5
|
|
200
|
+
|
|
201
|
+
# Weighted aggregation — maps model value to weight (default 1.0 for unlisted)
|
|
202
|
+
# Example: {"turing_flash": 2.0, "openai-moderation": 1.0}
|
|
203
|
+
# weighted_threshold: fraction of total weight needed to block (0.5 = majority)
|
|
204
|
+
model_weights: Dict[str, float] = field(default_factory=dict)
|
|
205
|
+
weighted_threshold: float = 0.5
|
|
206
|
+
|
|
207
|
+
# Fallback behavior
|
|
208
|
+
fail_open: bool = False
|
|
209
|
+
fallback_model: Optional[GuardrailModel] = None
|
|
210
|
+
|
|
211
|
+
# Scanner configuration
|
|
212
|
+
scanners: Optional[ScannerConfig] = None
|
|
213
|
+
|
|
214
|
+
def __post_init__(self):
|
|
215
|
+
"""Validate configuration."""
|
|
216
|
+
if not self.models:
|
|
217
|
+
raise ValueError("At least one model must be specified")
|
|
218
|
+
if self.timeout_ms <= 0:
|
|
219
|
+
raise ValueError(f"timeout_ms must be positive, got {self.timeout_ms}")
|
|
220
|
+
if self.max_workers <= 0:
|
|
221
|
+
raise ValueError(f"max_workers must be positive, got {self.max_workers}")
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Backend Discovery for Guardrails.
|
|
3
|
+
|
|
4
|
+
Auto-detects available backends based on environment variables,
|
|
5
|
+
API keys, and hardware capabilities.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from typing import Dict, List, Optional, Tuple
|
|
10
|
+
|
|
11
|
+
from fi.evals.guardrails.config import GuardrailModel
|
|
12
|
+
from fi.evals.guardrails.registry import MODEL_REGISTRY
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class BackendDiscovery:
|
|
16
|
+
"""
|
|
17
|
+
Auto-discover available guardrail backends.
|
|
18
|
+
|
|
19
|
+
Checks for:
|
|
20
|
+
- API keys (OpenAI, Azure, FutureAGI)
|
|
21
|
+
- VLLM servers (via environment variables)
|
|
22
|
+
- GPU availability for local models
|
|
23
|
+
|
|
24
|
+
Usage:
|
|
25
|
+
discovery = BackendDiscovery()
|
|
26
|
+
available = discovery.discover()
|
|
27
|
+
print(f"Available backends: {[m.value for m in available]}")
|
|
28
|
+
|
|
29
|
+
# Get details
|
|
30
|
+
details = discovery.get_availability_details()
|
|
31
|
+
for model, info in details.items():
|
|
32
|
+
print(f"{model}: {info['status']} - {info['reason']}")
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def __init__(self):
|
|
36
|
+
"""Initialize discovery."""
|
|
37
|
+
self._cache: Optional[List[GuardrailModel]] = None
|
|
38
|
+
self._details_cache: Optional[Dict[str, Dict]] = None
|
|
39
|
+
|
|
40
|
+
def discover(self, force_refresh: bool = False) -> List[GuardrailModel]:
|
|
41
|
+
"""
|
|
42
|
+
Discover available backends.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
force_refresh: Bypass cache and re-check
|
|
46
|
+
|
|
47
|
+
Returns:
|
|
48
|
+
List of available GuardrailModel values
|
|
49
|
+
"""
|
|
50
|
+
if self._cache is not None and not force_refresh:
|
|
51
|
+
return self._cache
|
|
52
|
+
|
|
53
|
+
available = []
|
|
54
|
+
|
|
55
|
+
# Check API backends
|
|
56
|
+
if self._check_fi_credentials():
|
|
57
|
+
available.append(GuardrailModel.TURING_FLASH)
|
|
58
|
+
available.append(GuardrailModel.TURING_SAFETY)
|
|
59
|
+
|
|
60
|
+
if self._check_openai_key():
|
|
61
|
+
available.append(GuardrailModel.OPENAI_MODERATION)
|
|
62
|
+
|
|
63
|
+
if self._check_azure_credentials():
|
|
64
|
+
available.append(GuardrailModel.AZURE_CONTENT_SAFETY)
|
|
65
|
+
|
|
66
|
+
# Check local models via VLLM servers
|
|
67
|
+
for model_value, info in MODEL_REGISTRY.items():
|
|
68
|
+
if info.model_type == "local":
|
|
69
|
+
vllm_url = self._get_vllm_url(model_value)
|
|
70
|
+
if vllm_url and self._check_vllm_health(vllm_url):
|
|
71
|
+
available.append(info.model)
|
|
72
|
+
|
|
73
|
+
self._cache = available
|
|
74
|
+
return available
|
|
75
|
+
|
|
76
|
+
def get_availability_details(self) -> Dict[str, Dict]:
|
|
77
|
+
"""
|
|
78
|
+
Get detailed availability information for all models.
|
|
79
|
+
|
|
80
|
+
Returns:
|
|
81
|
+
Dict mapping model names to availability details
|
|
82
|
+
"""
|
|
83
|
+
if self._details_cache is not None:
|
|
84
|
+
return self._details_cache
|
|
85
|
+
|
|
86
|
+
details = {}
|
|
87
|
+
|
|
88
|
+
for model_value, info in MODEL_REGISTRY.items():
|
|
89
|
+
status = "unavailable"
|
|
90
|
+
reason = ""
|
|
91
|
+
|
|
92
|
+
if info.model_type == "api":
|
|
93
|
+
if model_value.startswith("turing"):
|
|
94
|
+
if self._check_fi_credentials():
|
|
95
|
+
status = "available"
|
|
96
|
+
reason = "FutureAGI credentials found"
|
|
97
|
+
else:
|
|
98
|
+
reason = "Missing FI_API_KEY or FI_SECRET_KEY"
|
|
99
|
+
elif model_value == "openai-moderation":
|
|
100
|
+
if self._check_openai_key():
|
|
101
|
+
status = "available"
|
|
102
|
+
reason = "OPENAI_API_KEY found"
|
|
103
|
+
else:
|
|
104
|
+
reason = "Missing OPENAI_API_KEY"
|
|
105
|
+
elif model_value == "azure-content-safety":
|
|
106
|
+
if self._check_azure_credentials():
|
|
107
|
+
status = "available"
|
|
108
|
+
reason = "Azure credentials found"
|
|
109
|
+
else:
|
|
110
|
+
reason = "Missing AZURE_CONTENT_SAFETY_ENDPOINT or AZURE_CONTENT_SAFETY_KEY"
|
|
111
|
+
|
|
112
|
+
elif info.model_type == "local":
|
|
113
|
+
vllm_url = self._get_vllm_url(model_value)
|
|
114
|
+
if vllm_url:
|
|
115
|
+
if self._check_vllm_health(vllm_url):
|
|
116
|
+
status = "available"
|
|
117
|
+
reason = f"VLLM server at {vllm_url}"
|
|
118
|
+
else:
|
|
119
|
+
status = "unavailable"
|
|
120
|
+
reason = f"VLLM server at {vllm_url} not responding"
|
|
121
|
+
else:
|
|
122
|
+
gpu_status = self._check_gpu_available()
|
|
123
|
+
if gpu_status[0]:
|
|
124
|
+
vram = gpu_status[1]
|
|
125
|
+
if info.vram_required_gb and vram and vram >= info.vram_required_gb:
|
|
126
|
+
status = "available"
|
|
127
|
+
reason = f"GPU available ({vram:.1f}GB VRAM)"
|
|
128
|
+
elif info.vram_required_gb:
|
|
129
|
+
status = "unavailable"
|
|
130
|
+
reason = f"Insufficient VRAM ({vram:.1f}GB < {info.vram_required_gb}GB required)"
|
|
131
|
+
else:
|
|
132
|
+
status = "available"
|
|
133
|
+
reason = "GPU available"
|
|
134
|
+
else:
|
|
135
|
+
status = "unavailable"
|
|
136
|
+
reason = "No GPU available and no VLLM server configured"
|
|
137
|
+
|
|
138
|
+
details[model_value] = {
|
|
139
|
+
"status": status,
|
|
140
|
+
"reason": reason,
|
|
141
|
+
"model_type": info.model_type,
|
|
142
|
+
"description": info.description,
|
|
143
|
+
"hf_model": info.hf_model_name,
|
|
144
|
+
"vram_required": info.vram_required_gb,
|
|
145
|
+
"is_gated": info.is_gated,
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
self._details_cache = details
|
|
149
|
+
return details
|
|
150
|
+
|
|
151
|
+
def _check_fi_credentials(self) -> bool:
|
|
152
|
+
"""Check if FutureAGI credentials are available."""
|
|
153
|
+
return bool(
|
|
154
|
+
os.environ.get("FI_API_KEY") and
|
|
155
|
+
os.environ.get("FI_SECRET_KEY")
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
def _check_openai_key(self) -> bool:
|
|
159
|
+
"""Check if OpenAI API key is available."""
|
|
160
|
+
return bool(os.environ.get("OPENAI_API_KEY"))
|
|
161
|
+
|
|
162
|
+
def _check_azure_credentials(self) -> bool:
|
|
163
|
+
"""Check if Azure credentials are available."""
|
|
164
|
+
return bool(
|
|
165
|
+
os.environ.get("AZURE_CONTENT_SAFETY_ENDPOINT") and
|
|
166
|
+
os.environ.get("AZURE_CONTENT_SAFETY_KEY")
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
def _get_vllm_url(self, model_value: str) -> Optional[str]:
|
|
170
|
+
"""Get VLLM server URL for a model."""
|
|
171
|
+
# Try model-specific env var
|
|
172
|
+
env_var = f"VLLM_{model_value.upper().replace('-', '_')}_URL"
|
|
173
|
+
url = os.environ.get(env_var)
|
|
174
|
+
if url:
|
|
175
|
+
return url
|
|
176
|
+
|
|
177
|
+
# Fall back to generic VLLM_SERVER_URL
|
|
178
|
+
return os.environ.get("VLLM_SERVER_URL")
|
|
179
|
+
|
|
180
|
+
def _check_vllm_health(self, url: str) -> bool:
|
|
181
|
+
"""Check if VLLM server is healthy."""
|
|
182
|
+
try:
|
|
183
|
+
import httpx
|
|
184
|
+
base = url.rstrip('/')
|
|
185
|
+
with httpx.Client(timeout=5.0) as client:
|
|
186
|
+
# Try /health (VLLM), fall back to / (ollama)
|
|
187
|
+
response = client.get(f"{base}/health")
|
|
188
|
+
if response.status_code == 200:
|
|
189
|
+
return True
|
|
190
|
+
response = client.get(base)
|
|
191
|
+
return response.status_code == 200
|
|
192
|
+
except Exception:
|
|
193
|
+
return False
|
|
194
|
+
|
|
195
|
+
def _check_gpu_available(self) -> Tuple[bool, Optional[float]]:
|
|
196
|
+
"""
|
|
197
|
+
Check if GPU is available and get VRAM.
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
Tuple of (is_available, vram_gb)
|
|
201
|
+
"""
|
|
202
|
+
try:
|
|
203
|
+
import torch
|
|
204
|
+
|
|
205
|
+
if torch.cuda.is_available():
|
|
206
|
+
vram = torch.cuda.get_device_properties(0).total_memory / (1024**3)
|
|
207
|
+
return (True, vram)
|
|
208
|
+
|
|
209
|
+
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
|
210
|
+
# MPS doesn't report VRAM, estimate based on typical Apple Silicon
|
|
211
|
+
return (True, 16.0) # Conservative estimate
|
|
212
|
+
|
|
213
|
+
except ImportError:
|
|
214
|
+
pass
|
|
215
|
+
|
|
216
|
+
return (False, None)
|
|
217
|
+
|
|
218
|
+
def _check_hf_token(self) -> bool:
|
|
219
|
+
"""Check if HuggingFace token is available."""
|
|
220
|
+
return bool(
|
|
221
|
+
os.environ.get("HF_TOKEN") or
|
|
222
|
+
os.environ.get("HUGGING_FACE_HUB_TOKEN")
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def discover_backends() -> List[GuardrailModel]:
|
|
227
|
+
"""
|
|
228
|
+
Convenience function to discover available backends.
|
|
229
|
+
|
|
230
|
+
Returns:
|
|
231
|
+
List of available GuardrailModel values
|
|
232
|
+
"""
|
|
233
|
+
return BackendDiscovery().discover()
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def get_backend_details() -> Dict[str, Dict]:
|
|
237
|
+
"""
|
|
238
|
+
Convenience function to get detailed availability info.
|
|
239
|
+
|
|
240
|
+
Returns:
|
|
241
|
+
Dict mapping model names to availability details
|
|
242
|
+
"""
|
|
243
|
+
return BackendDiscovery().get_availability_details()
|