agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Malicious URL Scanner for Guardrails.
|
|
3
|
+
|
|
4
|
+
Detects suspicious URLs, phishing attempts, and malicious links.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
import time
|
|
9
|
+
from typing import List, Optional, Set, Tuple
|
|
10
|
+
from urllib.parse import urlparse
|
|
11
|
+
|
|
12
|
+
from fi.evals.guardrails.scanners.base import (
|
|
13
|
+
BaseScanner,
|
|
14
|
+
ScanResult,
|
|
15
|
+
ScanMatch,
|
|
16
|
+
ScannerAction,
|
|
17
|
+
register_scanner,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# URL extraction pattern
|
|
22
|
+
URL_PATTERN = re.compile(
|
|
23
|
+
r'https?://[^\s<>"\']+|'
|
|
24
|
+
r'www\.[^\s<>"\']+|'
|
|
25
|
+
r'[a-zA-Z0-9][-a-zA-Z0-9]*\.[a-zA-Z]{2,}(?:/[^\s<>"\']*)?'
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Suspicious TLDs often used in phishing
|
|
29
|
+
SUSPICIOUS_TLDS: Set[str] = {
|
|
30
|
+
"xyz", "top", "work", "click", "link", "gq", "ml", "cf", "ga", "tk",
|
|
31
|
+
"zip", "mov", "app", "dev", # New TLDs that can be confusing
|
|
32
|
+
"ru", "cn", "su", # Sometimes associated with malicious content
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
# Known URL shorteners
|
|
36
|
+
URL_SHORTENERS: Set[str] = {
|
|
37
|
+
"bit.ly", "tinyurl.com", "t.co", "goo.gl", "ow.ly", "is.gd",
|
|
38
|
+
"buff.ly", "adf.ly", "bc.vc", "j.mp", "tr.im", "tiny.cc",
|
|
39
|
+
"lnkd.in", "db.tt", "qr.ae", "cur.lv", "ity.im", "q.gs",
|
|
40
|
+
"po.st", "su.pr", "rebrand.ly", "shorturl.at",
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
# Legitimate domains that are often spoofed
|
|
44
|
+
SPOOFED_DOMAINS: List[Tuple[str, List[str]]] = [
|
|
45
|
+
("google", ["g00gle", "googie", "gooogle", "google-", "-google"]),
|
|
46
|
+
("facebook", ["faceb00k", "facebok", "faceboook", "facebook-", "-facebook"]),
|
|
47
|
+
("microsoft", ["micros0ft", "mircosoft", "microsoft-", "-microsoft"]),
|
|
48
|
+
("apple", ["app1e", "appie", "apple-", "-apple"]),
|
|
49
|
+
("amazon", ["amaz0n", "amazonn", "amazon-", "-amazon"]),
|
|
50
|
+
("paypal", ["paypa1", "paypai", "paypal-", "-paypal"]),
|
|
51
|
+
("netflix", ["netf1ix", "netiflix", "netflix-", "-netflix"]),
|
|
52
|
+
("linkedin", ["linkedln", "linkedin-", "-linkedin"]),
|
|
53
|
+
("twitter", ["tw1tter", "twitter-", "-twitter"]),
|
|
54
|
+
("instagram", ["lnstagram", "instagram-", "-instagram"]),
|
|
55
|
+
("whatsapp", ["whatsap", "whatsapp-", "-whatsapp"]),
|
|
56
|
+
("dropbox", ["dr0pbox", "dropbox-", "-dropbox"]),
|
|
57
|
+
("github", ["g1thub", "github-", "-github"]),
|
|
58
|
+
("openai", ["0penai", "openal", "openai-", "-openai"]),
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _extract_urls(text: str) -> List[Tuple[str, int, int]]:
|
|
63
|
+
"""Extract URLs from text with positions."""
|
|
64
|
+
urls = []
|
|
65
|
+
for match in URL_PATTERN.finditer(text):
|
|
66
|
+
url = match.group()
|
|
67
|
+
# Add protocol if missing
|
|
68
|
+
if not url.startswith(('http://', 'https://')):
|
|
69
|
+
url = 'https://' + url
|
|
70
|
+
urls.append((url, match.start(), match.end()))
|
|
71
|
+
return urls
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _get_domain(url: str) -> Optional[str]:
|
|
75
|
+
"""Extract domain from URL."""
|
|
76
|
+
try:
|
|
77
|
+
parsed = urlparse(url)
|
|
78
|
+
return parsed.netloc.lower()
|
|
79
|
+
except Exception:
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _check_homoglyph(domain: str) -> Optional[str]:
|
|
84
|
+
"""Check for homoglyph attacks (lookalike characters)."""
|
|
85
|
+
# Common homoglyphs
|
|
86
|
+
homoglyphs = {
|
|
87
|
+
'0': 'o', '1': 'i', '@': 'a',
|
|
88
|
+
'$': 's', '3': 'e', '4': 'a', '5': 's',
|
|
89
|
+
'6': 'b', '7': 't', '8': 'b', '9': 'g',
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
normalized = domain
|
|
93
|
+
for fake, real in homoglyphs.items():
|
|
94
|
+
normalized = normalized.replace(fake, real)
|
|
95
|
+
|
|
96
|
+
if normalized != domain:
|
|
97
|
+
return normalized
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@register_scanner("urls")
|
|
102
|
+
class MaliciousURLScanner(BaseScanner):
|
|
103
|
+
"""
|
|
104
|
+
Scanner for detecting malicious and suspicious URLs.
|
|
105
|
+
|
|
106
|
+
Detects:
|
|
107
|
+
- Phishing URLs (lookalike domains)
|
|
108
|
+
- IP-based URLs
|
|
109
|
+
- Suspicious TLDs
|
|
110
|
+
- URL shorteners (optional)
|
|
111
|
+
- Data URLs with executable content
|
|
112
|
+
- Encoded URLs
|
|
113
|
+
|
|
114
|
+
Usage:
|
|
115
|
+
scanner = MaliciousURLScanner()
|
|
116
|
+
result = scanner.scan("Visit https://g00gle.com/login")
|
|
117
|
+
if not result.passed:
|
|
118
|
+
print(f"Suspicious URL: {result.matched_patterns}")
|
|
119
|
+
"""
|
|
120
|
+
|
|
121
|
+
name = "urls"
|
|
122
|
+
category = "malicious_url"
|
|
123
|
+
description = "Detects phishing URLs and malicious links"
|
|
124
|
+
default_action = ScannerAction.FLAG # Flag rather than block by default
|
|
125
|
+
|
|
126
|
+
def __init__(
|
|
127
|
+
self,
|
|
128
|
+
action: Optional[ScannerAction] = None,
|
|
129
|
+
enabled: bool = True,
|
|
130
|
+
threshold: float = 0.7,
|
|
131
|
+
block_ip_urls: bool = True,
|
|
132
|
+
block_suspicious_tlds: bool = True,
|
|
133
|
+
block_shorteners: bool = False, # Disabled by default
|
|
134
|
+
block_data_urls: bool = True,
|
|
135
|
+
check_homoglyphs: bool = True,
|
|
136
|
+
allowed_domains: Optional[Set[str]] = None,
|
|
137
|
+
blocked_domains: Optional[Set[str]] = None,
|
|
138
|
+
):
|
|
139
|
+
"""
|
|
140
|
+
Initialize URL scanner.
|
|
141
|
+
|
|
142
|
+
Args:
|
|
143
|
+
action: Action on detection
|
|
144
|
+
enabled: Whether scanner is enabled
|
|
145
|
+
threshold: Minimum confidence to trigger
|
|
146
|
+
block_ip_urls: Block URLs with IP addresses
|
|
147
|
+
block_suspicious_tlds: Block URLs with suspicious TLDs
|
|
148
|
+
block_shorteners: Block URL shorteners
|
|
149
|
+
block_data_urls: Block data: URLs
|
|
150
|
+
check_homoglyphs: Check for lookalike domains
|
|
151
|
+
allowed_domains: Whitelist of allowed domains
|
|
152
|
+
blocked_domains: Blacklist of blocked domains
|
|
153
|
+
"""
|
|
154
|
+
super().__init__(action, enabled)
|
|
155
|
+
self.threshold = threshold
|
|
156
|
+
self.block_ip_urls = block_ip_urls
|
|
157
|
+
self.block_suspicious_tlds = block_suspicious_tlds
|
|
158
|
+
self.block_shorteners = block_shorteners
|
|
159
|
+
self.block_data_urls = block_data_urls
|
|
160
|
+
self.check_homoglyphs = check_homoglyphs
|
|
161
|
+
self.allowed_domains = allowed_domains or set()
|
|
162
|
+
self.blocked_domains = blocked_domains or set()
|
|
163
|
+
|
|
164
|
+
def _is_ip_address(self, domain: str) -> bool:
|
|
165
|
+
"""Check if domain is an IP address."""
|
|
166
|
+
# IPv4
|
|
167
|
+
ipv4_pattern = r'^\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}$'
|
|
168
|
+
if re.match(ipv4_pattern, domain):
|
|
169
|
+
return True
|
|
170
|
+
# IPv6
|
|
171
|
+
if ':' in domain and domain.replace(':', '').replace('.', '').isalnum():
|
|
172
|
+
return True
|
|
173
|
+
return False
|
|
174
|
+
|
|
175
|
+
def _check_phishing(self, domain: str) -> Optional[Tuple[str, str]]:
|
|
176
|
+
"""Check if domain is a phishing attempt."""
|
|
177
|
+
domain_lower = domain.lower()
|
|
178
|
+
|
|
179
|
+
for legit, fakes in SPOOFED_DOMAINS:
|
|
180
|
+
for fake in fakes:
|
|
181
|
+
if fake in domain_lower:
|
|
182
|
+
return (legit, fake)
|
|
183
|
+
|
|
184
|
+
# Also check for the legitimate name in suspicious context
|
|
185
|
+
if legit in domain_lower and not domain_lower.endswith(f".{legit}.com"):
|
|
186
|
+
# e.g., google.suspicious.com
|
|
187
|
+
parts = domain_lower.split('.')
|
|
188
|
+
if len(parts) > 2 and legit in parts[0]:
|
|
189
|
+
return (legit, f"{legit} subdomain")
|
|
190
|
+
|
|
191
|
+
return None
|
|
192
|
+
|
|
193
|
+
def scan(self, content: str, context: Optional[str] = None) -> ScanResult:
|
|
194
|
+
"""
|
|
195
|
+
Scan content for malicious URLs.
|
|
196
|
+
|
|
197
|
+
Args:
|
|
198
|
+
content: Content to scan
|
|
199
|
+
context: Optional context
|
|
200
|
+
|
|
201
|
+
Returns:
|
|
202
|
+
ScanResult with detection details
|
|
203
|
+
"""
|
|
204
|
+
start = time.perf_counter()
|
|
205
|
+
matches = []
|
|
206
|
+
max_confidence = 0.0
|
|
207
|
+
issues = set()
|
|
208
|
+
|
|
209
|
+
# Check for data URLs
|
|
210
|
+
if self.block_data_urls:
|
|
211
|
+
data_url_pattern = re.compile(r'data:[^;]+;base64,[a-zA-Z0-9+/=]+', re.IGNORECASE)
|
|
212
|
+
for match in data_url_pattern.finditer(content):
|
|
213
|
+
# Check if it's potentially executable
|
|
214
|
+
url = match.group().lower()
|
|
215
|
+
if any(t in url for t in ['javascript', 'text/html', 'application/']):
|
|
216
|
+
matches.append(ScanMatch(
|
|
217
|
+
pattern_name="data_url_executable",
|
|
218
|
+
matched_text=match.group()[:50] + "...",
|
|
219
|
+
start=match.start(),
|
|
220
|
+
end=match.end(),
|
|
221
|
+
confidence=0.95,
|
|
222
|
+
))
|
|
223
|
+
max_confidence = max(max_confidence, 0.95)
|
|
224
|
+
issues.add("Executable data URL")
|
|
225
|
+
|
|
226
|
+
# Extract and check URLs
|
|
227
|
+
urls = _extract_urls(content)
|
|
228
|
+
|
|
229
|
+
for url, start_pos, end_pos in urls:
|
|
230
|
+
domain = _get_domain(url)
|
|
231
|
+
if not domain:
|
|
232
|
+
continue
|
|
233
|
+
|
|
234
|
+
# Skip allowed domains
|
|
235
|
+
if domain in self.allowed_domains:
|
|
236
|
+
continue
|
|
237
|
+
|
|
238
|
+
# Check blocked domains
|
|
239
|
+
if domain in self.blocked_domains:
|
|
240
|
+
matches.append(ScanMatch(
|
|
241
|
+
pattern_name="blocked_domain",
|
|
242
|
+
matched_text=url,
|
|
243
|
+
start=start_pos,
|
|
244
|
+
end=end_pos,
|
|
245
|
+
confidence=1.0,
|
|
246
|
+
))
|
|
247
|
+
max_confidence = 1.0
|
|
248
|
+
issues.add("Blocked domain")
|
|
249
|
+
continue
|
|
250
|
+
|
|
251
|
+
# Check IP-based URLs
|
|
252
|
+
if self.block_ip_urls and self._is_ip_address(domain.split(':')[0]):
|
|
253
|
+
matches.append(ScanMatch(
|
|
254
|
+
pattern_name="ip_url",
|
|
255
|
+
matched_text=url,
|
|
256
|
+
start=start_pos,
|
|
257
|
+
end=end_pos,
|
|
258
|
+
confidence=0.8,
|
|
259
|
+
))
|
|
260
|
+
max_confidence = max(max_confidence, 0.8)
|
|
261
|
+
issues.add("IP-based URL")
|
|
262
|
+
|
|
263
|
+
# Check suspicious TLDs
|
|
264
|
+
if self.block_suspicious_tlds:
|
|
265
|
+
tld = domain.split('.')[-1]
|
|
266
|
+
if tld in SUSPICIOUS_TLDS:
|
|
267
|
+
matches.append(ScanMatch(
|
|
268
|
+
pattern_name="suspicious_tld",
|
|
269
|
+
matched_text=url,
|
|
270
|
+
start=start_pos,
|
|
271
|
+
end=end_pos,
|
|
272
|
+
confidence=0.6,
|
|
273
|
+
metadata={"tld": tld},
|
|
274
|
+
))
|
|
275
|
+
max_confidence = max(max_confidence, 0.6)
|
|
276
|
+
issues.add(f"Suspicious TLD (.{tld})")
|
|
277
|
+
|
|
278
|
+
# Check URL shorteners
|
|
279
|
+
if self.block_shorteners and domain in URL_SHORTENERS:
|
|
280
|
+
matches.append(ScanMatch(
|
|
281
|
+
pattern_name="url_shortener",
|
|
282
|
+
matched_text=url,
|
|
283
|
+
start=start_pos,
|
|
284
|
+
end=end_pos,
|
|
285
|
+
confidence=0.5,
|
|
286
|
+
))
|
|
287
|
+
max_confidence = max(max_confidence, 0.5)
|
|
288
|
+
issues.add("URL shortener")
|
|
289
|
+
|
|
290
|
+
# Check phishing (lookalike domains)
|
|
291
|
+
if self.check_homoglyphs:
|
|
292
|
+
phishing = self._check_phishing(domain)
|
|
293
|
+
if phishing:
|
|
294
|
+
legit, fake = phishing
|
|
295
|
+
matches.append(ScanMatch(
|
|
296
|
+
pattern_name="phishing_domain",
|
|
297
|
+
matched_text=url,
|
|
298
|
+
start=start_pos,
|
|
299
|
+
end=end_pos,
|
|
300
|
+
confidence=0.9,
|
|
301
|
+
metadata={"spoofed_brand": legit, "technique": fake},
|
|
302
|
+
))
|
|
303
|
+
max_confidence = max(max_confidence, 0.9)
|
|
304
|
+
issues.add(f"Phishing ({legit} lookalike)")
|
|
305
|
+
|
|
306
|
+
# Also check homoglyphs
|
|
307
|
+
normalized = _check_homoglyph(domain)
|
|
308
|
+
if normalized:
|
|
309
|
+
matches.append(ScanMatch(
|
|
310
|
+
pattern_name="homoglyph_attack",
|
|
311
|
+
matched_text=url,
|
|
312
|
+
start=start_pos,
|
|
313
|
+
end=end_pos,
|
|
314
|
+
confidence=0.85,
|
|
315
|
+
metadata={"normalized": normalized},
|
|
316
|
+
))
|
|
317
|
+
max_confidence = max(max_confidence, 0.85)
|
|
318
|
+
issues.add("Homoglyph attack")
|
|
319
|
+
|
|
320
|
+
latency = (time.perf_counter() - start) * 1000
|
|
321
|
+
|
|
322
|
+
# Filter by threshold
|
|
323
|
+
significant_matches = [m for m in matches if m.confidence >= self.threshold]
|
|
324
|
+
|
|
325
|
+
if significant_matches:
|
|
326
|
+
return self._create_result(
|
|
327
|
+
passed=False,
|
|
328
|
+
matches=significant_matches,
|
|
329
|
+
score=max_confidence,
|
|
330
|
+
reason=f"Suspicious URLs detected: {', '.join(issues)}",
|
|
331
|
+
latency_ms=latency,
|
|
332
|
+
metadata={"issues": list(issues)},
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
return self._create_result(
|
|
336
|
+
passed=True,
|
|
337
|
+
matches=[],
|
|
338
|
+
score=0.0,
|
|
339
|
+
reason="No malicious URLs detected",
|
|
340
|
+
latency_ms=latency,
|
|
341
|
+
)
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Guardrails Types Module.
|
|
3
|
+
|
|
4
|
+
Defines response types for the guardrails system:
|
|
5
|
+
- GuardrailResult: Result from a single model
|
|
6
|
+
- GuardrailsResponse: Aggregated response from all models
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import List, Optional
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class GuardrailResult:
|
|
15
|
+
"""Result from a single guardrail model check."""
|
|
16
|
+
|
|
17
|
+
passed: bool
|
|
18
|
+
category: str
|
|
19
|
+
score: float
|
|
20
|
+
model: str
|
|
21
|
+
reason: Optional[str] = None
|
|
22
|
+
action: str = "pass" # "block", "flag", "redact", "warn", "pass"
|
|
23
|
+
latency_ms: float = 0.0
|
|
24
|
+
|
|
25
|
+
def __post_init__(self):
|
|
26
|
+
"""Validate result."""
|
|
27
|
+
if not 0.0 <= self.score <= 1.0:
|
|
28
|
+
# Clamp score to valid range
|
|
29
|
+
self.score = max(0.0, min(1.0, self.score))
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class GuardrailsResponse:
|
|
34
|
+
"""Aggregated response from all guardrails."""
|
|
35
|
+
|
|
36
|
+
passed: bool
|
|
37
|
+
results: List[GuardrailResult] = field(default_factory=list)
|
|
38
|
+
blocked_categories: List[str] = field(default_factory=list)
|
|
39
|
+
flagged_categories: List[str] = field(default_factory=list)
|
|
40
|
+
redacted_content: Optional[str] = None
|
|
41
|
+
original_content: str = ""
|
|
42
|
+
total_latency_ms: float = 0.0
|
|
43
|
+
models_used: List[str] = field(default_factory=list)
|
|
44
|
+
error: Optional[str] = None
|
|
45
|
+
|
|
46
|
+
@classmethod
|
|
47
|
+
def create_passed(
|
|
48
|
+
cls,
|
|
49
|
+
content: str,
|
|
50
|
+
latency_ms: float = 0.0,
|
|
51
|
+
models_used: Optional[List[str]] = None,
|
|
52
|
+
results: Optional[List[GuardrailResult]] = None,
|
|
53
|
+
) -> "GuardrailsResponse":
|
|
54
|
+
"""Create a passed response."""
|
|
55
|
+
return cls(
|
|
56
|
+
passed=True,
|
|
57
|
+
original_content=content,
|
|
58
|
+
total_latency_ms=latency_ms,
|
|
59
|
+
models_used=models_used or [],
|
|
60
|
+
results=results or [],
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
@classmethod
|
|
64
|
+
def create_blocked(
|
|
65
|
+
cls,
|
|
66
|
+
content: str,
|
|
67
|
+
blocked_categories: List[str],
|
|
68
|
+
latency_ms: float = 0.0,
|
|
69
|
+
models_used: Optional[List[str]] = None,
|
|
70
|
+
results: Optional[List[GuardrailResult]] = None,
|
|
71
|
+
reason: Optional[str] = None,
|
|
72
|
+
) -> "GuardrailsResponse":
|
|
73
|
+
"""Create a blocked response."""
|
|
74
|
+
return cls(
|
|
75
|
+
passed=False,
|
|
76
|
+
original_content=content,
|
|
77
|
+
blocked_categories=blocked_categories,
|
|
78
|
+
total_latency_ms=latency_ms,
|
|
79
|
+
models_used=models_used or [],
|
|
80
|
+
results=results or [],
|
|
81
|
+
error=reason,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
@classmethod
|
|
85
|
+
def create_error(
|
|
86
|
+
cls,
|
|
87
|
+
content: str,
|
|
88
|
+
error: str,
|
|
89
|
+
fail_open: bool = False,
|
|
90
|
+
) -> "GuardrailsResponse":
|
|
91
|
+
"""Create an error response."""
|
|
92
|
+
return cls(
|
|
93
|
+
passed=fail_open, # If fail_open, allow content
|
|
94
|
+
original_content=content,
|
|
95
|
+
error=error,
|
|
96
|
+
)
|
fi/evals/llm/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import List, Dict, Any, Optional, Type
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class LLMProvider(ABC):
|
|
8
|
+
"""
|
|
9
|
+
Abstract base class for LLM providers.
|
|
10
|
+
|
|
11
|
+
This defines the standard interface that all LLM inference backends
|
|
12
|
+
must implement to be compatible with the LLM-as-a-judge framework.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
@abstractmethod
|
|
16
|
+
def get_completion(
|
|
17
|
+
self,
|
|
18
|
+
model: str,
|
|
19
|
+
messages: List[Dict[str, Any]],
|
|
20
|
+
response_format: Optional[Type[BaseModel] | Dict[str, str]] = None,
|
|
21
|
+
**kwargs: Any,
|
|
22
|
+
) -> str:
|
|
23
|
+
"""
|
|
24
|
+
Generates a text completion from a list of messages.
|
|
25
|
+
|
|
26
|
+
Args:
|
|
27
|
+
model (str): The name or identifier of the model to use.
|
|
28
|
+
messages (List[Dict[str, Any]]): The chat messages, following the OpenAI format.
|
|
29
|
+
Content can be a string or a list of content parts for multimodal inputs.
|
|
30
|
+
**kwargs: Provider-specific arguments like temperature, max_tokens, etc.
|
|
31
|
+
|
|
32
|
+
Returns:
|
|
33
|
+
str: The content of the generated message.
|
|
34
|
+
"""
|
|
35
|
+
raise NotImplementedError
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from typing import Any, Dict, List, Optional, Type
|
|
2
|
+
import litellm
|
|
3
|
+
from pydantic import BaseModel
|
|
4
|
+
from ..base_llm_provider import LLMProvider
|
|
5
|
+
import openai
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class LiteLLMProvider(LLMProvider):
|
|
9
|
+
"""The default provider, using the litellm library to connect to any API."""
|
|
10
|
+
|
|
11
|
+
def __init__(self, credentials: Optional[Dict[str, Any]] = None):
|
|
12
|
+
"""
|
|
13
|
+
Initializes the LiteLLMProvider.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
credentials (Optional[Dict[str, Any]]): A dictionary containing authentication
|
|
17
|
+
details that map directly to litellm.completion() arguments.
|
|
18
|
+
|
|
19
|
+
Examples:
|
|
20
|
+
- For OpenAI: `{"api_key": "sk-..."}`
|
|
21
|
+
- For Azure: `{"api_key": "...", "api_base": "...", "api_version": "..."}`
|
|
22
|
+
- For other providers, see LiteLLM documentation.
|
|
23
|
+
|
|
24
|
+
If not provided, LiteLLM will fall back to its default behavior
|
|
25
|
+
(i.e., checking for environment variables like OPENAI_API_KEY).
|
|
26
|
+
"""
|
|
27
|
+
self.credentials = credentials or {}
|
|
28
|
+
self.schema_support_cache: Dict[str, bool] = {}
|
|
29
|
+
|
|
30
|
+
def _supports_schema(self, model: str) -> bool:
|
|
31
|
+
"""Checks if a model supports response_format with caching."""
|
|
32
|
+
if model not in self.schema_support_cache:
|
|
33
|
+
try:
|
|
34
|
+
self.schema_support_cache[model] = litellm.supports_response_schema(
|
|
35
|
+
model
|
|
36
|
+
)
|
|
37
|
+
except Exception:
|
|
38
|
+
self.schema_support_cache[model] = False
|
|
39
|
+
return self.schema_support_cache[model]
|
|
40
|
+
|
|
41
|
+
def get_completion(
|
|
42
|
+
self,
|
|
43
|
+
model: str,
|
|
44
|
+
messages: List[Dict[str, Any]],
|
|
45
|
+
response_format: Optional[Type[BaseModel] | Dict[str, str]] = None,
|
|
46
|
+
**kwargs: Any,
|
|
47
|
+
):
|
|
48
|
+
completion_args = {**self.credentials, **kwargs}
|
|
49
|
+
|
|
50
|
+
if response_format and self._supports_schema(model):
|
|
51
|
+
completion_args["response_format"] = response_format
|
|
52
|
+
|
|
53
|
+
try:
|
|
54
|
+
# drop unknown/unsupported params in kwargs
|
|
55
|
+
litellm.drop_params = True
|
|
56
|
+
response = litellm.completion(
|
|
57
|
+
model=model, messages=messages, **completion_args
|
|
58
|
+
)
|
|
59
|
+
content = response.choices[0].message.content
|
|
60
|
+
if content is None:
|
|
61
|
+
raise ValueError("Received null content from the LLM API.")
|
|
62
|
+
return content
|
|
63
|
+
|
|
64
|
+
except openai.APIError as openai_error:
|
|
65
|
+
# Wrap litellm exceptions in a standard error
|
|
66
|
+
raise RuntimeError(
|
|
67
|
+
f"LiteLLM provider failed: {openai_error}"
|
|
68
|
+
) from openai_error
|
|
69
|
+
except Exception as e:
|
|
70
|
+
raise RuntimeError(f"Unknown error occured in LiteLLM :{e}")
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Local execution module for running evaluations without API calls.
|
|
2
|
+
|
|
3
|
+
This module provides the infrastructure for running heuristic metrics locally,
|
|
4
|
+
enabling offline evaluation and faster feedback loops during development.
|
|
5
|
+
|
|
6
|
+
It also supports local LLM inference via Ollama for running LLM-as-judge
|
|
7
|
+
evaluations without cloud API calls.
|
|
8
|
+
|
|
9
|
+
Example:
|
|
10
|
+
>>> from fi.evals.local import LocalEvaluator, ExecutionMode
|
|
11
|
+
>>>
|
|
12
|
+
>>> # Run a metric locally
|
|
13
|
+
>>> evaluator = LocalEvaluator()
|
|
14
|
+
>>> result = evaluator.evaluate(
|
|
15
|
+
... metric_name="contains",
|
|
16
|
+
... inputs=[{"response": "Hello world"}],
|
|
17
|
+
... config={"keyword": "world"}
|
|
18
|
+
... )
|
|
19
|
+
>>> print(result.results.eval_results[0].output)
|
|
20
|
+
1.0
|
|
21
|
+
|
|
22
|
+
>>> # Check if a metric can run locally
|
|
23
|
+
>>> evaluator.can_run_locally("contains") # True
|
|
24
|
+
>>> evaluator.can_run_locally("groundedness") # False (requires LLM)
|
|
25
|
+
|
|
26
|
+
>>> # Use hybrid mode to automatically route
|
|
27
|
+
>>> from fi.evals.local import HybridEvaluator
|
|
28
|
+
>>> hybrid = HybridEvaluator()
|
|
29
|
+
>>> partitions = hybrid.partition_evaluations([
|
|
30
|
+
... {"metric_name": "contains", "inputs": [...]},
|
|
31
|
+
... {"metric_name": "groundedness", "inputs": [...]},
|
|
32
|
+
... ])
|
|
33
|
+
>>> # partitions[ExecutionMode.LOCAL] = [contains eval]
|
|
34
|
+
>>> # partitions[ExecutionMode.CLOUD] = [groundedness eval]
|
|
35
|
+
|
|
36
|
+
>>> # Use local LLM for LLM-based evaluations
|
|
37
|
+
>>> from fi.evals.local import OllamaLLM, HybridEvaluator
|
|
38
|
+
>>> llm = OllamaLLM()
|
|
39
|
+
>>> hybrid = HybridEvaluator(local_llm=llm)
|
|
40
|
+
>>> result = llm.judge(
|
|
41
|
+
... query="What is AI?",
|
|
42
|
+
... response="AI is artificial intelligence.",
|
|
43
|
+
... criteria="Evaluate if the response correctly answers the question."
|
|
44
|
+
... )
|
|
45
|
+
>>> print(result["score"])
|
|
46
|
+
0.9
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
from .execution_mode import (
|
|
50
|
+
RoutingMode,
|
|
51
|
+
LOCAL_CAPABLE_METRICS,
|
|
52
|
+
can_run_locally,
|
|
53
|
+
select_routing_mode,
|
|
54
|
+
)
|
|
55
|
+
from .registry import (
|
|
56
|
+
LocalMetricRegistry,
|
|
57
|
+
get_registry,
|
|
58
|
+
)
|
|
59
|
+
from .evaluator import (
|
|
60
|
+
LocalEvaluator,
|
|
61
|
+
LocalEvaluatorConfig,
|
|
62
|
+
LocalEvaluationResult,
|
|
63
|
+
HybridEvaluator,
|
|
64
|
+
)
|
|
65
|
+
from .llm import (
|
|
66
|
+
LocalLLMConfig,
|
|
67
|
+
OllamaLLM,
|
|
68
|
+
LocalLLMFactory,
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
__all__ = [
|
|
73
|
+
# Routing mode
|
|
74
|
+
"RoutingMode",
|
|
75
|
+
"LOCAL_CAPABLE_METRICS",
|
|
76
|
+
"can_run_locally",
|
|
77
|
+
"select_routing_mode",
|
|
78
|
+
# Registry
|
|
79
|
+
"LocalMetricRegistry",
|
|
80
|
+
"get_registry",
|
|
81
|
+
# Evaluator
|
|
82
|
+
"LocalEvaluator",
|
|
83
|
+
"LocalEvaluatorConfig",
|
|
84
|
+
"LocalEvaluationResult",
|
|
85
|
+
"HybridEvaluator",
|
|
86
|
+
# Local LLM
|
|
87
|
+
"LocalLLMConfig",
|
|
88
|
+
"OllamaLLM",
|
|
89
|
+
"LocalLLMFactory",
|
|
90
|
+
]
|