agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""LLM prompt templates for AutoEval."""
|
|
2
|
+
|
|
3
|
+
ANALYSIS_SYSTEM_PROMPT = """You are an AI evaluation expert. Your task is to analyze application descriptions and recommend appropriate evaluations and safety scanners.
|
|
4
|
+
|
|
5
|
+
You must respond in valid JSON format with the following structure:
|
|
6
|
+
{
|
|
7
|
+
"category": "<category>",
|
|
8
|
+
"risk_level": "<risk_level>",
|
|
9
|
+
"domain_sensitivity": "<sensitivity>",
|
|
10
|
+
"detected_features": ["<feature1>", "<feature2>"],
|
|
11
|
+
"requirements": [
|
|
12
|
+
{
|
|
13
|
+
"category": "<requirement_category>",
|
|
14
|
+
"importance": "<importance>",
|
|
15
|
+
"reason": "<explanation>",
|
|
16
|
+
"suggested_evals": ["<eval1>", "<eval2>"],
|
|
17
|
+
"suggested_scanners": ["<scanner1>", "<scanner2>"]
|
|
18
|
+
}
|
|
19
|
+
],
|
|
20
|
+
"explanation": "<brief explanation of the analysis>"
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
## Valid Values
|
|
24
|
+
|
|
25
|
+
### Categories
|
|
26
|
+
- customer_support: Customer service chatbots, help desks
|
|
27
|
+
- rag_system: Retrieval-augmented generation, document Q&A
|
|
28
|
+
- code_assistant: Code generation, debugging, review
|
|
29
|
+
- content_moderation: Content filtering, safety systems
|
|
30
|
+
- agent_workflow: Autonomous agents with tool use
|
|
31
|
+
- chatbot: General conversational AI
|
|
32
|
+
- summarization: Text summarization
|
|
33
|
+
- translation: Language translation
|
|
34
|
+
- creative_writing: Content generation, copywriting
|
|
35
|
+
- data_extraction: Information extraction, parsing
|
|
36
|
+
- search: Search systems
|
|
37
|
+
- question_answering: Q&A systems
|
|
38
|
+
- unknown: Cannot determine
|
|
39
|
+
|
|
40
|
+
### Risk Levels
|
|
41
|
+
- low: Internal tools, development, testing (threshold: 0.6)
|
|
42
|
+
- medium: General public-facing applications (threshold: 0.7)
|
|
43
|
+
- high: Healthcare, finance, legal domains (threshold: 0.8)
|
|
44
|
+
- critical: Safety-critical systems (threshold: 0.9)
|
|
45
|
+
|
|
46
|
+
### Domain Sensitivity
|
|
47
|
+
- general: No special sensitivity
|
|
48
|
+
- pii_sensitive: Handles personal information
|
|
49
|
+
- financial: Banking, payments, investments
|
|
50
|
+
- healthcare: Medical, patient data, HIPAA
|
|
51
|
+
- legal: Legal documents, contracts
|
|
52
|
+
- children: Content for minors, COPPA
|
|
53
|
+
- government: Government/public sector
|
|
54
|
+
|
|
55
|
+
### Importance
|
|
56
|
+
- required: Must have for the application to be safe/functional
|
|
57
|
+
- recommended: Strongly advised but not mandatory
|
|
58
|
+
- optional: Nice to have
|
|
59
|
+
|
|
60
|
+
## Available Evaluations
|
|
61
|
+
|
|
62
|
+
### Semantic Evaluations
|
|
63
|
+
- CoherenceEval: Checks response coherence and logical flow
|
|
64
|
+
|
|
65
|
+
### Agentic Evaluations
|
|
66
|
+
- ActionSafetyEval: Detects unsafe agent actions
|
|
67
|
+
- ReasoningQualityEval: Evaluates reasoning quality
|
|
68
|
+
|
|
69
|
+
## Available Scanners
|
|
70
|
+
|
|
71
|
+
### Security Scanners
|
|
72
|
+
- jailbreak: Detects jailbreak and prompt manipulation attempts
|
|
73
|
+
- code_injection: Detects SQL, shell, path traversal injection
|
|
74
|
+
- secrets: Detects leaked API keys, passwords, credentials
|
|
75
|
+
- prompt_injection: Detects prompt injection attacks
|
|
76
|
+
|
|
77
|
+
### Safety Scanners
|
|
78
|
+
- toxicity: Detects toxic, harmful content
|
|
79
|
+
- bias: Detects biased content
|
|
80
|
+
- pii: Detects personally identifiable information
|
|
81
|
+
|
|
82
|
+
### Content Scanners
|
|
83
|
+
- malicious_url: Detects phishing and suspicious URLs
|
|
84
|
+
- invisible_chars: Detects Unicode manipulation
|
|
85
|
+
- language: Validates language requirements
|
|
86
|
+
- topic_restriction: Enforces topic boundaries
|
|
87
|
+
|
|
88
|
+
## Guidelines
|
|
89
|
+
|
|
90
|
+
1. Match the app type to the most specific category
|
|
91
|
+
2. Consider the domain - healthcare, finance, legal are high-risk
|
|
92
|
+
3. Recommend evaluations based on the app's primary function
|
|
93
|
+
4. Recommend scanners based on safety requirements
|
|
94
|
+
5. Higher risk = higher thresholds and more scanners
|
|
95
|
+
6. Always include basic safety scanners for public-facing apps
|
|
96
|
+
7. For RAG systems, always include faithfulness evaluations
|
|
97
|
+
8. For agents, always include tool use and safety evaluations"""
|
|
98
|
+
|
|
99
|
+
ANALYSIS_USER_PROMPT = """Analyze this application description and recommend evaluations and scanners:
|
|
100
|
+
|
|
101
|
+
{description}
|
|
102
|
+
|
|
103
|
+
Consider:
|
|
104
|
+
1. What type of application is this?
|
|
105
|
+
2. What are the risk factors?
|
|
106
|
+
3. What domain sensitivity applies?
|
|
107
|
+
4. What evaluations would ensure quality?
|
|
108
|
+
5. What scanners would ensure safety?
|
|
109
|
+
|
|
110
|
+
Respond with valid JSON only, no additional text."""
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
CLARIFICATION_QUESTIONS_PROMPT = """Based on the application description, generate clarifying questions to better configure the evaluation pipeline.
|
|
114
|
+
|
|
115
|
+
Application Description:
|
|
116
|
+
{description}
|
|
117
|
+
|
|
118
|
+
Current Analysis:
|
|
119
|
+
- Category: {category}
|
|
120
|
+
- Risk Level: {risk_level}
|
|
121
|
+
- Domain Sensitivity: {domain_sensitivity}
|
|
122
|
+
- Confidence: {confidence}
|
|
123
|
+
|
|
124
|
+
Generate 1-3 clarifying questions that would help improve the evaluation configuration. Focus on:
|
|
125
|
+
1. Deployment environment (internal vs production)
|
|
126
|
+
2. Data sensitivity (PII, financial, health data)
|
|
127
|
+
3. User base (general public, employees, children)
|
|
128
|
+
4. Special requirements (compliance, real-time)
|
|
129
|
+
|
|
130
|
+
Respond in JSON format:
|
|
131
|
+
{
|
|
132
|
+
"questions": [
|
|
133
|
+
{
|
|
134
|
+
"question": "<question text>",
|
|
135
|
+
"options": ["<option1>", "<option2>", "<option3>"],
|
|
136
|
+
"impact": "<what this answer affects>"
|
|
137
|
+
}
|
|
138
|
+
]
|
|
139
|
+
}"""
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Evaluation recommender for AutoEval.
|
|
2
|
+
|
|
3
|
+
Maps application requirements to specific evaluations and scanners.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from typing import List, Dict, Tuple, Set
|
|
7
|
+
from .types import AppAnalysis, RiskLevel, DomainSensitivity
|
|
8
|
+
from .config import EvalConfig, ScannerConfig
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# Mapping from requirement eval names to core metric names
|
|
12
|
+
EVAL_MAPPINGS: Dict[str, str] = {
|
|
13
|
+
# Quality / Relevancy
|
|
14
|
+
"coherence": "answer_relevancy",
|
|
15
|
+
"CoherenceEval": "answer_relevancy",
|
|
16
|
+
"answer_relevancy": "answer_relevancy",
|
|
17
|
+
# RAG / Faithfulness
|
|
18
|
+
"faithfulness": "faithfulness",
|
|
19
|
+
"groundedness": "groundedness",
|
|
20
|
+
"rag_faithfulness": "rag_faithfulness",
|
|
21
|
+
"context_utilization": "context_utilization",
|
|
22
|
+
# Hallucination
|
|
23
|
+
"hallucination": "hallucination_score",
|
|
24
|
+
"factual_consistency": "factual_consistency",
|
|
25
|
+
# Agent
|
|
26
|
+
"action_safety": "action_safety",
|
|
27
|
+
"ActionSafetyEval": "action_safety",
|
|
28
|
+
"reasoning_quality": "reasoning_quality",
|
|
29
|
+
"ReasoningQualityEval": "reasoning_quality",
|
|
30
|
+
"trajectory_score": "trajectory_score",
|
|
31
|
+
"task_completion": "task_completion",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
# Mapping from requirement scanner names to actual scanner names
|
|
35
|
+
SCANNER_MAPPINGS: Dict[str, str] = {
|
|
36
|
+
# Security scanners
|
|
37
|
+
"jailbreak": "JailbreakScanner",
|
|
38
|
+
"JailbreakScanner": "JailbreakScanner",
|
|
39
|
+
"code_injection": "CodeInjectionScanner",
|
|
40
|
+
"CodeInjectionScanner": "CodeInjectionScanner",
|
|
41
|
+
"secrets": "SecretsScanner",
|
|
42
|
+
"SecretsScanner": "SecretsScanner",
|
|
43
|
+
"prompt_injection": "JailbreakScanner", # Use jailbreak scanner for prompt injection
|
|
44
|
+
# Safety scanners (via EvalDelegateScanner)
|
|
45
|
+
"toxicity": "ToxicityScanner",
|
|
46
|
+
"ToxicityScanner": "ToxicityScanner",
|
|
47
|
+
"bias": "BiasScanner",
|
|
48
|
+
"BiasScanner": "BiasScanner",
|
|
49
|
+
"pii": "PIIScanner",
|
|
50
|
+
"PIIScanner": "PIIScanner",
|
|
51
|
+
# Content scanners
|
|
52
|
+
"malicious_url": "MaliciousURLScanner",
|
|
53
|
+
"MaliciousURLScanner": "MaliciousURLScanner",
|
|
54
|
+
"invisible_chars": "InvisibleCharScanner",
|
|
55
|
+
"InvisibleCharScanner": "InvisibleCharScanner",
|
|
56
|
+
"language": "LanguageScanner",
|
|
57
|
+
"LanguageScanner": "LanguageScanner",
|
|
58
|
+
"topic_restriction": "TopicRestrictionScanner",
|
|
59
|
+
"TopicRestrictionScanner": "TopicRestrictionScanner",
|
|
60
|
+
"regex": "RegexScanner",
|
|
61
|
+
"RegexScanner": "RegexScanner",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
# Risk-based thresholds
|
|
65
|
+
RISK_THRESHOLDS: Dict[RiskLevel, float] = {
|
|
66
|
+
RiskLevel.LOW: 0.6,
|
|
67
|
+
RiskLevel.MEDIUM: 0.7,
|
|
68
|
+
RiskLevel.HIGH: 0.8,
|
|
69
|
+
RiskLevel.CRITICAL: 0.9,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
# Scanner actions by importance
|
|
73
|
+
SCANNER_ACTIONS: Dict[str, str] = {
|
|
74
|
+
"required": "block",
|
|
75
|
+
"recommended": "flag",
|
|
76
|
+
"optional": "warn",
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class EvalRecommender:
|
|
81
|
+
"""
|
|
82
|
+
Maps application requirements to specific evaluations and scanners.
|
|
83
|
+
|
|
84
|
+
Example:
|
|
85
|
+
recommender = EvalRecommender()
|
|
86
|
+
evals, scanners = recommender.recommend(analysis)
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
def __init__(self):
|
|
90
|
+
"""Initialize the recommender."""
|
|
91
|
+
self.eval_mappings = EVAL_MAPPINGS
|
|
92
|
+
self.scanner_mappings = SCANNER_MAPPINGS
|
|
93
|
+
|
|
94
|
+
def recommend(
|
|
95
|
+
self,
|
|
96
|
+
analysis: AppAnalysis,
|
|
97
|
+
) -> Tuple[List[EvalConfig], List[ScannerConfig]]:
|
|
98
|
+
"""
|
|
99
|
+
Generate evaluation and scanner recommendations.
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
analysis: Result from AppAnalyzer
|
|
103
|
+
|
|
104
|
+
Returns:
|
|
105
|
+
Tuple of (eval configs, scanner configs)
|
|
106
|
+
"""
|
|
107
|
+
evals: List[EvalConfig] = []
|
|
108
|
+
scanners: List[ScannerConfig] = []
|
|
109
|
+
|
|
110
|
+
# Base threshold from risk level
|
|
111
|
+
base_threshold = RISK_THRESHOLDS.get(analysis.risk_level, 0.7)
|
|
112
|
+
|
|
113
|
+
# Track what we've added to avoid duplicates
|
|
114
|
+
added_evals: Set[str] = set()
|
|
115
|
+
added_scanners: Set[str] = set()
|
|
116
|
+
|
|
117
|
+
# Process each requirement
|
|
118
|
+
for req in analysis.requirements:
|
|
119
|
+
# Add suggested evaluations
|
|
120
|
+
for eval_name in req.suggested_evals:
|
|
121
|
+
mapped_name = self.eval_mappings.get(eval_name)
|
|
122
|
+
if mapped_name and mapped_name not in added_evals:
|
|
123
|
+
threshold = base_threshold
|
|
124
|
+
weight = 1.0
|
|
125
|
+
|
|
126
|
+
# Increase threshold for required items
|
|
127
|
+
if req.importance == "required":
|
|
128
|
+
threshold = min(0.95, threshold + 0.05)
|
|
129
|
+
weight = 1.5
|
|
130
|
+
|
|
131
|
+
evals.append(
|
|
132
|
+
EvalConfig(
|
|
133
|
+
name=mapped_name,
|
|
134
|
+
threshold=threshold,
|
|
135
|
+
weight=weight,
|
|
136
|
+
)
|
|
137
|
+
)
|
|
138
|
+
added_evals.add(mapped_name)
|
|
139
|
+
|
|
140
|
+
# Add suggested scanners
|
|
141
|
+
for scanner_name in req.suggested_scanners:
|
|
142
|
+
mapped_name = self.scanner_mappings.get(scanner_name)
|
|
143
|
+
if mapped_name and mapped_name not in added_scanners:
|
|
144
|
+
action = SCANNER_ACTIONS.get(req.importance, "flag")
|
|
145
|
+
|
|
146
|
+
scanners.append(
|
|
147
|
+
ScannerConfig(
|
|
148
|
+
name=mapped_name,
|
|
149
|
+
threshold=base_threshold,
|
|
150
|
+
action=action,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
added_scanners.add(mapped_name)
|
|
154
|
+
|
|
155
|
+
# Add domain-specific recommendations
|
|
156
|
+
domain_evals, domain_scanners = self._add_domain_recommendations(
|
|
157
|
+
analysis, base_threshold, added_evals, added_scanners
|
|
158
|
+
)
|
|
159
|
+
evals.extend(domain_evals)
|
|
160
|
+
scanners.extend(domain_scanners)
|
|
161
|
+
|
|
162
|
+
return evals, scanners
|
|
163
|
+
|
|
164
|
+
def _add_domain_recommendations(
|
|
165
|
+
self,
|
|
166
|
+
analysis: AppAnalysis,
|
|
167
|
+
base_threshold: float,
|
|
168
|
+
added_evals: Set[str],
|
|
169
|
+
added_scanners: Set[str],
|
|
170
|
+
) -> Tuple[List[EvalConfig], List[ScannerConfig]]:
|
|
171
|
+
"""Add domain-specific recommendations."""
|
|
172
|
+
evals: List[EvalConfig] = []
|
|
173
|
+
scanners: List[ScannerConfig] = []
|
|
174
|
+
|
|
175
|
+
# PII-sensitive domains always need PII scanner
|
|
176
|
+
pii_domains = {
|
|
177
|
+
DomainSensitivity.PII_SENSITIVE,
|
|
178
|
+
DomainSensitivity.HEALTHCARE,
|
|
179
|
+
DomainSensitivity.FINANCIAL,
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
if analysis.domain_sensitivity in pii_domains:
|
|
183
|
+
if "PIIScanner" not in added_scanners:
|
|
184
|
+
scanners.append(
|
|
185
|
+
ScannerConfig(
|
|
186
|
+
name="PIIScanner",
|
|
187
|
+
threshold=base_threshold,
|
|
188
|
+
action="redact", # Redact PII instead of blocking
|
|
189
|
+
)
|
|
190
|
+
)
|
|
191
|
+
if "SecretsScanner" not in added_scanners:
|
|
192
|
+
scanners.append(
|
|
193
|
+
ScannerConfig(
|
|
194
|
+
name="SecretsScanner",
|
|
195
|
+
threshold=base_threshold,
|
|
196
|
+
action="block",
|
|
197
|
+
)
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
# Healthcare needs extra safety
|
|
201
|
+
if analysis.domain_sensitivity == DomainSensitivity.HEALTHCARE:
|
|
202
|
+
if "ToxicityScanner" not in added_scanners:
|
|
203
|
+
scanners.append(
|
|
204
|
+
ScannerConfig(
|
|
205
|
+
name="ToxicityScanner",
|
|
206
|
+
threshold=base_threshold + 0.1,
|
|
207
|
+
action="block",
|
|
208
|
+
)
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
# Children's content needs strict safety
|
|
212
|
+
if analysis.domain_sensitivity == DomainSensitivity.CHILDREN:
|
|
213
|
+
for scanner_name in ["ToxicityScanner", "BiasScanner"]:
|
|
214
|
+
if scanner_name not in added_scanners:
|
|
215
|
+
scanners.append(
|
|
216
|
+
ScannerConfig(
|
|
217
|
+
name=scanner_name,
|
|
218
|
+
threshold=0.9,
|
|
219
|
+
action="block",
|
|
220
|
+
)
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
# High/critical risk always needs jailbreak protection
|
|
224
|
+
if analysis.risk_level in {RiskLevel.HIGH, RiskLevel.CRITICAL}:
|
|
225
|
+
if "JailbreakScanner" not in added_scanners:
|
|
226
|
+
scanners.append(
|
|
227
|
+
ScannerConfig(
|
|
228
|
+
name="JailbreakScanner",
|
|
229
|
+
threshold=base_threshold,
|
|
230
|
+
action="block",
|
|
231
|
+
)
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
return evals, scanners
|
|
235
|
+
|
|
236
|
+
def get_available_evals(self) -> List[str]:
|
|
237
|
+
"""Get list of available evaluation names."""
|
|
238
|
+
return list(set(self.eval_mappings.values()))
|
|
239
|
+
|
|
240
|
+
def get_available_scanners(self) -> List[str]:
|
|
241
|
+
"""Get list of available scanner names."""
|
|
242
|
+
return list(set(self.scanner_mappings.values()))
|