agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
"""Early stop policies for streaming evaluation.
|
|
2
|
+
|
|
3
|
+
Defines conditions under which streaming evaluation should stop early.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Dict, List, Optional, Callable, Any
|
|
8
|
+
from .types import EarlyStopReason, EarlyStopCondition, ChunkResult
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class PolicyState:
|
|
13
|
+
"""Tracks state for policy evaluation."""
|
|
14
|
+
|
|
15
|
+
consecutive_failures: Dict[str, int] = field(default_factory=dict)
|
|
16
|
+
total_failures: Dict[str, int] = field(default_factory=dict)
|
|
17
|
+
triggered_conditions: List[str] = field(default_factory=list)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class EarlyStopPolicy:
|
|
21
|
+
"""
|
|
22
|
+
Policy that determines when to stop streaming evaluation early.
|
|
23
|
+
|
|
24
|
+
Manages a set of conditions that can trigger early stopping, such as
|
|
25
|
+
toxicity thresholds, safety violations, or custom conditions.
|
|
26
|
+
|
|
27
|
+
Example:
|
|
28
|
+
policy = EarlyStopPolicy()
|
|
29
|
+
policy.add_condition(
|
|
30
|
+
name="high_toxicity",
|
|
31
|
+
eval_name="toxicity",
|
|
32
|
+
threshold=0.7,
|
|
33
|
+
comparison="above",
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
for chunk_result in stream:
|
|
37
|
+
should_stop, reason = policy.check(chunk_result)
|
|
38
|
+
if should_stop:
|
|
39
|
+
break
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(self):
|
|
43
|
+
"""Initialize the policy."""
|
|
44
|
+
self._conditions: List[EarlyStopCondition] = []
|
|
45
|
+
self._custom_checks: List[Callable[[ChunkResult], Optional[EarlyStopReason]]] = []
|
|
46
|
+
self._state = PolicyState()
|
|
47
|
+
|
|
48
|
+
def add_condition(
|
|
49
|
+
self,
|
|
50
|
+
name: str,
|
|
51
|
+
eval_name: str,
|
|
52
|
+
threshold: float,
|
|
53
|
+
comparison: str = "below",
|
|
54
|
+
consecutive_chunks: int = 1,
|
|
55
|
+
) -> "EarlyStopPolicy":
|
|
56
|
+
"""
|
|
57
|
+
Add a threshold-based stop condition.
|
|
58
|
+
|
|
59
|
+
Args:
|
|
60
|
+
name: Name for this condition
|
|
61
|
+
eval_name: Name of the evaluation to check
|
|
62
|
+
threshold: Threshold value
|
|
63
|
+
comparison: "below" (stop if score < threshold) or "above" (stop if score > threshold)
|
|
64
|
+
consecutive_chunks: Number of consecutive chunks that must fail
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
Self for chaining
|
|
68
|
+
"""
|
|
69
|
+
condition = EarlyStopCondition(
|
|
70
|
+
name=name,
|
|
71
|
+
eval_name=eval_name,
|
|
72
|
+
threshold=threshold,
|
|
73
|
+
comparison=comparison,
|
|
74
|
+
consecutive_chunks=consecutive_chunks,
|
|
75
|
+
)
|
|
76
|
+
self._conditions.append(condition)
|
|
77
|
+
return self
|
|
78
|
+
|
|
79
|
+
def add_toxicity_stop(
|
|
80
|
+
self,
|
|
81
|
+
threshold: float = 0.7,
|
|
82
|
+
consecutive: int = 1,
|
|
83
|
+
) -> "EarlyStopPolicy":
|
|
84
|
+
"""
|
|
85
|
+
Add toxicity-based stop condition.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
threshold: Stop if toxicity score exceeds this
|
|
89
|
+
consecutive: Number of consecutive chunks
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
Self for chaining
|
|
93
|
+
"""
|
|
94
|
+
return self.add_condition(
|
|
95
|
+
name="toxicity_stop",
|
|
96
|
+
eval_name="toxicity",
|
|
97
|
+
threshold=threshold,
|
|
98
|
+
comparison="above",
|
|
99
|
+
consecutive_chunks=consecutive,
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
def add_safety_stop(
|
|
103
|
+
self,
|
|
104
|
+
threshold: float = 0.3,
|
|
105
|
+
consecutive: int = 1,
|
|
106
|
+
) -> "EarlyStopPolicy":
|
|
107
|
+
"""
|
|
108
|
+
Add safety-based stop condition.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
threshold: Stop if safety score drops below this
|
|
112
|
+
consecutive: Number of consecutive chunks
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
Self for chaining
|
|
116
|
+
"""
|
|
117
|
+
return self.add_condition(
|
|
118
|
+
name="safety_stop",
|
|
119
|
+
eval_name="safety",
|
|
120
|
+
threshold=threshold,
|
|
121
|
+
comparison="below",
|
|
122
|
+
consecutive_chunks=consecutive,
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
def add_quality_stop(
|
|
126
|
+
self,
|
|
127
|
+
threshold: float = 0.3,
|
|
128
|
+
consecutive: int = 3,
|
|
129
|
+
) -> "EarlyStopPolicy":
|
|
130
|
+
"""
|
|
131
|
+
Add quality-based stop condition.
|
|
132
|
+
|
|
133
|
+
Args:
|
|
134
|
+
threshold: Stop if quality score stays below this
|
|
135
|
+
consecutive: Number of consecutive chunks
|
|
136
|
+
|
|
137
|
+
Returns:
|
|
138
|
+
Self for chaining
|
|
139
|
+
"""
|
|
140
|
+
return self.add_condition(
|
|
141
|
+
name="quality_stop",
|
|
142
|
+
eval_name="quality",
|
|
143
|
+
threshold=threshold,
|
|
144
|
+
comparison="below",
|
|
145
|
+
consecutive_chunks=consecutive,
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
def add_custom_check(
|
|
149
|
+
self,
|
|
150
|
+
check_fn: Callable[[ChunkResult], Optional[EarlyStopReason]],
|
|
151
|
+
) -> "EarlyStopPolicy":
|
|
152
|
+
"""
|
|
153
|
+
Add a custom check function.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
check_fn: Function that takes ChunkResult and returns
|
|
157
|
+
EarlyStopReason if should stop, None otherwise
|
|
158
|
+
|
|
159
|
+
Returns:
|
|
160
|
+
Self for chaining
|
|
161
|
+
"""
|
|
162
|
+
self._custom_checks.append(check_fn)
|
|
163
|
+
return self
|
|
164
|
+
|
|
165
|
+
def check(self, chunk_result: ChunkResult) -> tuple:
|
|
166
|
+
"""
|
|
167
|
+
Check if any stop conditions are triggered.
|
|
168
|
+
|
|
169
|
+
Args:
|
|
170
|
+
chunk_result: Result from evaluating a chunk
|
|
171
|
+
|
|
172
|
+
Returns:
|
|
173
|
+
Tuple of (should_stop: bool, reason: EarlyStopReason)
|
|
174
|
+
"""
|
|
175
|
+
# Check threshold-based conditions
|
|
176
|
+
for condition in self._conditions:
|
|
177
|
+
if not condition.enabled:
|
|
178
|
+
continue
|
|
179
|
+
|
|
180
|
+
score = chunk_result.scores.get(condition.eval_name)
|
|
181
|
+
if score is None:
|
|
182
|
+
continue
|
|
183
|
+
|
|
184
|
+
# Track consecutive failures
|
|
185
|
+
key = condition.name
|
|
186
|
+
if self._check_threshold(score, condition.threshold, condition.comparison):
|
|
187
|
+
self._state.consecutive_failures[key] = (
|
|
188
|
+
self._state.consecutive_failures.get(key, 0) + 1
|
|
189
|
+
)
|
|
190
|
+
self._state.total_failures[key] = (
|
|
191
|
+
self._state.total_failures.get(key, 0) + 1
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
# Check if consecutive threshold met
|
|
195
|
+
if condition.check(score, self._state.consecutive_failures[key]):
|
|
196
|
+
self._state.triggered_conditions.append(condition.name)
|
|
197
|
+
return True, self._get_reason_for_condition(condition)
|
|
198
|
+
else:
|
|
199
|
+
# Reset consecutive count
|
|
200
|
+
self._state.consecutive_failures[key] = 0
|
|
201
|
+
|
|
202
|
+
# Check custom conditions
|
|
203
|
+
for check_fn in self._custom_checks:
|
|
204
|
+
reason = check_fn(chunk_result)
|
|
205
|
+
if reason is not None:
|
|
206
|
+
return True, reason
|
|
207
|
+
|
|
208
|
+
return False, EarlyStopReason.NONE
|
|
209
|
+
|
|
210
|
+
def _check_threshold(
|
|
211
|
+
self,
|
|
212
|
+
score: float,
|
|
213
|
+
threshold: float,
|
|
214
|
+
comparison: str,
|
|
215
|
+
) -> bool:
|
|
216
|
+
"""Check if score triggers threshold."""
|
|
217
|
+
if comparison == "below":
|
|
218
|
+
return score < threshold
|
|
219
|
+
else:
|
|
220
|
+
return score > threshold
|
|
221
|
+
|
|
222
|
+
def _get_reason_for_condition(
|
|
223
|
+
self,
|
|
224
|
+
condition: EarlyStopCondition,
|
|
225
|
+
) -> EarlyStopReason:
|
|
226
|
+
"""Get the appropriate stop reason for a condition."""
|
|
227
|
+
name_lower = condition.name.lower()
|
|
228
|
+
eval_lower = condition.eval_name.lower()
|
|
229
|
+
|
|
230
|
+
if "toxic" in name_lower or "toxic" in eval_lower:
|
|
231
|
+
return EarlyStopReason.TOXICITY
|
|
232
|
+
elif "safe" in name_lower or "safe" in eval_lower:
|
|
233
|
+
return EarlyStopReason.SAFETY
|
|
234
|
+
elif "pii" in name_lower or "pii" in eval_lower:
|
|
235
|
+
return EarlyStopReason.PII
|
|
236
|
+
elif "jailbreak" in name_lower or "jailbreak" in eval_lower:
|
|
237
|
+
return EarlyStopReason.JAILBREAK
|
|
238
|
+
else:
|
|
239
|
+
return EarlyStopReason.THRESHOLD
|
|
240
|
+
|
|
241
|
+
def reset(self) -> None:
|
|
242
|
+
"""Reset policy state."""
|
|
243
|
+
self._state = PolicyState()
|
|
244
|
+
|
|
245
|
+
def enable_condition(self, name: str) -> None:
|
|
246
|
+
"""Enable a condition by name."""
|
|
247
|
+
for condition in self._conditions:
|
|
248
|
+
if condition.name == name:
|
|
249
|
+
condition.enabled = True
|
|
250
|
+
break
|
|
251
|
+
|
|
252
|
+
def disable_condition(self, name: str) -> None:
|
|
253
|
+
"""Disable a condition by name."""
|
|
254
|
+
for condition in self._conditions:
|
|
255
|
+
if condition.name == name:
|
|
256
|
+
condition.enabled = False
|
|
257
|
+
break
|
|
258
|
+
|
|
259
|
+
def get_stats(self) -> Dict[str, Any]:
|
|
260
|
+
"""Get policy statistics."""
|
|
261
|
+
return {
|
|
262
|
+
"conditions": [c.to_dict() for c in self._conditions],
|
|
263
|
+
"consecutive_failures": dict(self._state.consecutive_failures),
|
|
264
|
+
"total_failures": dict(self._state.total_failures),
|
|
265
|
+
"triggered_conditions": list(self._state.triggered_conditions),
|
|
266
|
+
"custom_checks": len(self._custom_checks),
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
@classmethod
|
|
270
|
+
def default(cls) -> "EarlyStopPolicy":
|
|
271
|
+
"""
|
|
272
|
+
Create a policy with sensible defaults.
|
|
273
|
+
|
|
274
|
+
Returns:
|
|
275
|
+
EarlyStopPolicy with toxicity and safety stops
|
|
276
|
+
"""
|
|
277
|
+
policy = cls()
|
|
278
|
+
policy.add_toxicity_stop(threshold=0.7, consecutive=1)
|
|
279
|
+
policy.add_safety_stop(threshold=0.3, consecutive=1)
|
|
280
|
+
return policy
|
|
281
|
+
|
|
282
|
+
@classmethod
|
|
283
|
+
def strict(cls) -> "EarlyStopPolicy":
|
|
284
|
+
"""
|
|
285
|
+
Create a strict policy for high-risk applications.
|
|
286
|
+
|
|
287
|
+
Returns:
|
|
288
|
+
EarlyStopPolicy with strict thresholds
|
|
289
|
+
"""
|
|
290
|
+
policy = cls()
|
|
291
|
+
policy.add_toxicity_stop(threshold=0.5, consecutive=1)
|
|
292
|
+
policy.add_safety_stop(threshold=0.5, consecutive=1)
|
|
293
|
+
policy.add_quality_stop(threshold=0.4, consecutive=2)
|
|
294
|
+
return policy
|
|
295
|
+
|
|
296
|
+
@classmethod
|
|
297
|
+
def permissive(cls) -> "EarlyStopPolicy":
|
|
298
|
+
"""
|
|
299
|
+
Create a permissive policy that only stops on severe issues.
|
|
300
|
+
|
|
301
|
+
Returns:
|
|
302
|
+
EarlyStopPolicy with high thresholds
|
|
303
|
+
"""
|
|
304
|
+
policy = cls()
|
|
305
|
+
policy.add_toxicity_stop(threshold=0.9, consecutive=2)
|
|
306
|
+
policy.add_safety_stop(threshold=0.1, consecutive=2)
|
|
307
|
+
return policy
|
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
"""Streaming-compatible scorer functions.
|
|
2
|
+
|
|
3
|
+
Provides lightweight evaluation functions optimized for streaming evaluation.
|
|
4
|
+
These scorers are designed to be fast and work with incremental text.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
from typing import Callable, List, Set
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# Toxicity word lists (simplified for demonstration)
|
|
12
|
+
TOXIC_WORDS: Set[str] = {
|
|
13
|
+
"hate", "kill", "attack", "destroy", "violent", "threat",
|
|
14
|
+
"abuse", "harass", "racist", "sexist", "discriminate",
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
PROFANITY_WORDS: Set[str] = {
|
|
18
|
+
# Basic profanity patterns (simplified)
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
# PII patterns
|
|
22
|
+
PII_PATTERNS = {
|
|
23
|
+
"email": re.compile(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b'),
|
|
24
|
+
"phone": re.compile(r'\b(?:\+?1[-.\s]?)?\(?[0-9]{3}\)?[-.\s]?[0-9]{3}[-.\s]?[0-9]{4}\b'),
|
|
25
|
+
"ssn": re.compile(r'\b\d{3}-\d{2}-\d{4}\b'),
|
|
26
|
+
"credit_card": re.compile(r'\b(?:\d{4}[-\s]?){3}\d{4}\b'),
|
|
27
|
+
"ip_address": re.compile(r'\b\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}\b'),
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
# Jailbreak patterns
|
|
31
|
+
JAILBREAK_PATTERNS = [
|
|
32
|
+
re.compile(r'ignore\s+(?:all\s+)?(?:previous\s+)?instructions?', re.IGNORECASE),
|
|
33
|
+
re.compile(r'disregard\s+(?:all\s+)?(?:previous\s+)?instructions?', re.IGNORECASE),
|
|
34
|
+
re.compile(r'forget\s+(?:all\s+)?(?:previous\s+)?instructions?', re.IGNORECASE),
|
|
35
|
+
re.compile(r'you\s+are\s+now\s+(?:a\s+)?(?:different|new)', re.IGNORECASE),
|
|
36
|
+
re.compile(r'pretend\s+(?:you\s+are|to\s+be)', re.IGNORECASE),
|
|
37
|
+
re.compile(r'act\s+as\s+(?:if|though)', re.IGNORECASE),
|
|
38
|
+
]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def toxicity_scorer(chunk: str, cumulative: str) -> float:
|
|
42
|
+
"""
|
|
43
|
+
Score text for toxicity.
|
|
44
|
+
|
|
45
|
+
Returns a score from 0.0 (not toxic) to 1.0 (highly toxic).
|
|
46
|
+
Uses the cumulative text for better context.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
chunk: Current chunk text
|
|
50
|
+
cumulative: All text so far
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
Toxicity score (0.0 = safe, 1.0 = toxic)
|
|
54
|
+
"""
|
|
55
|
+
text = cumulative.lower()
|
|
56
|
+
words = set(re.findall(r'\b\w+\b', text))
|
|
57
|
+
|
|
58
|
+
toxic_count = len(words.intersection(TOXIC_WORDS))
|
|
59
|
+
total_words = len(words) if words else 1
|
|
60
|
+
|
|
61
|
+
# Calculate toxicity ratio with diminishing returns
|
|
62
|
+
raw_score = toxic_count / max(total_words, 10)
|
|
63
|
+
|
|
64
|
+
# Scale to 0-1 range with sensitivity adjustment
|
|
65
|
+
score = min(1.0, raw_score * 5)
|
|
66
|
+
|
|
67
|
+
return score
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def safety_scorer(chunk: str, cumulative: str) -> float:
|
|
71
|
+
"""
|
|
72
|
+
Score text for general safety.
|
|
73
|
+
|
|
74
|
+
Returns a score from 0.0 (unsafe) to 1.0 (safe).
|
|
75
|
+
Higher is better (opposite of toxicity).
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
chunk: Current chunk text
|
|
79
|
+
cumulative: All text so far
|
|
80
|
+
|
|
81
|
+
Returns:
|
|
82
|
+
Safety score (0.0 = unsafe, 1.0 = safe)
|
|
83
|
+
"""
|
|
84
|
+
# Inverse of toxicity
|
|
85
|
+
toxicity = toxicity_scorer(chunk, cumulative)
|
|
86
|
+
return 1.0 - toxicity
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def pii_scorer(chunk: str, cumulative: str) -> float:
|
|
90
|
+
"""
|
|
91
|
+
Score text for PII presence.
|
|
92
|
+
|
|
93
|
+
Returns a score from 0.0 (no PII) to 1.0 (contains PII).
|
|
94
|
+
Lower is better (no PII is good).
|
|
95
|
+
|
|
96
|
+
Args:
|
|
97
|
+
chunk: Current chunk text
|
|
98
|
+
cumulative: All text so far
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
PII score (0.0 = no PII, 1.0 = contains PII)
|
|
102
|
+
"""
|
|
103
|
+
text = cumulative
|
|
104
|
+
pii_found = 0
|
|
105
|
+
|
|
106
|
+
for pattern_name, pattern in PII_PATTERNS.items():
|
|
107
|
+
matches = pattern.findall(text)
|
|
108
|
+
pii_found += len(matches)
|
|
109
|
+
|
|
110
|
+
# Return 1.0 if any PII found, otherwise 0.0
|
|
111
|
+
# Could be weighted by severity in production
|
|
112
|
+
return min(1.0, pii_found * 0.5)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def jailbreak_scorer(chunk: str, cumulative: str) -> float:
|
|
116
|
+
"""
|
|
117
|
+
Score text for jailbreak attempt patterns.
|
|
118
|
+
|
|
119
|
+
Returns a score from 0.0 (no jailbreak) to 1.0 (jailbreak detected).
|
|
120
|
+
Lower is better.
|
|
121
|
+
|
|
122
|
+
Args:
|
|
123
|
+
chunk: Current chunk text
|
|
124
|
+
cumulative: All text so far
|
|
125
|
+
|
|
126
|
+
Returns:
|
|
127
|
+
Jailbreak score (0.0 = safe, 1.0 = jailbreak detected)
|
|
128
|
+
"""
|
|
129
|
+
text = cumulative
|
|
130
|
+
|
|
131
|
+
for pattern in JAILBREAK_PATTERNS:
|
|
132
|
+
if pattern.search(text):
|
|
133
|
+
return 1.0
|
|
134
|
+
|
|
135
|
+
return 0.0
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def coherence_scorer(chunk: str, cumulative: str) -> float:
|
|
139
|
+
"""
|
|
140
|
+
Score text for coherence.
|
|
141
|
+
|
|
142
|
+
A simple heuristic based on sentence structure.
|
|
143
|
+
Returns 1.0 for coherent text, lower for incoherent.
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
chunk: Current chunk text
|
|
147
|
+
cumulative: All text so far
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
Coherence score (0.0 = incoherent, 1.0 = coherent)
|
|
151
|
+
"""
|
|
152
|
+
text = cumulative.strip()
|
|
153
|
+
|
|
154
|
+
if not text:
|
|
155
|
+
return 1.0
|
|
156
|
+
|
|
157
|
+
# Count sentences
|
|
158
|
+
sentences = re.split(r'[.!?]+', text)
|
|
159
|
+
sentences = [s.strip() for s in sentences if s.strip()]
|
|
160
|
+
|
|
161
|
+
if not sentences:
|
|
162
|
+
return 0.5
|
|
163
|
+
|
|
164
|
+
# Check for basic coherence indicators
|
|
165
|
+
score = 1.0
|
|
166
|
+
|
|
167
|
+
# Penalize very short average sentence length
|
|
168
|
+
avg_words = sum(len(s.split()) for s in sentences) / len(sentences)
|
|
169
|
+
if avg_words < 3:
|
|
170
|
+
score -= 0.3
|
|
171
|
+
|
|
172
|
+
# Penalize excessive repetition
|
|
173
|
+
words = cumulative.lower().split()
|
|
174
|
+
if len(words) > 10:
|
|
175
|
+
unique_ratio = len(set(words)) / len(words)
|
|
176
|
+
if unique_ratio < 0.3:
|
|
177
|
+
score -= 0.4
|
|
178
|
+
|
|
179
|
+
# Penalize gibberish (high non-alpha ratio)
|
|
180
|
+
alpha_chars = sum(1 for c in text if c.isalpha() or c.isspace())
|
|
181
|
+
if len(text) > 0:
|
|
182
|
+
alpha_ratio = alpha_chars / len(text)
|
|
183
|
+
if alpha_ratio < 0.7:
|
|
184
|
+
score -= 0.3
|
|
185
|
+
|
|
186
|
+
return max(0.0, score)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def quality_scorer(chunk: str, cumulative: str) -> float:
|
|
190
|
+
"""
|
|
191
|
+
Score text for overall quality.
|
|
192
|
+
|
|
193
|
+
Combines multiple heuristics for a quality assessment.
|
|
194
|
+
|
|
195
|
+
Args:
|
|
196
|
+
chunk: Current chunk text
|
|
197
|
+
cumulative: All text so far
|
|
198
|
+
|
|
199
|
+
Returns:
|
|
200
|
+
Quality score (0.0 = poor, 1.0 = high quality)
|
|
201
|
+
"""
|
|
202
|
+
text = cumulative.strip()
|
|
203
|
+
|
|
204
|
+
if not text:
|
|
205
|
+
return 0.5
|
|
206
|
+
|
|
207
|
+
scores = []
|
|
208
|
+
|
|
209
|
+
# Coherence component
|
|
210
|
+
scores.append(coherence_scorer(chunk, cumulative))
|
|
211
|
+
|
|
212
|
+
# Length appropriateness (not too short, not repetitive)
|
|
213
|
+
words = text.split()
|
|
214
|
+
if len(words) > 5:
|
|
215
|
+
scores.append(0.8)
|
|
216
|
+
else:
|
|
217
|
+
scores.append(0.5)
|
|
218
|
+
|
|
219
|
+
# Punctuation presence
|
|
220
|
+
if re.search(r'[.!?,]', text):
|
|
221
|
+
scores.append(0.9)
|
|
222
|
+
else:
|
|
223
|
+
scores.append(0.6)
|
|
224
|
+
|
|
225
|
+
return sum(scores) / len(scores) if scores else 0.5
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def create_keyword_scorer(
|
|
229
|
+
keywords: Set[str],
|
|
230
|
+
return_high_on_match: bool = True,
|
|
231
|
+
) -> Callable[[str, str], float]:
|
|
232
|
+
"""
|
|
233
|
+
Create a custom keyword-based scorer.
|
|
234
|
+
|
|
235
|
+
Args:
|
|
236
|
+
keywords: Set of keywords to detect
|
|
237
|
+
return_high_on_match: If True, returns high score on match
|
|
238
|
+
|
|
239
|
+
Returns:
|
|
240
|
+
Scorer function
|
|
241
|
+
"""
|
|
242
|
+
keywords_lower = {k.lower() for k in keywords}
|
|
243
|
+
|
|
244
|
+
def scorer(chunk: str, cumulative: str) -> float:
|
|
245
|
+
text = cumulative.lower()
|
|
246
|
+
words = set(re.findall(r'\b\w+\b', text))
|
|
247
|
+
|
|
248
|
+
matches = len(words.intersection(keywords_lower))
|
|
249
|
+
|
|
250
|
+
if return_high_on_match:
|
|
251
|
+
return min(1.0, matches * 0.2)
|
|
252
|
+
else:
|
|
253
|
+
return max(0.0, 1.0 - matches * 0.2)
|
|
254
|
+
|
|
255
|
+
return scorer
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def create_pattern_scorer(
|
|
259
|
+
patterns: List[re.Pattern],
|
|
260
|
+
return_high_on_match: bool = True,
|
|
261
|
+
) -> Callable[[str, str], float]:
|
|
262
|
+
"""
|
|
263
|
+
Create a custom regex pattern-based scorer.
|
|
264
|
+
|
|
265
|
+
Args:
|
|
266
|
+
patterns: List of compiled regex patterns
|
|
267
|
+
return_high_on_match: If True, returns high score on match
|
|
268
|
+
|
|
269
|
+
Returns:
|
|
270
|
+
Scorer function
|
|
271
|
+
"""
|
|
272
|
+
def scorer(chunk: str, cumulative: str) -> float:
|
|
273
|
+
text = cumulative
|
|
274
|
+
|
|
275
|
+
for pattern in patterns:
|
|
276
|
+
if pattern.search(text):
|
|
277
|
+
return 1.0 if return_high_on_match else 0.0
|
|
278
|
+
|
|
279
|
+
return 0.0 if return_high_on_match else 1.0
|
|
280
|
+
|
|
281
|
+
return scorer
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
class CompositeScorer:
|
|
285
|
+
"""
|
|
286
|
+
Combines multiple scorers with weights.
|
|
287
|
+
|
|
288
|
+
Example:
|
|
289
|
+
scorer = CompositeScorer()
|
|
290
|
+
scorer.add(toxicity_scorer, weight=2.0)
|
|
291
|
+
scorer.add(coherence_scorer, weight=1.0)
|
|
292
|
+
|
|
293
|
+
combined_score = scorer(chunk, cumulative)
|
|
294
|
+
"""
|
|
295
|
+
|
|
296
|
+
def __init__(self):
|
|
297
|
+
"""Initialize composite scorer."""
|
|
298
|
+
self._scorers: List[tuple] = [] # (scorer_fn, weight)
|
|
299
|
+
|
|
300
|
+
def add(
|
|
301
|
+
self,
|
|
302
|
+
scorer: Callable[[str, str], float],
|
|
303
|
+
weight: float = 1.0,
|
|
304
|
+
) -> "CompositeScorer":
|
|
305
|
+
"""
|
|
306
|
+
Add a scorer with weight.
|
|
307
|
+
|
|
308
|
+
Args:
|
|
309
|
+
scorer: Scorer function
|
|
310
|
+
weight: Weight for this scorer
|
|
311
|
+
|
|
312
|
+
Returns:
|
|
313
|
+
Self for chaining
|
|
314
|
+
"""
|
|
315
|
+
self._scorers.append((scorer, weight))
|
|
316
|
+
return self
|
|
317
|
+
|
|
318
|
+
def __call__(self, chunk: str, cumulative: str) -> float:
|
|
319
|
+
"""
|
|
320
|
+
Calculate weighted average of all scorers.
|
|
321
|
+
|
|
322
|
+
Args:
|
|
323
|
+
chunk: Current chunk text
|
|
324
|
+
cumulative: All text so far
|
|
325
|
+
|
|
326
|
+
Returns:
|
|
327
|
+
Weighted average score
|
|
328
|
+
"""
|
|
329
|
+
if not self._scorers:
|
|
330
|
+
return 0.5
|
|
331
|
+
|
|
332
|
+
total_weight = sum(w for _, w in self._scorers)
|
|
333
|
+
weighted_sum = sum(
|
|
334
|
+
scorer(chunk, cumulative) * weight
|
|
335
|
+
for scorer, weight in self._scorers
|
|
336
|
+
)
|
|
337
|
+
|
|
338
|
+
return weighted_sum / total_weight if total_weight > 0 else 0.5
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
# Pre-configured composite scorers
|
|
342
|
+
def safety_composite_scorer(chunk: str, cumulative: str) -> float:
|
|
343
|
+
"""
|
|
344
|
+
Composite scorer for overall safety.
|
|
345
|
+
|
|
346
|
+
Combines toxicity, PII, and jailbreak detection.
|
|
347
|
+
Returns 1.0 for safe, 0.0 for unsafe.
|
|
348
|
+
"""
|
|
349
|
+
# Invert toxicity and PII scores (lower is better for them)
|
|
350
|
+
toxicity = 1.0 - toxicity_scorer(chunk, cumulative)
|
|
351
|
+
pii = 1.0 - pii_scorer(chunk, cumulative)
|
|
352
|
+
jailbreak = 1.0 - jailbreak_scorer(chunk, cumulative)
|
|
353
|
+
|
|
354
|
+
# Weighted combination (jailbreak is most critical)
|
|
355
|
+
return (toxicity * 0.3 + pii * 0.3 + jailbreak * 0.4)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def quality_composite_scorer(chunk: str, cumulative: str) -> float:
|
|
359
|
+
"""
|
|
360
|
+
Composite scorer for overall quality.
|
|
361
|
+
|
|
362
|
+
Combines coherence and quality metrics.
|
|
363
|
+
Returns 1.0 for high quality, 0.0 for low quality.
|
|
364
|
+
"""
|
|
365
|
+
coherence = coherence_scorer(chunk, cumulative)
|
|
366
|
+
quality = quality_scorer(chunk, cumulative)
|
|
367
|
+
|
|
368
|
+
return (coherence * 0.5 + quality * 0.5)
|