agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import logging
|
|
5
|
+
from collections.abc import Callable, Iterable
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from fi.simulate._logging import redacted_exc_info
|
|
10
|
+
from fi.simulate.agent.wrapper import AgentWrapper, SimulationArtifact, SimulationEvent
|
|
11
|
+
from fi.simulate.artifacts import ArtifactManifest
|
|
12
|
+
from fi.simulate.environment import EnvironmentAdapter
|
|
13
|
+
import fi.simulate.environments # noqa: F401 (registers builtin environment plugins)
|
|
14
|
+
from fi.simulate.registry import environment_registry
|
|
15
|
+
from fi.simulate.evidence import EvidenceSourceSummary
|
|
16
|
+
from fi.simulate.results.base import ResultSink
|
|
17
|
+
from fi.simulate.simulation.models import Persona
|
|
18
|
+
|
|
19
|
+
from .events import CanonicalEvent
|
|
20
|
+
from .failures import FailureStage, SimulationFailure
|
|
21
|
+
from .plan import SimulationPlan
|
|
22
|
+
from .planner import build_plan
|
|
23
|
+
from .report import SimulationReport, SimulationTestCaseResult
|
|
24
|
+
from .run import CleanupStatus, RunStatus
|
|
25
|
+
from .spec import SimulationSpec
|
|
26
|
+
|
|
27
|
+
logger = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class SimulationRunner:
|
|
31
|
+
async def run(
|
|
32
|
+
self,
|
|
33
|
+
spec: SimulationSpec,
|
|
34
|
+
*,
|
|
35
|
+
target: Callable[..., Any] | AgentWrapper | Any = None,
|
|
36
|
+
result_sink: ResultSink | None = None,
|
|
37
|
+
artifacts: list[SimulationArtifact | dict[str, Any]] | None = None,
|
|
38
|
+
events: list[SimulationEvent | dict[str, Any]] | None = None,
|
|
39
|
+
environment: EnvironmentAdapter | Iterable[EnvironmentAdapter] | None = None,
|
|
40
|
+
auto_execute_tools: bool = True,
|
|
41
|
+
stop_when: Callable[[list[dict[str, Any]], Persona], bool] | None = None,
|
|
42
|
+
agent_wrapper_kwargs: dict[str, Any] | None = None,
|
|
43
|
+
) -> SimulationReport:
|
|
44
|
+
started_at = datetime.now(timezone.utc)
|
|
45
|
+
plan: SimulationPlan | None = None
|
|
46
|
+
try:
|
|
47
|
+
plan = build_plan(spec)
|
|
48
|
+
if result_sink is not None:
|
|
49
|
+
result_sink.prepare(spec, plan)
|
|
50
|
+
# Hosted runs stream each case to the platform the moment it finishes
|
|
51
|
+
# (allocate CallExecution rows up front, PATCH by index). The sink
|
|
52
|
+
# decides eligibility; a non-streaming sink (local/chat, or missing
|
|
53
|
+
# config) returns False and we fall back to the batch-at-end path.
|
|
54
|
+
on_case_start, on_case_complete = self._begin_streaming(
|
|
55
|
+
result_sink, spec, plan
|
|
56
|
+
)
|
|
57
|
+
# Keep every case that finished. A deadline that fires while the plugin is still
|
|
58
|
+
# tidying up used to discard cases whose conversation had already completed, and the
|
|
59
|
+
# run then reported no test case at all: no turns, no transcript, and an infrastructure
|
|
60
|
+
# failure for a call that worked. Held here because this is the only layer that sees
|
|
61
|
+
# both the cases and the timeout.
|
|
62
|
+
finished: list[SimulationTestCaseResult] = []
|
|
63
|
+
# Bound to its own name before the wrapper is installed. Closing over
|
|
64
|
+
# ``on_case_complete`` and then rebinding it makes the wrapper call itself: the
|
|
65
|
+
# closure reads the name at call time, not the value it had at definition. It cost a
|
|
66
|
+
# whole run to find, and only after cases started completing at all, because a
|
|
67
|
+
# wrapper that never runs never recurses.
|
|
68
|
+
report_case = on_case_complete
|
|
69
|
+
|
|
70
|
+
async def _remember(index: int, legacy_case: Any) -> None:
|
|
71
|
+
try:
|
|
72
|
+
finished.append(
|
|
73
|
+
SimulationTestCaseResult.from_legacy_case(
|
|
74
|
+
legacy_case, index=index, run_id=spec.run_id
|
|
75
|
+
)
|
|
76
|
+
)
|
|
77
|
+
except Exception as exc: # noqa: BLE001 - never fail a case over bookkeeping
|
|
78
|
+
logger.warning(
|
|
79
|
+
"could not keep a finished case for the timeout path: %s", exc
|
|
80
|
+
)
|
|
81
|
+
if report_case is not None:
|
|
82
|
+
await report_case(index, legacy_case)
|
|
83
|
+
|
|
84
|
+
on_case_complete = _remember
|
|
85
|
+
self._write_event(
|
|
86
|
+
result_sink,
|
|
87
|
+
CanonicalEvent.create(
|
|
88
|
+
run_id=spec.run_id,
|
|
89
|
+
test_case_id="run",
|
|
90
|
+
event_type="session.started",
|
|
91
|
+
source="runtime",
|
|
92
|
+
sequence=0,
|
|
93
|
+
),
|
|
94
|
+
)
|
|
95
|
+
plugin = environment_registry.create(spec.environment.adapter)
|
|
96
|
+
legacy_report = await asyncio.wait_for(
|
|
97
|
+
plugin.run(
|
|
98
|
+
spec,
|
|
99
|
+
target=target,
|
|
100
|
+
artifacts=artifacts,
|
|
101
|
+
events=events,
|
|
102
|
+
environment=environment,
|
|
103
|
+
auto_execute_tools=auto_execute_tools,
|
|
104
|
+
stop_when=stop_when,
|
|
105
|
+
agent_wrapper_kwargs=agent_wrapper_kwargs,
|
|
106
|
+
on_case_start=on_case_start,
|
|
107
|
+
on_case_complete=on_case_complete,
|
|
108
|
+
),
|
|
109
|
+
timeout=spec.execution.timeout.run_seconds,
|
|
110
|
+
)
|
|
111
|
+
except asyncio.TimeoutError:
|
|
112
|
+
report = self._failure_report(
|
|
113
|
+
spec,
|
|
114
|
+
plan=plan,
|
|
115
|
+
started_at=started_at,
|
|
116
|
+
status=RunStatus.TIMED_OUT,
|
|
117
|
+
failure=SimulationFailure(
|
|
118
|
+
stage=FailureStage.RUNNING,
|
|
119
|
+
code="simulation_timeout",
|
|
120
|
+
message="Simulation exceeded its run deadline",
|
|
121
|
+
retryable=True,
|
|
122
|
+
),
|
|
123
|
+
cases=finished,
|
|
124
|
+
)
|
|
125
|
+
except Exception as exc:
|
|
126
|
+
stage = FailureStage.PLANNING if plan is None else FailureStage.RUNNING
|
|
127
|
+
report = self._failure_report(
|
|
128
|
+
spec,
|
|
129
|
+
plan=plan,
|
|
130
|
+
started_at=started_at,
|
|
131
|
+
status=RunStatus.FAILED,
|
|
132
|
+
failure=SimulationFailure(
|
|
133
|
+
stage=stage,
|
|
134
|
+
code="simulation_failed",
|
|
135
|
+
message="Simulation execution failed",
|
|
136
|
+
retryable=False,
|
|
137
|
+
details={"exception_type": type(exc).__name__},
|
|
138
|
+
),
|
|
139
|
+
)
|
|
140
|
+
logger.error(
|
|
141
|
+
"Simulation run failed",
|
|
142
|
+
exc_info=redacted_exc_info(exc),
|
|
143
|
+
extra={
|
|
144
|
+
"run_id": spec.run_id,
|
|
145
|
+
"exception_type": type(exc).__name__,
|
|
146
|
+
},
|
|
147
|
+
)
|
|
148
|
+
else:
|
|
149
|
+
ended_at = datetime.now(timezone.utc)
|
|
150
|
+
report = SimulationReport.from_legacy(
|
|
151
|
+
legacy_report,
|
|
152
|
+
run_id=spec.run_id,
|
|
153
|
+
plan_id=plan.plan_id,
|
|
154
|
+
spec_hash=spec.spec_hash or spec.content_hash(),
|
|
155
|
+
status=RunStatus.COMPLETED,
|
|
156
|
+
started_at=started_at,
|
|
157
|
+
ended_at=ended_at,
|
|
158
|
+
artifacts=ArtifactManifest(run_id=spec.run_id),
|
|
159
|
+
evidence=[
|
|
160
|
+
EvidenceSourceSummary(
|
|
161
|
+
source_id=source.source_id,
|
|
162
|
+
adapter=source.adapter,
|
|
163
|
+
evidence_class=source.evidence_class,
|
|
164
|
+
capabilities=source.capabilities,
|
|
165
|
+
)
|
|
166
|
+
for source in spec.evidence.sources
|
|
167
|
+
],
|
|
168
|
+
)
|
|
169
|
+
report.cleanup_status = CleanupStatus.COMPLETED
|
|
170
|
+
report = SimulationReport.model_validate(
|
|
171
|
+
report.model_dump(exclude={"report_hash"})
|
|
172
|
+
)
|
|
173
|
+
# Environments enforce terminal conditions (plan §3): let the plugin
|
|
174
|
+
# override the run status from per-case results (e.g. voice marks an
|
|
175
|
+
# all-cases-failed run FAILED). Absent hook -> COMPLETED, as before.
|
|
176
|
+
finalize = getattr(plugin, "finalize_run_status", None)
|
|
177
|
+
if finalize is not None:
|
|
178
|
+
report = finalize(report)
|
|
179
|
+
self._write_event(
|
|
180
|
+
result_sink,
|
|
181
|
+
CanonicalEvent.create(
|
|
182
|
+
run_id=spec.run_id,
|
|
183
|
+
test_case_id="run",
|
|
184
|
+
event_type="session.ended",
|
|
185
|
+
source="runtime",
|
|
186
|
+
sequence=1,
|
|
187
|
+
payload={"status": report.status.value},
|
|
188
|
+
),
|
|
189
|
+
)
|
|
190
|
+
self._write_report(result_sink, report)
|
|
191
|
+
return report
|
|
192
|
+
|
|
193
|
+
def _begin_streaming(
|
|
194
|
+
self,
|
|
195
|
+
result_sink: ResultSink | None,
|
|
196
|
+
spec: SimulationSpec,
|
|
197
|
+
plan: SimulationPlan | None,
|
|
198
|
+
) -> tuple[Callable[[int], Any] | None, Callable[[int, Any], Any] | None]:
|
|
199
|
+
"""Open the sink's streaming session; return ``(on_case_start,
|
|
200
|
+
on_case_complete)`` — either may be ``None``.
|
|
201
|
+
|
|
202
|
+
``on_case_complete`` converts a legacy ``TestCaseResult`` to its canonical
|
|
203
|
+
form (identical to the finalized report) and submits it off the event loop
|
|
204
|
+
so a slow result PATCH never blocks case concurrency. ``on_case_start``
|
|
205
|
+
fires a best-effort ONGOING status ping the moment a case begins, and is
|
|
206
|
+
present only when the sink exposes a ``case_started`` method. Both are
|
|
207
|
+
engine-agnostic — any engine that invokes the hooks gets the behaviour.
|
|
208
|
+
A streaming error is swallowed here — the case is left un-streamed and
|
|
209
|
+
``finalize`` reconciles it; a broken upload must never fail a case.
|
|
210
|
+
"""
|
|
211
|
+
if result_sink is None:
|
|
212
|
+
return None, None
|
|
213
|
+
begin = getattr(result_sink, "begin_stream", None)
|
|
214
|
+
submit_case = getattr(result_sink, "submit_case", None)
|
|
215
|
+
if begin is None or submit_case is None:
|
|
216
|
+
return None, None
|
|
217
|
+
try:
|
|
218
|
+
streaming = bool(begin(spec, plan))
|
|
219
|
+
except Exception as exc:
|
|
220
|
+
logger.error(
|
|
221
|
+
"Simulation stream begin failed",
|
|
222
|
+
exc_info=redacted_exc_info(exc),
|
|
223
|
+
extra={"run_id": spec.run_id},
|
|
224
|
+
)
|
|
225
|
+
return None, None
|
|
226
|
+
if not streaming:
|
|
227
|
+
return None, None
|
|
228
|
+
|
|
229
|
+
evidence = [
|
|
230
|
+
EvidenceSourceSummary(
|
|
231
|
+
source_id=source.source_id,
|
|
232
|
+
adapter=source.adapter,
|
|
233
|
+
evidence_class=source.evidence_class,
|
|
234
|
+
capabilities=source.capabilities,
|
|
235
|
+
)
|
|
236
|
+
for source in spec.evidence.sources
|
|
237
|
+
]
|
|
238
|
+
run_id = spec.run_id
|
|
239
|
+
|
|
240
|
+
async def _on_case_complete(index: int, legacy_case: Any) -> None:
|
|
241
|
+
try:
|
|
242
|
+
canonical = SimulationTestCaseResult.from_legacy_case(
|
|
243
|
+
legacy_case, index=index, run_id=run_id, evidence=evidence
|
|
244
|
+
)
|
|
245
|
+
await asyncio.to_thread(submit_case, index, canonical)
|
|
246
|
+
except Exception as exc:
|
|
247
|
+
logger.error(
|
|
248
|
+
"Simulation stream case submit failed",
|
|
249
|
+
exc_info=redacted_exc_info(exc),
|
|
250
|
+
extra={"run_id": run_id, "case_index": index},
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
# Optional per-case start hook — present only if the sink supports it.
|
|
254
|
+
# Marks the row ONGOING when its case begins; never fails the case.
|
|
255
|
+
case_started = getattr(result_sink, "case_started", None)
|
|
256
|
+
on_case_start: Callable[[int], Any] | None = None
|
|
257
|
+
if case_started is not None:
|
|
258
|
+
|
|
259
|
+
async def _on_case_start(index: int) -> None:
|
|
260
|
+
try:
|
|
261
|
+
await asyncio.to_thread(case_started, index)
|
|
262
|
+
except Exception as exc:
|
|
263
|
+
logger.error(
|
|
264
|
+
"Simulation stream case start ping failed",
|
|
265
|
+
exc_info=redacted_exc_info(exc),
|
|
266
|
+
extra={"run_id": run_id, "case_index": index},
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
on_case_start = _on_case_start
|
|
270
|
+
|
|
271
|
+
return on_case_start, _on_case_complete
|
|
272
|
+
|
|
273
|
+
def _failure_report(
|
|
274
|
+
self,
|
|
275
|
+
spec: SimulationSpec,
|
|
276
|
+
*,
|
|
277
|
+
plan: SimulationPlan | None,
|
|
278
|
+
started_at: datetime,
|
|
279
|
+
status: RunStatus,
|
|
280
|
+
failure: SimulationFailure,
|
|
281
|
+
cases: list[SimulationTestCaseResult] | None = None,
|
|
282
|
+
) -> SimulationReport:
|
|
283
|
+
return SimulationReport(
|
|
284
|
+
run_id=spec.run_id,
|
|
285
|
+
plan_id=plan.plan_id if plan is not None else None,
|
|
286
|
+
spec_hash=spec.spec_hash or spec.content_hash(),
|
|
287
|
+
status=status,
|
|
288
|
+
cleanup_status=CleanupStatus.COMPLETED,
|
|
289
|
+
started_at=started_at,
|
|
290
|
+
ended_at=datetime.now(timezone.utc),
|
|
291
|
+
artifacts=ArtifactManifest(run_id=spec.run_id),
|
|
292
|
+
failure=failure,
|
|
293
|
+
test_cases=list(cases or []),
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
def _write_event(
|
|
297
|
+
self,
|
|
298
|
+
result_sink: ResultSink | None,
|
|
299
|
+
event: CanonicalEvent,
|
|
300
|
+
) -> None:
|
|
301
|
+
if result_sink is None:
|
|
302
|
+
return
|
|
303
|
+
try:
|
|
304
|
+
result_sink.write_event(event)
|
|
305
|
+
except Exception as exc:
|
|
306
|
+
logger.error(
|
|
307
|
+
"Simulation event sink failed",
|
|
308
|
+
exc_info=redacted_exc_info(exc),
|
|
309
|
+
extra={
|
|
310
|
+
"run_id": event.run_id,
|
|
311
|
+
"event_id": event.event_id,
|
|
312
|
+
"exception_type": type(exc).__name__,
|
|
313
|
+
},
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
def _write_report(
|
|
317
|
+
self,
|
|
318
|
+
result_sink: ResultSink | None,
|
|
319
|
+
report: SimulationReport,
|
|
320
|
+
) -> None:
|
|
321
|
+
if result_sink is None:
|
|
322
|
+
return
|
|
323
|
+
try:
|
|
324
|
+
result_sink.write_report(report)
|
|
325
|
+
except Exception as exc:
|
|
326
|
+
logger.error(
|
|
327
|
+
"Simulation report sink failed",
|
|
328
|
+
exc_info=redacted_exc_info(exc),
|
|
329
|
+
extra={
|
|
330
|
+
"run_id": report.run_id,
|
|
331
|
+
"exception_type": type(exc).__name__,
|
|
332
|
+
},
|
|
333
|
+
)
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping, Sequence
|
|
4
|
+
from enum import Enum
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, Field, JsonValue, model_validator
|
|
7
|
+
|
|
8
|
+
from fi.simulate._hashing import content_hash
|
|
9
|
+
from fi.simulate.evidence import EvidenceSourceSpec
|
|
10
|
+
from fi.simulate.simulation.models import Scenario
|
|
11
|
+
|
|
12
|
+
SIMULATION_SPEC_SCHEMA_VERSION = "futureagi.simulation-spec.v1"
|
|
13
|
+
|
|
14
|
+
_SECRET_KEYS = {
|
|
15
|
+
"api_key",
|
|
16
|
+
"api_secret",
|
|
17
|
+
"authorization",
|
|
18
|
+
"credential",
|
|
19
|
+
"credentials",
|
|
20
|
+
"password",
|
|
21
|
+
"private_key",
|
|
22
|
+
"secret",
|
|
23
|
+
"token",
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SecretRef(BaseModel):
|
|
28
|
+
manager: str
|
|
29
|
+
key: str
|
|
30
|
+
version: str | None = None
|
|
31
|
+
purpose: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class AdapterSpec(BaseModel):
|
|
35
|
+
adapter: str
|
|
36
|
+
adapter_version: str = "1"
|
|
37
|
+
config: dict[str, JsonValue] = Field(default_factory=dict)
|
|
38
|
+
secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class EnvironmentSpec(AdapterSpec):
|
|
42
|
+
world_kind: str
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AgentEndpointSpec(AdapterSpec):
|
|
46
|
+
required_capabilities: list[str] = Field(default_factory=list)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class SimulatorPolicySpec(AdapterSpec):
|
|
50
|
+
pass
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class ConversationDirection(str, Enum):
|
|
54
|
+
SIMULATOR_FIRST = "simulator_first"
|
|
55
|
+
AGENT_FIRST = "agent_first"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class RuntimeIsolation(str, Enum):
|
|
59
|
+
SHARED_RUNNER_PROCESS = "shared_runner_process"
|
|
60
|
+
DEDICATED_POD = "dedicated_pod"
|
|
61
|
+
DEDICATED_VM = "dedicated_vm"
|
|
62
|
+
EXTERNAL = "external"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class RuntimeRequirements(BaseModel):
|
|
66
|
+
isolation: RuntimeIsolation = RuntimeIsolation.SHARED_RUNNER_PROCESS
|
|
67
|
+
cpu_units: int = Field(default=1, ge=1)
|
|
68
|
+
memory_mb: int = Field(default=512, ge=128)
|
|
69
|
+
parallelism: int = Field(default=1, ge=1, le=8)
|
|
70
|
+
concurrency_weight: int = Field(default=1, ge=1)
|
|
71
|
+
max_duration_seconds: int = Field(default=300, ge=1)
|
|
72
|
+
network_policy: str = "live"
|
|
73
|
+
# World count for the hosted harness (seam contract §1); the gateway caps
|
|
74
|
+
# it at admission, so the model stays permissive beyond ge=1.
|
|
75
|
+
parallelism: int = Field(default=1, ge=1)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class TimeoutPolicy(BaseModel):
|
|
79
|
+
connect_seconds: float = Field(default=15.0, gt=0)
|
|
80
|
+
readiness_seconds: float = Field(default=30.0, gt=0)
|
|
81
|
+
run_seconds: float = Field(default=300.0, gt=0)
|
|
82
|
+
finalize_seconds: float = Field(default=30.0, gt=0)
|
|
83
|
+
cleanup_seconds: float = Field(default=30.0, gt=0)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class RetryPolicy(BaseModel):
|
|
87
|
+
max_attempts: int = Field(default=1, ge=1)
|
|
88
|
+
initial_backoff_seconds: float = Field(default=1.0, ge=0)
|
|
89
|
+
max_backoff_seconds: float = Field(default=30.0, ge=0)
|
|
90
|
+
|
|
91
|
+
@model_validator(mode="after")
|
|
92
|
+
def _validate_backoff(self) -> RetryPolicy:
|
|
93
|
+
if self.max_backoff_seconds < self.initial_backoff_seconds:
|
|
94
|
+
raise ValueError(
|
|
95
|
+
"retry_policy_invalid: max backoff is below initial backoff"
|
|
96
|
+
)
|
|
97
|
+
return self
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class CleanupPolicy(BaseModel):
|
|
101
|
+
always: bool = True
|
|
102
|
+
reconcile_before_create: bool = True
|
|
103
|
+
orphan_cleanup: bool = True
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class ExecutionPolicy(BaseModel):
|
|
107
|
+
direction: ConversationDirection = ConversationDirection.SIMULATOR_FIRST
|
|
108
|
+
runtime: RuntimeRequirements = Field(default_factory=RuntimeRequirements)
|
|
109
|
+
timeout: TimeoutPolicy = Field(default_factory=TimeoutPolicy)
|
|
110
|
+
retry: RetryPolicy = Field(default_factory=RetryPolicy)
|
|
111
|
+
cleanup: CleanupPolicy = Field(default_factory=CleanupPolicy)
|
|
112
|
+
max_parallel_cases: int = Field(default=1, ge=1)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
class EvidencePolicy(BaseModel):
|
|
116
|
+
sources: list[EvidenceSourceSpec] = Field(default_factory=list)
|
|
117
|
+
required_capabilities: list[str] = Field(default_factory=list)
|
|
118
|
+
|
|
119
|
+
@model_validator(mode="after")
|
|
120
|
+
def _validate_source_ids(self) -> EvidencePolicy:
|
|
121
|
+
source_ids = [source.source_id for source in self.sources]
|
|
122
|
+
if len(source_ids) != len(set(source_ids)):
|
|
123
|
+
raise ValueError(
|
|
124
|
+
"evidence_source_duplicate: source_id values must be unique"
|
|
125
|
+
)
|
|
126
|
+
return self
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class ArtifactPolicy(BaseModel):
|
|
130
|
+
enabled: bool = True
|
|
131
|
+
record_audio: bool = False
|
|
132
|
+
root_directory: str | None = None
|
|
133
|
+
max_inline_bytes: int = Field(default=65_536, ge=0)
|
|
134
|
+
required_types: list[str] = Field(default_factory=list)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class EvaluationRef(BaseModel):
|
|
138
|
+
evaluation_id: str
|
|
139
|
+
version: str | None = None
|
|
140
|
+
config: dict[str, JsonValue] = Field(default_factory=dict)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class SimulationSpec(BaseModel):
|
|
144
|
+
schema_version: str = SIMULATION_SPEC_SCHEMA_VERSION
|
|
145
|
+
run_id: str
|
|
146
|
+
environment: EnvironmentSpec
|
|
147
|
+
target: AgentEndpointSpec
|
|
148
|
+
simulator: SimulatorPolicySpec
|
|
149
|
+
scenario: Scenario
|
|
150
|
+
execution: ExecutionPolicy = Field(default_factory=ExecutionPolicy)
|
|
151
|
+
evidence: EvidencePolicy = Field(default_factory=EvidencePolicy)
|
|
152
|
+
artifacts: ArtifactPolicy = Field(default_factory=ArtifactPolicy)
|
|
153
|
+
evaluation_refs: list[EvaluationRef] = Field(default_factory=list)
|
|
154
|
+
metadata: dict[str, JsonValue] = Field(default_factory=dict)
|
|
155
|
+
spec_hash: str | None = None
|
|
156
|
+
|
|
157
|
+
def content_hash(self) -> str:
|
|
158
|
+
return content_hash(self.model_dump(exclude={"spec_hash"}, exclude_none=True))
|
|
159
|
+
|
|
160
|
+
@model_validator(mode="after")
|
|
161
|
+
def _validate_and_stamp(self) -> SimulationSpec:
|
|
162
|
+
if self.schema_version != SIMULATION_SPEC_SCHEMA_VERSION:
|
|
163
|
+
raise ValueError(
|
|
164
|
+
f"simulation_spec_version_unsupported: {self.schema_version}"
|
|
165
|
+
)
|
|
166
|
+
_reject_resolved_secrets(self.model_dump(exclude={"spec_hash"}))
|
|
167
|
+
expected = self.content_hash()
|
|
168
|
+
if self.spec_hash is not None and self.spec_hash != expected:
|
|
169
|
+
raise ValueError("simulation_spec_hash_mismatch")
|
|
170
|
+
object.__setattr__(self, "spec_hash", expected)
|
|
171
|
+
return self
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _reject_resolved_secrets(value: object, path: tuple[str, ...] = ()) -> None:
|
|
175
|
+
if isinstance(value, Mapping):
|
|
176
|
+
for key, item in value.items():
|
|
177
|
+
name = str(key).lower().replace("-", "_")
|
|
178
|
+
current_path = (*path, str(key))
|
|
179
|
+
if name == "secret_refs":
|
|
180
|
+
continue
|
|
181
|
+
if name in _SECRET_KEYS and item not in (None, "", {}, []):
|
|
182
|
+
raise ValueError("resolved_secret_forbidden: " + ".".join(current_path))
|
|
183
|
+
_reject_resolved_secrets(item, current_path)
|
|
184
|
+
elif isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
185
|
+
for index, item in enumerate(value):
|
|
186
|
+
_reject_resolved_secrets(item, (*path, str(index)))
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
from .models import Persona, Scenario, TestReport, TestCaseResult
|
|
2
|
+
from .runner import TestRunner
|
|
3
|
+
from .generator import ScenarioGenerator
|
|
4
|
+
from .synthetic import (
|
|
5
|
+
AttackDefinition,
|
|
6
|
+
AttackVector,
|
|
7
|
+
SyntheticDataGenerator,
|
|
8
|
+
SyntheticScenarioConfig,
|
|
9
|
+
SyntheticTrajectoryTemplateBundle,
|
|
10
|
+
SyntheticTrajectoryTemplateConfig,
|
|
11
|
+
SyntheticToolTaskBundle,
|
|
12
|
+
SyntheticToolTaskConfig,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"Persona",
|
|
17
|
+
"Scenario",
|
|
18
|
+
"TestReport",
|
|
19
|
+
"TestCaseResult",
|
|
20
|
+
"TestRunner",
|
|
21
|
+
"ScenarioGenerator",
|
|
22
|
+
"AttackDefinition",
|
|
23
|
+
"AttackVector",
|
|
24
|
+
"SyntheticDataGenerator",
|
|
25
|
+
"SyntheticScenarioConfig",
|
|
26
|
+
"SyntheticTrajectoryTemplateBundle",
|
|
27
|
+
"SyntheticTrajectoryTemplateConfig",
|
|
28
|
+
"SyntheticToolTaskBundle",
|
|
29
|
+
"SyntheticToolTaskConfig",
|
|
30
|
+
]
|