agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,419 @@
|
|
|
1
|
+
"""Unit 3 (BBG U3 / ARCH §2a) — the generic SIMULATION contract models.
|
|
2
|
+
|
|
3
|
+
``agent-learning.simulation.v1``: a typed, content-addressed world definition
|
|
4
|
+
that sits ABOVE the adapters (13D-D6). Engine-side home (AD-A): this module
|
|
5
|
+
imports only from ``.models`` / ``.goal_machine`` / stdlib and NEVER from
|
|
6
|
+
``fi.alk`` (the studio one-way rule). Canonicalization is the Persona
|
|
7
|
+
rule verbatim (AD-D); ``world`` (incl. every tool-mock block) is inside the
|
|
8
|
+
identity (R4/AD-O).
|
|
9
|
+
|
|
10
|
+
R5 dispositions honored: APPLY the A1/A3-A6/A8-A12/A16 shape fields + A2
|
|
11
|
+
emulator_backend discriminator; DEFER A15/A18 (TOOL_MOCK_LEVELS stays the
|
|
12
|
+
4-tuple, no 5th evidence class, emulated stays typed-only); STAGE A7
|
|
13
|
+
(GOAL_CHECK_KINDS stays the 5-kind set, imported from goal_machine); A19 (no
|
|
14
|
+
new world kinds — the 6-kind vocabulary is frozen).
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import hashlib
|
|
19
|
+
import json
|
|
20
|
+
from typing import Any, Dict, List, Mapping, Optional
|
|
21
|
+
|
|
22
|
+
from pydantic import BaseModel, Field, model_validator
|
|
23
|
+
|
|
24
|
+
from .goal_machine import ( # re-exported canon (ARCH §3; the gate byte-compares)
|
|
25
|
+
GOAL_CHECK_KINDS,
|
|
26
|
+
GOAL_CHECK_RUNGS,
|
|
27
|
+
GOAL_PREDICATE_OPS,
|
|
28
|
+
)
|
|
29
|
+
from .models import Persona, Scenario, ScenarioGoal, VerificationSpec
|
|
30
|
+
|
|
31
|
+
# Single-home canon re-exported here so contract consumers / the gate mirror can
|
|
32
|
+
# read goal-machine vocab from the contract module (ARCH §3).
|
|
33
|
+
__all__ = [
|
|
34
|
+
"GOAL_CHECK_KINDS", "GOAL_CHECK_RUNGS", "GOAL_PREDICATE_OPS",
|
|
35
|
+
"SIMULATION_KIND", "SIMULATION_CAST_ROLES", "SIMULATION_WORLD_KINDS",
|
|
36
|
+
"EXECUTABLE_WORLD_KINDS_V1", "TYPED_ONLY_WORLD_KINDS_V1", "TOOL_MOCK_LEVELS",
|
|
37
|
+
"EMULATOR_BACKENDS", "RECORDED_REPLAY_MISS_POLICIES", "STATE_CONSISTENCY_CLASSES",
|
|
38
|
+
"RESET_SEMANTICS", "REQUIRES_NETWORK_POLICIES", "ORACLE_SOLVER_KINDS",
|
|
39
|
+
"TOOL_CALL_ANSWERED_BY", "WORLD_RUNGS", "DYNAMICS_EVENT_KINDS",
|
|
40
|
+
"EPISODE_PERSISTENCE", "WORLD_EXECUTION_MODES",
|
|
41
|
+
"Simulation", "ScenarioBinding", "CastMember", "WorldSpec", "ToolBinding",
|
|
42
|
+
"ClockSpec", "DynamicsEvent", "EpisodeSpec", "AdmissionSpec",
|
|
43
|
+
"register_cast_role", "register_world_kind", "register_environment_type",
|
|
44
|
+
"resolved_cast_roles", "resolved_world_kinds",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
# ===========================================================================
|
|
48
|
+
# Canon constants (ARCH §3 — this is the single home; trinity.py mirrors them,
|
|
49
|
+
# the gate byte-compares; registration NEVER mutates them, AD-J).
|
|
50
|
+
# ===========================================================================
|
|
51
|
+
SIMULATION_KIND = "agent-learning.simulation.v1"
|
|
52
|
+
|
|
53
|
+
SIMULATION_CAST_ROLES = ("user", "opponent", "coworker", "counterpart")
|
|
54
|
+
|
|
55
|
+
SIMULATION_WORLD_KINDS = (
|
|
56
|
+
"conversation",
|
|
57
|
+
"tool_api",
|
|
58
|
+
"browser",
|
|
59
|
+
"computer_use",
|
|
60
|
+
"code_exec",
|
|
61
|
+
"voice_telephony",
|
|
62
|
+
)
|
|
63
|
+
EXECUTABLE_WORLD_KINDS_V1 = ("conversation", "tool_api")
|
|
64
|
+
TYPED_ONLY_WORLD_KINDS_V1 = ("browser", "computer_use", "code_exec", "voice_telephony")
|
|
65
|
+
|
|
66
|
+
# Tool-mock vocabulary — closed 4-level set UNCHANGED (A15 deferred).
|
|
67
|
+
TOOL_MOCK_LEVELS = ("static_fixture", "recorded_replay", "emulated", "live")
|
|
68
|
+
EMULATOR_BACKENDS = ("code", "prompted_lm", "finetuned_lm") # A2 sub-discriminator
|
|
69
|
+
RECORDED_REPLAY_MISS_POLICIES = ("fail", "fallthrough_emulated", "fallthrough_live", "re_record") # A3
|
|
70
|
+
STATE_CONSISTENCY_CLASSES = ("shared_programmatic", "per_call_context", "declared_preconditions") # A4
|
|
71
|
+
RESET_SEMANTICS = (
|
|
72
|
+
"stateless_fixture", "scripted_init", "image_copy", "snapshot_revert",
|
|
73
|
+
"memory_branch", "ephemeral_tenant", "container_provisioned",
|
|
74
|
+
) # A6 — v1 executes only the first two
|
|
75
|
+
REQUIRES_NETWORK_POLICIES = ("off", "allowlist", "proxy_recorded", "live") # A9
|
|
76
|
+
ORACLE_SOLVER_KINDS = ("script", "trajectory") # A10
|
|
77
|
+
TOOL_CALL_ANSWERED_BY = (
|
|
78
|
+
"fixture_hit", "cassette_hit", "emulated:code",
|
|
79
|
+
"emulated:prompted_lm", "emulated:finetuned_lm", "live",
|
|
80
|
+
) # A11
|
|
81
|
+
|
|
82
|
+
WORLD_RUNGS = (1, 2, 3) # rung→evidence: 1→local_gate, 2→captured_fixture, 3→live
|
|
83
|
+
DYNAMICS_EVENT_KINDS = ("env_state_patch", "counterpart_message", "tool_outcome_shift", "fault_profile")
|
|
84
|
+
EPISODE_PERSISTENCE = ("fresh", "carry_state", "carry_memory")
|
|
85
|
+
WORLD_EXECUTION_MODES = ("derived_legacy", "contract_native")
|
|
86
|
+
|
|
87
|
+
# ===========================================================================
|
|
88
|
+
# Extension registries (Appendix C-1): contract.py owns private tables with
|
|
89
|
+
# narrow setters; fi/alk/extensions.py (facade) is their ONLY writer
|
|
90
|
+
# (downward push). Built-ins shadow extensions at resolution; canon never
|
|
91
|
+
# mutates.
|
|
92
|
+
# ===========================================================================
|
|
93
|
+
_EXTRA_CAST_ROLES: Dict[str, dict] = {}
|
|
94
|
+
_EXTRA_WORLD_KINDS: Dict[str, dict] = {}
|
|
95
|
+
_EXTRA_ENVIRONMENT_TYPES: Dict[str, dict] = {}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def register_cast_role(name: str, record: Mapping[str, Any]) -> None:
|
|
99
|
+
_EXTRA_CAST_ROLES[str(name)] = dict(record)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def register_world_kind(name: str, record: Mapping[str, Any]) -> None:
|
|
103
|
+
_EXTRA_WORLD_KINDS[str(name)] = dict(record)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def register_environment_type(name: str, record: Mapping[str, Any]) -> None:
|
|
107
|
+
_EXTRA_ENVIRONMENT_TYPES[str(name)] = dict(record)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def resolved_cast_roles() -> tuple[str, ...]:
|
|
111
|
+
return tuple(SIMULATION_CAST_ROLES) + tuple(sorted(_EXTRA_CAST_ROLES))
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def resolved_world_kinds() -> tuple[str, ...]:
|
|
115
|
+
return tuple(SIMULATION_WORLD_KINDS) + tuple(sorted(_EXTRA_WORLD_KINDS))
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _reset_contract_extensions() -> None: # test-only
|
|
119
|
+
_EXTRA_CAST_ROLES.clear()
|
|
120
|
+
_EXTRA_WORLD_KINDS.clear()
|
|
121
|
+
_EXTRA_ENVIRONMENT_TYPES.clear()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# ===========================================================================
|
|
125
|
+
# Content-hash helper — the Persona rule verbatim (models.py:117-119), with
|
|
126
|
+
# 6-place float rounding applied to float leaves before dump.
|
|
127
|
+
# ===========================================================================
|
|
128
|
+
def _round_floats(value: Any) -> Any:
|
|
129
|
+
if isinstance(value, bool):
|
|
130
|
+
return value
|
|
131
|
+
if isinstance(value, float):
|
|
132
|
+
return round(value, 6)
|
|
133
|
+
if isinstance(value, Mapping):
|
|
134
|
+
return {k: _round_floats(v) for k, v in value.items()}
|
|
135
|
+
if isinstance(value, (list, tuple)):
|
|
136
|
+
return [_round_floats(v) for v in value]
|
|
137
|
+
return value
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _content_hash(payload: Mapping[str, Any]) -> str:
|
|
141
|
+
rounded = _round_floats(dict(payload))
|
|
142
|
+
canonical = json.dumps(rounded, sort_keys=True, separators=(",", ":"), default=str)
|
|
143
|
+
return "sha256:" + hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# ===========================================================================
|
|
147
|
+
# Models (ARCH §2a field tables — verbatim; no field added/dropped/renamed).
|
|
148
|
+
# ===========================================================================
|
|
149
|
+
class ToolBinding(BaseModel):
|
|
150
|
+
"""First-class tool mocking (R4).
|
|
151
|
+
|
|
152
|
+
Note: the ``schema`` field name is ARCH §2a verbatim; Pydantic emits a
|
|
153
|
+
benign class-definition warning because it shadows ``BaseModel.schema``.
|
|
154
|
+
Renaming is forbidden by the BBG ("do not add, drop, or rename a field")
|
|
155
|
+
and the field works correctly (validators/serialization unaffected).
|
|
156
|
+
"""
|
|
157
|
+
name: str
|
|
158
|
+
schema: Optional[Dict[str, Any]] = None
|
|
159
|
+
mock: Dict[str, Any] = Field(default_factory=dict) # {level, source, emulator_backend, ...}
|
|
160
|
+
required_env: List[str] = Field(default_factory=list)
|
|
161
|
+
requires: Optional[Dict[str, Any]] = None # A9 per-rung block
|
|
162
|
+
|
|
163
|
+
@model_validator(mode="after")
|
|
164
|
+
def _validate_mock(self) -> "ToolBinding":
|
|
165
|
+
level = self.mock.get("level")
|
|
166
|
+
if not level:
|
|
167
|
+
raise ValueError(
|
|
168
|
+
"tool_mock_level_undeclared: ToolBinding.mock.level is required "
|
|
169
|
+
f"(one of {TOOL_MOCK_LEVELS}) for tool {self.name!r}"
|
|
170
|
+
)
|
|
171
|
+
if level not in TOOL_MOCK_LEVELS:
|
|
172
|
+
raise ValueError(
|
|
173
|
+
f"tool_mock_level_undeclared: mock.level {level!r} not in {TOOL_MOCK_LEVELS}"
|
|
174
|
+
)
|
|
175
|
+
if level == "recorded_replay":
|
|
176
|
+
source = self.mock.get("source")
|
|
177
|
+
prov = self.mock.get("provenance") or {}
|
|
178
|
+
if not source or not prov.get("capture"):
|
|
179
|
+
raise ValueError(
|
|
180
|
+
"tool_mock_replay_missing: recorded_replay requires mock.source "
|
|
181
|
+
"(scrubbed capture ref) + mock.provenance.capture (sha256)"
|
|
182
|
+
)
|
|
183
|
+
sub = self.mock.get("recorded_replay") or {}
|
|
184
|
+
miss = sub.get("miss_policy")
|
|
185
|
+
if miss is not None and miss not in RECORDED_REPLAY_MISS_POLICIES:
|
|
186
|
+
raise ValueError(
|
|
187
|
+
f"tool_mock_replay_missing: miss_policy {miss!r} not in "
|
|
188
|
+
f"{RECORDED_REPLAY_MISS_POLICIES}"
|
|
189
|
+
)
|
|
190
|
+
if level == "emulated":
|
|
191
|
+
backend = self.mock.get("emulator_backend")
|
|
192
|
+
if backend is not None and backend not in EMULATOR_BACKENDS:
|
|
193
|
+
raise ValueError(
|
|
194
|
+
f"tool_mock_level_undeclared: emulator_backend {backend!r} not in "
|
|
195
|
+
f"{EMULATOR_BACKENDS}"
|
|
196
|
+
)
|
|
197
|
+
if level == "live" and not self.required_env:
|
|
198
|
+
raise ValueError(
|
|
199
|
+
"tool_mock_live_unkeyed: live mock requires required_env NAMES "
|
|
200
|
+
f"for tool {self.name!r} (lane-only; refused in gate/release)"
|
|
201
|
+
)
|
|
202
|
+
return self
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
class WorldSpec(BaseModel):
|
|
206
|
+
"""Typed world: kind + environments + tools (R4)."""
|
|
207
|
+
kind: str
|
|
208
|
+
environments: List[Dict[str, Any]] = Field(default_factory=list)
|
|
209
|
+
spec: Dict[str, Any] = Field(default_factory=dict)
|
|
210
|
+
tools: List[ToolBinding] = Field(default_factory=list)
|
|
211
|
+
rung: int = 1
|
|
212
|
+
state_consistency: str = "shared_programmatic" # A4
|
|
213
|
+
reset_semantics: str = "stateless_fixture" # A6
|
|
214
|
+
perturbation_profile: Optional[Dict[str, Any]] = None # A8
|
|
215
|
+
policies: List[Dict[str, Any]] = Field(default_factory=list) # A16
|
|
216
|
+
stochasticity_profile: Optional[Dict[str, Any]] = None # A12
|
|
217
|
+
|
|
218
|
+
@model_validator(mode="after")
|
|
219
|
+
def _validate_world(self) -> "WorldSpec":
|
|
220
|
+
if self.kind not in resolved_world_kinds():
|
|
221
|
+
raise ValueError(
|
|
222
|
+
f"world_kind_unsupported: world.kind {self.kind!r} not in "
|
|
223
|
+
f"{resolved_world_kinds()}"
|
|
224
|
+
)
|
|
225
|
+
if self.rung not in WORLD_RUNGS:
|
|
226
|
+
raise ValueError(f"world.rung {self.rung!r} not in {WORLD_RUNGS}")
|
|
227
|
+
if self.state_consistency not in STATE_CONSISTENCY_CLASSES:
|
|
228
|
+
raise ValueError(
|
|
229
|
+
f"world.state_consistency {self.state_consistency!r} not in "
|
|
230
|
+
f"{STATE_CONSISTENCY_CLASSES}"
|
|
231
|
+
)
|
|
232
|
+
if self.reset_semantics not in RESET_SEMANTICS:
|
|
233
|
+
raise ValueError(
|
|
234
|
+
f"world.reset_semantics {self.reset_semantics!r} not in {RESET_SEMANTICS}"
|
|
235
|
+
)
|
|
236
|
+
return self
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
class ClockSpec(BaseModel):
|
|
240
|
+
model: str = "turn"
|
|
241
|
+
step_s: Optional[float] = None
|
|
242
|
+
horizon: Dict[str, Any] = Field(default_factory=dict)
|
|
243
|
+
|
|
244
|
+
@model_validator(mode="after")
|
|
245
|
+
def _validate_clock(self) -> "ClockSpec":
|
|
246
|
+
if self.model not in ("turn", "simulated"):
|
|
247
|
+
raise ValueError(f"clock.model {self.model!r} not in ('turn', 'simulated')")
|
|
248
|
+
if self.model == "simulated" and (self.step_s is None or self.step_s <= 0):
|
|
249
|
+
raise ValueError("clock.step_s is required (>0) when model == 'simulated'")
|
|
250
|
+
return self
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
class DynamicsEvent(BaseModel):
|
|
254
|
+
at: Dict[str, Any]
|
|
255
|
+
event: str
|
|
256
|
+
payload: Dict[str, Any] = Field(default_factory=dict)
|
|
257
|
+
seed: Optional[int] = None
|
|
258
|
+
provenance: Dict[str, Any] = Field(default_factory=dict)
|
|
259
|
+
|
|
260
|
+
@model_validator(mode="after")
|
|
261
|
+
def _validate_dynamics(self) -> "DynamicsEvent":
|
|
262
|
+
if self.event not in DYNAMICS_EVENT_KINDS:
|
|
263
|
+
raise ValueError(
|
|
264
|
+
f"dynamics.event {self.event!r} not in {DYNAMICS_EVENT_KINDS}"
|
|
265
|
+
)
|
|
266
|
+
keys = set(self.at)
|
|
267
|
+
valid = (
|
|
268
|
+
keys == {"turn"}
|
|
269
|
+
or keys == {"time_s"}
|
|
270
|
+
or keys <= {"every", "phase"} and "every" in keys
|
|
271
|
+
)
|
|
272
|
+
if not valid:
|
|
273
|
+
raise ValueError(
|
|
274
|
+
"dynamics.at must be exactly one of {turn}, {time_s}, {every, phase?}"
|
|
275
|
+
)
|
|
276
|
+
return self
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
class EpisodeSpec(BaseModel):
|
|
280
|
+
count: int = 1
|
|
281
|
+
persistence: str = "fresh"
|
|
282
|
+
settle: List[Any] = Field(default_factory=list)
|
|
283
|
+
|
|
284
|
+
@model_validator(mode="after")
|
|
285
|
+
def _validate_episode(self) -> "EpisodeSpec":
|
|
286
|
+
if self.count < 1:
|
|
287
|
+
raise ValueError("episodes.count must be >= 1")
|
|
288
|
+
if self.persistence not in EPISODE_PERSISTENCE:
|
|
289
|
+
raise ValueError(
|
|
290
|
+
f"episodes.persistence {self.persistence!r} not in {EPISODE_PERSISTENCE}"
|
|
291
|
+
)
|
|
292
|
+
return self
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
class AdmissionSpec(BaseModel):
|
|
296
|
+
fidelity_floors: Dict[str, float] = Field(default_factory=dict)
|
|
297
|
+
oracle_adequacy: Optional[Dict[str, Any]] = None
|
|
298
|
+
realism_certificate: Optional[str] = None
|
|
299
|
+
epidemic_rate: Optional[float] = None
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
class CastMember(BaseModel):
|
|
303
|
+
"""R2: a cast member holds turns; a dynamics entry never does."""
|
|
304
|
+
persona: str # persona version hash (ref into simulation.personas)
|
|
305
|
+
role: str = "user"
|
|
306
|
+
alias: Optional[str] = None
|
|
307
|
+
|
|
308
|
+
@model_validator(mode="after")
|
|
309
|
+
def _validate_role(self) -> "CastMember":
|
|
310
|
+
if self.role not in resolved_cast_roles():
|
|
311
|
+
raise ValueError(
|
|
312
|
+
f"cast_role_unknown: role {self.role!r} not in {resolved_cast_roles()}"
|
|
313
|
+
)
|
|
314
|
+
return self
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
class ScenarioBinding(BaseModel):
|
|
318
|
+
"""Cast selection wrapping a Scenario (AD-B: Scenario stays byte-stable)."""
|
|
319
|
+
scenario: Optional[Scenario] = None
|
|
320
|
+
scenario_ref: Optional[str] = None
|
|
321
|
+
cast: List[CastMember]
|
|
322
|
+
casting: str = "each"
|
|
323
|
+
goal: Optional[ScenarioGoal] = None
|
|
324
|
+
verification: Optional[VerificationSpec] = None
|
|
325
|
+
oracle_solver: Optional[Dict[str, Any]] = None # A10
|
|
326
|
+
weight: float = 1.0
|
|
327
|
+
|
|
328
|
+
@model_validator(mode="after")
|
|
329
|
+
def _validate_binding(self) -> "ScenarioBinding":
|
|
330
|
+
if not self.cast:
|
|
331
|
+
raise ValueError("simulation_contract_invalid: ScenarioBinding.cast requires >= 1 member")
|
|
332
|
+
if self.casting not in ("each", "together"):
|
|
333
|
+
raise ValueError(f"simulation_contract_invalid: casting {self.casting!r} not in ('each','together')")
|
|
334
|
+
if self.weight <= 0:
|
|
335
|
+
raise ValueError("simulation_contract_invalid: ScenarioBinding.weight must be > 0")
|
|
336
|
+
if self.oracle_solver is not None:
|
|
337
|
+
kind = self.oracle_solver.get("kind")
|
|
338
|
+
if kind is not None and kind not in ORACLE_SOLVER_KINDS:
|
|
339
|
+
raise ValueError(
|
|
340
|
+
f"simulation_contract_invalid: oracle_solver.kind {kind!r} not in {ORACLE_SOLVER_KINDS}"
|
|
341
|
+
)
|
|
342
|
+
# contract-native scenarios MUST have empty/absent legacy dataset
|
|
343
|
+
if self.scenario is not None and self.scenario.dataset:
|
|
344
|
+
# auto-lift produces non-empty dataset; flag only when explicitly built.
|
|
345
|
+
pass
|
|
346
|
+
return self
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
class Simulation(BaseModel):
|
|
350
|
+
"""The top-level ``agent-learning.simulation.v1`` object."""
|
|
351
|
+
kind: str = SIMULATION_KIND
|
|
352
|
+
name: str
|
|
353
|
+
description: Optional[str] = None
|
|
354
|
+
version: Optional[str] = None
|
|
355
|
+
personas: List[Persona] = Field(default_factory=list)
|
|
356
|
+
scenarios: List[ScenarioBinding]
|
|
357
|
+
world: WorldSpec
|
|
358
|
+
clock: ClockSpec = Field(default_factory=ClockSpec)
|
|
359
|
+
dynamics: List[DynamicsEvent] = Field(default_factory=list)
|
|
360
|
+
episodes: EpisodeSpec = Field(default_factory=EpisodeSpec)
|
|
361
|
+
goal: Optional[ScenarioGoal] = None
|
|
362
|
+
verification: Optional[VerificationSpec] = None
|
|
363
|
+
objective: Optional[Dict[str, Any]] = None
|
|
364
|
+
admission: AdmissionSpec = Field(default_factory=AdmissionSpec)
|
|
365
|
+
seed: Optional[int] = None
|
|
366
|
+
provenance: Dict[str, Any] = Field(default_factory=dict)
|
|
367
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
368
|
+
|
|
369
|
+
def content_hash(self) -> str:
|
|
370
|
+
payload = self.model_dump(exclude={"version"}, exclude_none=True)
|
|
371
|
+
return _content_hash(payload)
|
|
372
|
+
|
|
373
|
+
@model_validator(mode="after")
|
|
374
|
+
def _validate_and_stamp(self) -> "Simulation":
|
|
375
|
+
if self.kind != SIMULATION_KIND:
|
|
376
|
+
raise ValueError(f"simulation_contract_invalid: kind must be {SIMULATION_KIND!r}")
|
|
377
|
+
# duplicate persona version hashes rejected
|
|
378
|
+
seen: set[str] = set()
|
|
379
|
+
persona_hashes: set[str] = set()
|
|
380
|
+
for persona in self.personas:
|
|
381
|
+
digest = persona.version or persona.content_hash()
|
|
382
|
+
if digest in seen:
|
|
383
|
+
raise ValueError(
|
|
384
|
+
"simulation_contract_invalid: duplicate persona version hashes in personas"
|
|
385
|
+
)
|
|
386
|
+
seen.add(digest)
|
|
387
|
+
persona_hashes.add(digest)
|
|
388
|
+
# every cast ref resolves into the persona set (closed world)
|
|
389
|
+
for binding in self.scenarios:
|
|
390
|
+
for member in binding.cast:
|
|
391
|
+
if member.persona not in persona_hashes:
|
|
392
|
+
raise ValueError(
|
|
393
|
+
f"simulation_contract_invalid: cast persona ref {member.persona!r} "
|
|
394
|
+
"does not resolve into simulation.personas (closed world)"
|
|
395
|
+
)
|
|
396
|
+
turn_holders = len(binding.cast)
|
|
397
|
+
if binding.casting == "each" and turn_holders < 1:
|
|
398
|
+
raise ValueError(
|
|
399
|
+
"simulation_contract_invalid: casting 'each' requires >= 1 turn-holder"
|
|
400
|
+
)
|
|
401
|
+
# casting 'together' validates structurally; execution refuses until U23c.
|
|
402
|
+
# R2 litmus (ambient side): a dynamics payload may not declare turn-holding
|
|
403
|
+
# capability (responds_to / utterance templates) — that is a persona.
|
|
404
|
+
for evt in self.dynamics:
|
|
405
|
+
if evt.event == "counterpart_message":
|
|
406
|
+
payload = evt.payload or {}
|
|
407
|
+
if "responds_to" in payload or "utterance_templates" in payload:
|
|
408
|
+
raise ValueError(
|
|
409
|
+
"counterpart_misclassified: a dynamics entry never holds a turn "
|
|
410
|
+
"(responds_to/utterance_templates ⇒ this is a persona with a role, "
|
|
411
|
+
"not a dynamics event). R2 litmus: does it hold a turn?"
|
|
412
|
+
)
|
|
413
|
+
# objective is structure-only on the engine side (semantic validation lives
|
|
414
|
+
# facade-side in loss.py — the engine never imports the loss module, AD-A).
|
|
415
|
+
if self.objective is not None and not isinstance(self.objective, Mapping):
|
|
416
|
+
raise ValueError("simulation_contract_invalid: objective must be a mapping")
|
|
417
|
+
if self.version is None:
|
|
418
|
+
object.__setattr__(self, "version", self.content_hash())
|
|
419
|
+
return self
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from fi.simulate.simulation.engines.base import BaseEngine
|
|
2
|
+
from fi.simulate.simulation.engines.cloud import CloudEngine
|
|
3
|
+
from fi.simulate.simulation.engines.local_text import LocalTextEngine
|
|
4
|
+
|
|
5
|
+
# LiveKit is an optional dependency. Keep cloud-mode imports working even when
|
|
6
|
+
# LiveKit isn't installed (or version mismatches exist).
|
|
7
|
+
try: # pragma: no cover
|
|
8
|
+
from fi.simulate.simulation.engines.livekit import LiveKitEngine
|
|
9
|
+
except ImportError: # pragma: no cover
|
|
10
|
+
LiveKitEngine = None # type: ignore
|
|
11
|
+
|
|
12
|
+
__all__ = ["BaseEngine", "CloudEngine", "LiveKitEngine", "LocalTextEngine"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
|
|
2
|
+
from abc import ABC, abstractmethod
|
|
3
|
+
from typing import Optional
|
|
4
|
+
from fi.simulate.agent.definition import AgentDefinition, SimulatorAgentDefinition
|
|
5
|
+
from fi.simulate.simulation.models import Scenario, TestReport
|
|
6
|
+
|
|
7
|
+
class BaseEngine(ABC):
|
|
8
|
+
"""
|
|
9
|
+
Abstract base class for simulation engines.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
@abstractmethod
|
|
13
|
+
async def run(
|
|
14
|
+
self,
|
|
15
|
+
agent_definition: Optional[AgentDefinition] = None,
|
|
16
|
+
scenario: Optional[Scenario] = None,
|
|
17
|
+
simulator: Optional[SimulatorAgentDefinition] = None,
|
|
18
|
+
**kwargs
|
|
19
|
+
) -> TestReport:
|
|
20
|
+
pass
|
|
21
|
+
|