agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""Child process a ``simulation-runner`` worker spawns per hosted job.
|
|
2
|
+
|
|
3
|
+
python -m fi.simulate.hosted.child_entrypoint <job.json> [--status-file PATH]
|
|
4
|
+
|
|
5
|
+
It runs the released SDK for the job's mode and submits results through
|
|
6
|
+
``FutureAGIResultSink``. It is the only place hosted execution differs from a
|
|
7
|
+
local run — the simulation itself is the same ``SimulationRunner``/engine code.
|
|
8
|
+
|
|
9
|
+
Lifecycle is reported as newline-delimited JSON ``RunnerJobStatus`` objects, both
|
|
10
|
+
to stdout (the worker tails these for Temporal heartbeats) and to an optional
|
|
11
|
+
status file. SIGTERM triggers a graceful cancel + cleanup.
|
|
12
|
+
|
|
13
|
+
Slice 1 wires the chat mode only; the voice modes raise until their slices land.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import asyncio
|
|
20
|
+
import json
|
|
21
|
+
import logging
|
|
22
|
+
import os
|
|
23
|
+
import signal
|
|
24
|
+
import sys
|
|
25
|
+
from datetime import datetime, timezone
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import TYPE_CHECKING, Any
|
|
28
|
+
|
|
29
|
+
from fi.simulate.results.futureagi import FutureAGIResultSink
|
|
30
|
+
from fi.simulate.runtime.report import SimulationReport
|
|
31
|
+
from fi.simulate.runtime.run import RunStatus
|
|
32
|
+
from fi.simulate.runtime.runner import SimulationRunner
|
|
33
|
+
|
|
34
|
+
from .job import RunnerJobPhase, RunnerJobStatus, RunnerMode, StartRunnerJob
|
|
35
|
+
from .targets import resolve_chat_target
|
|
36
|
+
|
|
37
|
+
if TYPE_CHECKING:
|
|
38
|
+
from fi.simulate.runtime.spec import SimulationSpec
|
|
39
|
+
|
|
40
|
+
_HEARTBEAT_INTERVAL_SECONDS = 10.0
|
|
41
|
+
_CANCEL_GRACE_SECONDS = 30.0
|
|
42
|
+
|
|
43
|
+
logger = logging.getLogger("fi.simulate.hosted.runner")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _job_log_fields(job: StartRunnerJob) -> dict[str, Any]:
|
|
47
|
+
fields: dict[str, Any] = {"job_id": job.job_id, "mode": job.mode.value}
|
|
48
|
+
if job.voice is not None:
|
|
49
|
+
target = dict(job.voice.agent_definition or {}).get("target") or {}
|
|
50
|
+
fields["provider"] = target.get("provider")
|
|
51
|
+
dataset = dict(job.voice.scenario or {}).get("dataset") or []
|
|
52
|
+
fields["cases"] = len(dataset)
|
|
53
|
+
if job.sink is not None:
|
|
54
|
+
fields["run_test_id"] = job.sink.run_test_id
|
|
55
|
+
fields["test_execution_id"] = job.sink.test_execution_id
|
|
56
|
+
return fields
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class _StatusReporter:
|
|
60
|
+
def __init__(self, job_id: str, status_file: Path | None) -> None:
|
|
61
|
+
self._job_id = job_id
|
|
62
|
+
self._status_file = status_file
|
|
63
|
+
|
|
64
|
+
def emit(
|
|
65
|
+
self,
|
|
66
|
+
phase: RunnerJobPhase,
|
|
67
|
+
*,
|
|
68
|
+
detail: str | None = None,
|
|
69
|
+
report_hash: str | None = None,
|
|
70
|
+
submission_status: str | None = None,
|
|
71
|
+
) -> None:
|
|
72
|
+
status = RunnerJobStatus(
|
|
73
|
+
job_id=self._job_id,
|
|
74
|
+
phase=phase,
|
|
75
|
+
detail=detail,
|
|
76
|
+
report_hash=report_hash,
|
|
77
|
+
submission_status=submission_status,
|
|
78
|
+
updated_at=datetime.now(timezone.utc),
|
|
79
|
+
)
|
|
80
|
+
line = status.model_dump_json()
|
|
81
|
+
print(line, flush=True)
|
|
82
|
+
if self._status_file is not None:
|
|
83
|
+
with self._status_file.open("a", encoding="utf-8") as handle:
|
|
84
|
+
handle.write(line + "\n")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _load_job(path: Path) -> StartRunnerJob:
|
|
88
|
+
return StartRunnerJob.model_validate_json(path.read_text(encoding="utf-8"))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _build_sink(job: StartRunnerJob) -> FutureAGIResultSink:
|
|
92
|
+
root = job.sink.root_directory or os.environ.get("FI_RUN_ROOT") or ".fagi/runs"
|
|
93
|
+
return FutureAGIResultSink(
|
|
94
|
+
root=root,
|
|
95
|
+
api_url=job.sink.api_url,
|
|
96
|
+
run_test_id=job.sink.run_test_id,
|
|
97
|
+
test_execution_id=job.sink.test_execution_id,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _read_submission(run_directory: Path | None) -> dict[str, Any]:
|
|
102
|
+
if run_directory is None:
|
|
103
|
+
return {}
|
|
104
|
+
submission_path = run_directory / "submission.json"
|
|
105
|
+
if not submission_path.exists():
|
|
106
|
+
return {}
|
|
107
|
+
try:
|
|
108
|
+
return json.loads(submission_path.read_text(encoding="utf-8"))
|
|
109
|
+
except (ValueError, OSError):
|
|
110
|
+
return {}
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _build_voice_spec(job: StartRunnerJob) -> "SimulationSpec":
|
|
114
|
+
"""Translate a voice job into a ``SimulationSpec`` so the voice run flows
|
|
115
|
+
through the same ``SimulationRunner`` spine as chat (plan §3). The typed voice
|
|
116
|
+
inputs ride in ``environment.config`` — secret-free, since providers are
|
|
117
|
+
referenced by ``*_env`` name, never raw values. ``transport.kind`` selects the
|
|
118
|
+
target adapter. The DID pool is leased by the runner activity (telephone
|
|
119
|
+
only), not here — the leased number arrives via the agent definition / params.
|
|
120
|
+
"""
|
|
121
|
+
from fi.simulate.runtime import new_run_id
|
|
122
|
+
from fi.simulate.runtime.spec import (
|
|
123
|
+
AgentEndpointSpec,
|
|
124
|
+
EnvironmentSpec,
|
|
125
|
+
EvidencePolicy,
|
|
126
|
+
ExecutionPolicy,
|
|
127
|
+
SimulationSpec,
|
|
128
|
+
SimulatorPolicySpec,
|
|
129
|
+
TimeoutPolicy,
|
|
130
|
+
)
|
|
131
|
+
from fi.simulate.simulation.models import Scenario
|
|
132
|
+
|
|
133
|
+
cfg = job.voice
|
|
134
|
+
run_id = str((job.spec.run_id if job.spec else None) or new_run_id())
|
|
135
|
+
params = dict(cfg.params or {})
|
|
136
|
+
transport = (dict(cfg.agent_definition or {}).get("transport") or {})
|
|
137
|
+
transport_kind = transport.get("kind") or "livekit"
|
|
138
|
+
|
|
139
|
+
# The runner's outer deadline must clear the voice call's own budget.
|
|
140
|
+
run_seconds = max(
|
|
141
|
+
300.0,
|
|
142
|
+
float(params.get("max_seconds", 45.0))
|
|
143
|
+
+ float(params.get("connect_timeout", 15.0))
|
|
144
|
+
+ float(params.get("readiness_timeout", 30.0))
|
|
145
|
+
+ float(params.get("cleanup_timeout", 30.0))
|
|
146
|
+
+ 60.0,
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
return SimulationSpec(
|
|
150
|
+
run_id=run_id,
|
|
151
|
+
environment=EnvironmentSpec(
|
|
152
|
+
adapter="voice",
|
|
153
|
+
world_kind="voice",
|
|
154
|
+
config={
|
|
155
|
+
"agent_definition": cfg.agent_definition,
|
|
156
|
+
"livekit_runtime": cfg.livekit_runtime,
|
|
157
|
+
"simulator": cfg.simulator,
|
|
158
|
+
"params": cfg.params,
|
|
159
|
+
},
|
|
160
|
+
),
|
|
161
|
+
target=AgentEndpointSpec(adapter=transport_kind),
|
|
162
|
+
simulator=SimulatorPolicySpec(adapter="livekit_simulator"),
|
|
163
|
+
scenario=Scenario.model_validate(cfg.scenario),
|
|
164
|
+
execution=ExecutionPolicy(timeout=TimeoutPolicy(run_seconds=run_seconds)),
|
|
165
|
+
evidence=EvidencePolicy(),
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
async def _heartbeat(reporter: _StatusReporter) -> None:
|
|
170
|
+
while True:
|
|
171
|
+
await asyncio.sleep(_HEARTBEAT_INTERVAL_SECONDS)
|
|
172
|
+
reporter.emit(RunnerJobPhase.RUNNING, detail="heartbeat")
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
async def _execute(job: StartRunnerJob, reporter: _StatusReporter) -> int:
|
|
176
|
+
reporter.emit(RunnerJobPhase.PREPARING)
|
|
177
|
+
logger.info("hosted job start", extra=_job_log_fields(job))
|
|
178
|
+
sink = _build_sink(job)
|
|
179
|
+
|
|
180
|
+
if job.mode is RunnerMode.CHAT:
|
|
181
|
+
target = resolve_chat_target(job.spec)
|
|
182
|
+
run_coro = SimulationRunner().run(job.spec, target=target, result_sink=sink)
|
|
183
|
+
elif job.mode.is_voice:
|
|
184
|
+
run_coro = SimulationRunner().run(_build_voice_spec(job), result_sink=sink)
|
|
185
|
+
else:
|
|
186
|
+
raise NotImplementedError(f"runner mode not wired: {job.mode.value}")
|
|
187
|
+
|
|
188
|
+
run_task = asyncio.ensure_future(run_coro)
|
|
189
|
+
heartbeat_task = asyncio.ensure_future(_heartbeat(reporter))
|
|
190
|
+
reporter.emit(RunnerJobPhase.RUNNING)
|
|
191
|
+
try:
|
|
192
|
+
report: SimulationReport = await run_task
|
|
193
|
+
except asyncio.CancelledError:
|
|
194
|
+
reporter.emit(RunnerJobPhase.CANCELED, detail="cancelled")
|
|
195
|
+
logger.warning("hosted job cancelled", extra={"job_id": job.job_id})
|
|
196
|
+
# Cancelling this coroutine does not cancel ``run_task``; without an
|
|
197
|
+
# explicit cancel ``asyncio.run`` shutdown waits on it forever and the
|
|
198
|
+
# child leaks past SIGTERM.
|
|
199
|
+
run_task.cancel()
|
|
200
|
+
try:
|
|
201
|
+
await asyncio.wait({run_task}, timeout=_CANCEL_GRACE_SECONDS)
|
|
202
|
+
except asyncio.CancelledError:
|
|
203
|
+
pass
|
|
204
|
+
if not run_task.done():
|
|
205
|
+
os._exit(2)
|
|
206
|
+
raise
|
|
207
|
+
finally:
|
|
208
|
+
heartbeat_task.cancel()
|
|
209
|
+
|
|
210
|
+
reporter.emit(RunnerJobPhase.FINALIZING)
|
|
211
|
+
submission = _read_submission(sink.run_directory)
|
|
212
|
+
submission_status = submission.get("status")
|
|
213
|
+
run_completed = report.status is RunStatus.COMPLETED
|
|
214
|
+
# When a submission target is configured (a hosted run), a failed or omitted
|
|
215
|
+
# submission is a job failure — otherwise a broken upload reports as green.
|
|
216
|
+
submission_expected = bool(job.sink.run_test_id)
|
|
217
|
+
submission_ok = (not submission_expected) or submission_status == "submitted"
|
|
218
|
+
completed = run_completed and submission_ok
|
|
219
|
+
if completed:
|
|
220
|
+
detail = None
|
|
221
|
+
elif not run_completed:
|
|
222
|
+
detail = report.failure.code if report.failure else "run_failed"
|
|
223
|
+
else:
|
|
224
|
+
detail = f"submission_{submission_status or 'missing'}"
|
|
225
|
+
outcome_fields = {
|
|
226
|
+
**_job_log_fields(job),
|
|
227
|
+
"run_status": getattr(report.status, "value", str(report.status)),
|
|
228
|
+
"submission_status": submission_status,
|
|
229
|
+
"report_hash": report.report_hash,
|
|
230
|
+
"detail": detail,
|
|
231
|
+
}
|
|
232
|
+
if completed:
|
|
233
|
+
logger.info("hosted job completed", extra=outcome_fields)
|
|
234
|
+
else:
|
|
235
|
+
logger.error("hosted job failed", extra=outcome_fields)
|
|
236
|
+
reporter.emit(
|
|
237
|
+
RunnerJobPhase.COMPLETED if completed else RunnerJobPhase.FAILED,
|
|
238
|
+
detail=detail,
|
|
239
|
+
report_hash=report.report_hash,
|
|
240
|
+
submission_status=submission_status,
|
|
241
|
+
)
|
|
242
|
+
return 0 if completed else 1
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _install_cancellation(run_task_holder: dict[str, asyncio.Task[int]]) -> None:
|
|
246
|
+
loop = asyncio.get_running_loop()
|
|
247
|
+
|
|
248
|
+
def _cancel() -> None:
|
|
249
|
+
task = run_task_holder.get("task")
|
|
250
|
+
if task is not None and not task.done():
|
|
251
|
+
task.cancel()
|
|
252
|
+
|
|
253
|
+
for sig in (signal.SIGTERM, signal.SIGINT):
|
|
254
|
+
try:
|
|
255
|
+
loop.add_signal_handler(sig, _cancel)
|
|
256
|
+
except (NotImplementedError, ValueError):
|
|
257
|
+
pass
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
async def _main_async(job: StartRunnerJob, reporter: _StatusReporter) -> int:
|
|
261
|
+
holder: dict[str, asyncio.Task[int]] = {}
|
|
262
|
+
_install_cancellation(holder)
|
|
263
|
+
task = asyncio.ensure_future(_execute(job, reporter))
|
|
264
|
+
holder["task"] = task
|
|
265
|
+
try:
|
|
266
|
+
return await task
|
|
267
|
+
except asyncio.CancelledError:
|
|
268
|
+
return 2
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _configure_logging() -> None:
|
|
272
|
+
"""The child runs with no logging config, so INFO seams (job start/outcome,
|
|
273
|
+
engine dispatch/join/stop_reason) were silently dropped by the WARNING-level
|
|
274
|
+
lastResort handler and never reached the runner's log capture."""
|
|
275
|
+
root = logging.getLogger()
|
|
276
|
+
if not root.handlers:
|
|
277
|
+
handler = logging.StreamHandler(sys.stderr)
|
|
278
|
+
handler.setFormatter(logging.Formatter("%(levelname)s:%(name)s:%(message)s"))
|
|
279
|
+
root.addHandler(handler)
|
|
280
|
+
root.setLevel(logging.WARNING)
|
|
281
|
+
logging.getLogger("fi.simulate").setLevel(logging.INFO)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def main(argv: list[str] | None = None) -> int:
|
|
285
|
+
_configure_logging()
|
|
286
|
+
parser = argparse.ArgumentParser(prog="fi.simulate.hosted.child_entrypoint")
|
|
287
|
+
parser.add_argument("job", help="path to the StartRunnerJob JSON file")
|
|
288
|
+
parser.add_argument("--status-file", default=None)
|
|
289
|
+
args = parser.parse_args(argv)
|
|
290
|
+
|
|
291
|
+
job = _load_job(Path(args.job))
|
|
292
|
+
status_file = Path(args.status_file) if args.status_file else None
|
|
293
|
+
reporter = _StatusReporter(job.job_id, status_file)
|
|
294
|
+
|
|
295
|
+
try:
|
|
296
|
+
return asyncio.run(_main_async(job, reporter))
|
|
297
|
+
except Exception as exc: # noqa: BLE001
|
|
298
|
+
logger.exception("hosted job crashed", extra={"job_id": job.job_id})
|
|
299
|
+
reporter.emit(
|
|
300
|
+
RunnerJobPhase.FAILED, detail=f"{type(exc).__name__}: {exc}"
|
|
301
|
+
)
|
|
302
|
+
return 1
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
if __name__ == "__main__":
|
|
306
|
+
sys.exit(main())
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Hosted-runner job contracts (plan §9.1).
|
|
2
|
+
|
|
3
|
+
A ``StartRunnerJob`` is the serializable unit the platform hands to a
|
|
4
|
+
``simulation-runner`` worker. It embeds an immutable ``SimulationSpec`` plus the
|
|
5
|
+
result-sink target; it carries only ``SecretRef``s, never resolved secrets (the
|
|
6
|
+
runner resolves those into the child process environment). The child process
|
|
7
|
+
(``fi.simulate.hosted.child_entrypoint``) consumes exactly this model.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from datetime import datetime
|
|
13
|
+
from enum import Enum
|
|
14
|
+
from typing import Protocol
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel, Field, JsonValue, model_validator
|
|
17
|
+
|
|
18
|
+
from fi.simulate.runtime.spec import SecretRef, SimulationSpec
|
|
19
|
+
|
|
20
|
+
RUNNER_JOB_SCHEMA_VERSION = "futureagi.runner-job.v1"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RunnerMode(str, Enum):
|
|
24
|
+
"""Execution mode the child selects an engine for. Only the SIP mode
|
|
25
|
+
leases a phone-number slot; chat and WebRTC never touch the pool."""
|
|
26
|
+
|
|
27
|
+
CHAT = "chat"
|
|
28
|
+
VOICE_WEBRTC = "voice_webrtc"
|
|
29
|
+
VOICE_SIP = "voice_sip"
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def needs_phone(self) -> bool:
|
|
33
|
+
return self is RunnerMode.VOICE_SIP
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def is_voice(self) -> bool:
|
|
37
|
+
return self in {RunnerMode.VOICE_WEBRTC, RunnerMode.VOICE_SIP}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ResultSinkConfig(BaseModel):
|
|
41
|
+
"""Where the child submits results. ``test_execution_id`` is set for hosted
|
|
42
|
+
runs (the platform pre-creates the execution); leaving it unset preserves
|
|
43
|
+
the local create-then-submit behavior."""
|
|
44
|
+
|
|
45
|
+
api_url: str | None = None
|
|
46
|
+
run_test_id: str | None = None
|
|
47
|
+
test_execution_id: str | None = None
|
|
48
|
+
root_directory: str | None = None
|
|
49
|
+
secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class VoiceRunConfig(BaseModel):
|
|
53
|
+
"""Voice runs use ``run_voice_simulation`` (LiveKit), not the chat
|
|
54
|
+
``SimulationRunner``. This carries the typed inputs as JSON-round-trippable
|
|
55
|
+
dicts the child hydrates into ``AgentDefinition`` / ``LiveKitSimulatorRuntime``
|
|
56
|
+
/ ``Scenario`` / ``SimulatorAgentDefinition``. ``transport.kind`` on the
|
|
57
|
+
agent definition selects webrtc vs sip."""
|
|
58
|
+
|
|
59
|
+
agent_definition: dict[str, JsonValue]
|
|
60
|
+
scenario: dict[str, JsonValue]
|
|
61
|
+
livekit_runtime: dict[str, JsonValue] | None = None
|
|
62
|
+
simulator: dict[str, JsonValue] | None = None
|
|
63
|
+
params: dict[str, JsonValue] = Field(default_factory=dict)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class StartRunnerJob(BaseModel):
|
|
67
|
+
schema_version: str = RUNNER_JOB_SCHEMA_VERSION
|
|
68
|
+
job_id: str
|
|
69
|
+
mode: RunnerMode = RunnerMode.CHAT
|
|
70
|
+
spec: SimulationSpec | None = None
|
|
71
|
+
voice: VoiceRunConfig | None = None
|
|
72
|
+
sink: ResultSinkConfig = Field(default_factory=ResultSinkConfig)
|
|
73
|
+
job_token_env: str | None = None
|
|
74
|
+
metadata: dict[str, JsonValue] = Field(default_factory=dict)
|
|
75
|
+
|
|
76
|
+
@model_validator(mode="after")
|
|
77
|
+
def _validate(self) -> "StartRunnerJob":
|
|
78
|
+
if self.schema_version != RUNNER_JOB_SCHEMA_VERSION:
|
|
79
|
+
raise ValueError(f"runner_job_version_unsupported: {self.schema_version}")
|
|
80
|
+
if self.mode is RunnerMode.CHAT and self.spec is None:
|
|
81
|
+
raise ValueError("chat runner job requires a spec")
|
|
82
|
+
if self.mode.is_voice and self.voice is None:
|
|
83
|
+
raise ValueError(f"{self.mode.value} runner job requires a voice config")
|
|
84
|
+
return self
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class RunnerJobPhase(str, Enum):
|
|
88
|
+
PENDING = "pending"
|
|
89
|
+
PREPARING = "preparing"
|
|
90
|
+
RUNNING = "running"
|
|
91
|
+
FINALIZING = "finalizing"
|
|
92
|
+
COMPLETED = "completed"
|
|
93
|
+
FAILED = "failed"
|
|
94
|
+
CANCELED = "canceled"
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def terminal(self) -> bool:
|
|
98
|
+
return self in {
|
|
99
|
+
RunnerJobPhase.COMPLETED,
|
|
100
|
+
RunnerJobPhase.FAILED,
|
|
101
|
+
RunnerJobPhase.CANCELED,
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class RunnerJobHandle(BaseModel):
|
|
106
|
+
job_id: str
|
|
107
|
+
run_id: str
|
|
108
|
+
pid: int | None = None
|
|
109
|
+
run_directory: str | None = None
|
|
110
|
+
metadata: dict[str, JsonValue] = Field(default_factory=dict)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class RunnerJobStatus(BaseModel):
|
|
114
|
+
job_id: str
|
|
115
|
+
phase: RunnerJobPhase
|
|
116
|
+
detail: str | None = None
|
|
117
|
+
report_hash: str | None = None
|
|
118
|
+
submission_status: str | None = None
|
|
119
|
+
updated_at: datetime
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class RunnerReconcileResult(BaseModel):
|
|
123
|
+
reconciled: bool
|
|
124
|
+
orphan_ids: list[str] = Field(default_factory=list)
|
|
125
|
+
metadata: dict[str, JsonValue] = Field(default_factory=dict)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class HostedRunnerPort(Protocol):
|
|
129
|
+
"""Scheduler-neutral port Temporal invokes (plan §9.1)."""
|
|
130
|
+
|
|
131
|
+
async def start(self, request: StartRunnerJob) -> RunnerJobHandle: ...
|
|
132
|
+
|
|
133
|
+
async def status(self, handle: RunnerJobHandle) -> RunnerJobStatus: ...
|
|
134
|
+
|
|
135
|
+
async def cancel(self, handle: RunnerJobHandle) -> None: ...
|
|
136
|
+
|
|
137
|
+
async def reconcile(self, handle: RunnerJobHandle) -> RunnerReconcileResult: ...
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
__all__ = [
|
|
141
|
+
"RUNNER_JOB_SCHEMA_VERSION",
|
|
142
|
+
"HostedRunnerPort",
|
|
143
|
+
"ResultSinkConfig",
|
|
144
|
+
"RunnerJobHandle",
|
|
145
|
+
"RunnerJobPhase",
|
|
146
|
+
"RunnerJobStatus",
|
|
147
|
+
"RunnerMode",
|
|
148
|
+
"RunnerReconcileResult",
|
|
149
|
+
"StartRunnerJob",
|
|
150
|
+
]
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Resolve the agent-under-test target from a job's ``SimulationSpec.target``.
|
|
2
|
+
|
|
3
|
+
The runner never runs the target agent itself — the target is the customer's
|
|
4
|
+
deployed (or supplied) agent. For chat runs the target is a turn-based surface,
|
|
5
|
+
resolved through the one endpoint registry: ``spec.target.adapter`` names the
|
|
6
|
+
actor-source kind (``callable`` / ``python_callable`` / ``import_object`` /
|
|
7
|
+
``factory`` / ``framework`` / ``system_prompt`` / ``http`` / …) and each
|
|
8
|
+
registered ``EndpointProfile`` carries the resolver. Adding a target kind is one
|
|
9
|
+
profile entry, no edits here (plan §4.1).
|
|
10
|
+
|
|
11
|
+
This is a HOSTED execution path (the runner runs it on our infra), so target
|
|
12
|
+
kinds that execute caller-supplied Python in-process (``callable`` /
|
|
13
|
+
``python_callable`` / ``import_object`` / ``factory`` / ``framework``) are
|
|
14
|
+
**rejected here** — deny-by-default via ``EndpointProfile.runs_caller_code``.
|
|
15
|
+
Hosted runs must reach the agent as a deployed endpoint (``http`` / ``websocket``)
|
|
16
|
+
or through the sandboxed runtime. The only in-process escape is a trusted
|
|
17
|
+
operator-configured default target, opted in explicitly with
|
|
18
|
+
``ALK_UNSAFE_INPROCESS_CODE_ACTORS`` — never set in prod for untrusted jobs.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from collections.abc import Callable
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
from fi.simulate.agent.wrapper import AgentWrapper
|
|
27
|
+
from fi.simulate.runtime.spec import SimulationSpec
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def resolve_chat_target(spec: SimulationSpec) -> Callable[..., Any] | AgentWrapper:
|
|
31
|
+
from fi.simulate.endpoints.actor_sources import (
|
|
32
|
+
ActorSourceError,
|
|
33
|
+
inprocess_code_allowed,
|
|
34
|
+
)
|
|
35
|
+
from fi.simulate.endpoints.profiles import get_profile
|
|
36
|
+
|
|
37
|
+
adapter = (spec.target.adapter or "").lower()
|
|
38
|
+
profile = get_profile(adapter)
|
|
39
|
+
if profile is None or not profile.is_turn_based_target:
|
|
40
|
+
raise ValueError(f"unsupported_chat_target_adapter: {spec.target.adapter}")
|
|
41
|
+
if profile.runs_caller_code and not inprocess_code_allowed():
|
|
42
|
+
raise ActorSourceError(
|
|
43
|
+
f"code_actor_denied_in_hosted: target {adapter!r} would run "
|
|
44
|
+
f"caller-supplied code in the runner process. Hosted runs must use a "
|
|
45
|
+
f"deployed endpoint (http/websocket) or the sandboxed runtime; "
|
|
46
|
+
f"in-process code is developer/local only."
|
|
47
|
+
)
|
|
48
|
+
return profile.resolve_target(
|
|
49
|
+
dict(spec.target.config or {}), spec.target.secret_refs, hosted=True
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
__all__ = ["resolve_chat_target"]
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""FutureAGIObserver — LiveKit AgentSession event tap (plan §6.6 skeleton).
|
|
2
|
+
|
|
3
|
+
Attach an observer to a ``livekit.agents.AgentSession`` and it will
|
|
4
|
+
subscribe to a documented list of session events, translate each into
|
|
5
|
+
a ``CanonicalEvent`` from ``fi.simulate.runtime``, and hand it to a
|
|
6
|
+
pluggable sink (defaulting to an in-memory list so tests can assert
|
|
7
|
+
against emitted events). The real OTLP wiring lands with
|
|
8
|
+
``OpenTelemetryEvidenceSource``; this observer only owns the SDK-side
|
|
9
|
+
event capture.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
from collections.abc import Callable
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from fi.simulate.runtime import CanonicalEvent, EventReliability
|
|
19
|
+
|
|
20
|
+
_SESSION_EVENT_MAP: dict[str, str] = {
|
|
21
|
+
"conversation_item_added": "transcript.final",
|
|
22
|
+
"user_state_changed": "speech.started",
|
|
23
|
+
"agent_state_changed": "session.ready",
|
|
24
|
+
"function_tool_execution_started": "tool.started",
|
|
25
|
+
"function_tool_execution_completed": "tool.completed",
|
|
26
|
+
"function_tool_execution_failed": "tool.failed",
|
|
27
|
+
"session_usage_updated": "usage.updated",
|
|
28
|
+
"close": "session.ended",
|
|
29
|
+
"error": "session.error",
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
logger = logging.getLogger(__name__)
|
|
33
|
+
|
|
34
|
+
EventSink = Callable[[CanonicalEvent], None]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class FutureAGIObserver:
|
|
38
|
+
def __init__(
|
|
39
|
+
self,
|
|
40
|
+
*,
|
|
41
|
+
run_id: str,
|
|
42
|
+
test_case_id: str,
|
|
43
|
+
sink: EventSink,
|
|
44
|
+
source: str = "livekit-observer",
|
|
45
|
+
) -> None:
|
|
46
|
+
self._run_id = run_id
|
|
47
|
+
self._test_case_id = test_case_id
|
|
48
|
+
self._sink = sink
|
|
49
|
+
self._source = source
|
|
50
|
+
self._sequence = 0
|
|
51
|
+
self._attached = False
|
|
52
|
+
|
|
53
|
+
def attach(self, session: Any) -> "FutureAGIObserver":
|
|
54
|
+
if self._attached:
|
|
55
|
+
raise RuntimeError("observer_already_attached")
|
|
56
|
+
if not hasattr(session, "on"):
|
|
57
|
+
raise TypeError("session_incompatible: object has no on(event, callback)")
|
|
58
|
+
for session_event, canonical_type in _SESSION_EVENT_MAP.items():
|
|
59
|
+
handler = self._handler_for(session_event, canonical_type)
|
|
60
|
+
try:
|
|
61
|
+
session.on(session_event, handler)
|
|
62
|
+
except (AttributeError, ValueError):
|
|
63
|
+
logger.debug(
|
|
64
|
+
"livekit observer: session does not expose event",
|
|
65
|
+
extra={"event": session_event},
|
|
66
|
+
)
|
|
67
|
+
self._attached = True
|
|
68
|
+
return self
|
|
69
|
+
|
|
70
|
+
def emit(
|
|
71
|
+
self,
|
|
72
|
+
event_type: str,
|
|
73
|
+
payload: dict[str, Any] | None = None,
|
|
74
|
+
*,
|
|
75
|
+
reliability: EventReliability = EventReliability.RELIABLE,
|
|
76
|
+
) -> CanonicalEvent:
|
|
77
|
+
self._sequence += 1
|
|
78
|
+
event = CanonicalEvent.create(
|
|
79
|
+
run_id=self._run_id,
|
|
80
|
+
test_case_id=self._test_case_id,
|
|
81
|
+
event_type=event_type,
|
|
82
|
+
source=self._source,
|
|
83
|
+
sequence=self._sequence,
|
|
84
|
+
reliability=reliability,
|
|
85
|
+
payload=payload or {},
|
|
86
|
+
)
|
|
87
|
+
self._sink(event)
|
|
88
|
+
return event
|
|
89
|
+
|
|
90
|
+
def _handler_for(self, session_event: str, canonical_type: str) -> Callable[..., None]:
|
|
91
|
+
def handler(*args: Any, **kwargs: Any) -> None:
|
|
92
|
+
payload = _summarize_payload(session_event, args, kwargs)
|
|
93
|
+
self.emit(canonical_type, payload)
|
|
94
|
+
|
|
95
|
+
return handler
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _summarize_payload(
|
|
99
|
+
session_event: str,
|
|
100
|
+
args: tuple[Any, ...],
|
|
101
|
+
kwargs: dict[str, Any],
|
|
102
|
+
) -> dict[str, Any]:
|
|
103
|
+
def _describe(value: Any) -> Any:
|
|
104
|
+
if hasattr(value, "model_dump"):
|
|
105
|
+
try:
|
|
106
|
+
return value.model_dump(mode="json", exclude_none=True)
|
|
107
|
+
except Exception: # noqa: BLE001
|
|
108
|
+
pass
|
|
109
|
+
if hasattr(value, "__dict__"):
|
|
110
|
+
return {"repr": type(value).__name__}
|
|
111
|
+
if isinstance(value, (str, int, float, bool)) or value is None:
|
|
112
|
+
return value
|
|
113
|
+
return type(value).__name__
|
|
114
|
+
|
|
115
|
+
return {
|
|
116
|
+
"session_event": session_event,
|
|
117
|
+
"args": [_describe(item) for item in args],
|
|
118
|
+
"kwargs": {key: _describe(value) for key, value in kwargs.items()},
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
__all__ = ["FutureAGIObserver"]
|