agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,1440 @@
|
|
|
1
|
+
"""The hosted lane's `CallRunner` — places one simulated LiveKit voice call and reports what
|
|
2
|
+
happened, satisfying `hosted_scheduler.CallRunner` exactly.
|
|
3
|
+
|
|
4
|
+
Three sub-systems (world-handle-interface.md, hosted-execution-seams.md v1.15 §2a):
|
|
5
|
+
|
|
6
|
+
1. **Placing the call.** The customer agent is already running INSIDE the Daytona sandbox, as a
|
|
7
|
+
world process the bundle's provisioner spawned (`process_runtime.py`) and registered with
|
|
8
|
+
LiveKit cloud under `LIVEKIT_AGENT_NAME=agent-w{WORLD_INDEX}`-style identity. This runner never
|
|
9
|
+
starts or manages that process. It drives `SimulationRunner` IN-PROCESS with a
|
|
10
|
+
`SimulationSpec` built by `simulator_voice.simulation_spec`, the same builder the local lane
|
|
11
|
+
uses; only the value lookup differs, resolving from job config and the bundle's scenario
|
|
12
|
+
document rather than `HARNESS_*` env vars. Do not rebuild the spec here: the two lanes drifted
|
|
13
|
+
for exactly that reason. The local-only webhook/subprocess plumbing `run/call.py` and
|
|
14
|
+
`run/live.py` use is neither available nor appropriate in the guest.
|
|
15
|
+
2. **Collecting evidence.** The bundle declares exactly one `runtime.evidence_seam`:
|
|
16
|
+
`http_tool` or `tool_trace`. `http_tool` has NO guest-side capture surface anywhere in this
|
|
17
|
+
repo today (see `_collect_http_tool_calls`'s docstring — a verified finding, not an assumption)
|
|
18
|
+
and is intentionally left returning zero calls rather than inventing a capture proxy.
|
|
19
|
+
`tool_trace` is read from the world's own postgres database against an unpinned, isolated
|
|
20
|
+
convention (see `_collect_tool_trace_calls`'s docstring). Either way, zero calls captured is
|
|
21
|
+
never fabricated into something else — the scheduler's own `evidence_missing` retry-once policy
|
|
22
|
+
is the contract-correct handling for "no evidence."
|
|
23
|
+
3. **Uploading artifacts.** The transcript and any produced recordings are uploaded through the
|
|
24
|
+
adapter's `upload_artifact` (content-addressed, budget/level-gated, returns `None` on refusal —
|
|
25
|
+
never an exception) BEFORE this runner returns, so `CallOutcome.transcript_artifact`/
|
|
26
|
+
`recording_artifacts` only ever carry ids the platform has already acked.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import atexit
|
|
32
|
+
import asyncio
|
|
33
|
+
import gc
|
|
34
|
+
import json
|
|
35
|
+
import logging
|
|
36
|
+
import os
|
|
37
|
+
import stat
|
|
38
|
+
import tempfile
|
|
39
|
+
from dataclasses import dataclass, field, replace
|
|
40
|
+
from datetime import datetime, timezone
|
|
41
|
+
from pathlib import Path
|
|
42
|
+
from typing import Any, Awaitable, Callable, Mapping, Protocol
|
|
43
|
+
|
|
44
|
+
from fi import simulate
|
|
45
|
+
from fi.simulate.runtime import (
|
|
46
|
+
SimulationSpec,
|
|
47
|
+
new_run_id,
|
|
48
|
+
)
|
|
49
|
+
from fi.simulate.runtime.report import SimulationReport
|
|
50
|
+
from fi.simulate.runtime.run import TestCaseStatus
|
|
51
|
+
from fi.simulate.runtime.runner import SimulationRunner
|
|
52
|
+
|
|
53
|
+
from .background_noise import scenario_source
|
|
54
|
+
from .bundle_v2 import EvidenceSeam
|
|
55
|
+
from .hosted_scheduler import CallAborted, CallOutcome
|
|
56
|
+
from .hosted_scheduler import Scenario as HostedScenario
|
|
57
|
+
from .job import ExecutionMode, HarnessJob, ProviderExecutionMode
|
|
58
|
+
from .outbound import ArtifactKind, format_rfc3339_millis
|
|
59
|
+
from .process_runtime import EnvironmentRuntime
|
|
60
|
+
from .scenario import DEFAULT_VOICEMAIL_STYLE, voicemail_enabled
|
|
61
|
+
from .voicemail_audio import clip_for
|
|
62
|
+
from .simulator_voice import (
|
|
63
|
+
CLEANUP_TIMEOUT_SECONDS,
|
|
64
|
+
CONNECT_TIMEOUT_SECONDS,
|
|
65
|
+
READINESS_TIMEOUT_SECONDS,
|
|
66
|
+
caller_scenario,
|
|
67
|
+
simulation_spec,
|
|
68
|
+
simulator_definition,
|
|
69
|
+
)
|
|
70
|
+
from .world.errors import WorldUnavailable
|
|
71
|
+
from .world.runtime import Call
|
|
72
|
+
|
|
73
|
+
logger = logging.getLogger(__name__)
|
|
74
|
+
|
|
75
|
+
# --- credential aliases / config keys a voice job must carry ---
|
|
76
|
+
|
|
77
|
+
LIVEKIT_API_KEY_ALIAS = "LIVEKIT_API_KEY"
|
|
78
|
+
LIVEKIT_API_SECRET_ALIAS = "LIVEKIT_API_SECRET"
|
|
79
|
+
LIVEKIT_URL_ALIAS = "LIVEKIT_URL"
|
|
80
|
+
DEEPGRAM_API_KEY_ALIAS = "DEEPGRAM_API_KEY"
|
|
81
|
+
CARTESIA_API_KEY_ALIAS = "CARTESIA_API_KEY"
|
|
82
|
+
GEMINI_API_KEY_ALIAS = "GEMINI_API_KEY"
|
|
83
|
+
GOOGLE_API_KEY_ALIAS = "GOOGLE_API_KEY"
|
|
84
|
+
GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS = "GOOGLE_APPLICATION_CREDENTIALS_JSON"
|
|
85
|
+
GOOGLE_APPLICATION_CREDENTIALS_ALIAS = "GOOGLE_APPLICATION_CREDENTIALS"
|
|
86
|
+
GOOGLE_CLOUD_PROJECT_ALIAS = "GOOGLE_CLOUD_PROJECT"
|
|
87
|
+
GOOGLE_CLOUD_LOCATION_ALIAS = "GOOGLE_CLOUD_LOCATION"
|
|
88
|
+
GOOGLE_GENAI_USE_VERTEXAI_ALIAS = "GOOGLE_GENAI_USE_VERTEXAI"
|
|
89
|
+
OPENAI_API_KEY_ALIAS = "OPENAI_API_KEY"
|
|
90
|
+
VAPI_API_KEY_ALIAS = "VAPI_API_KEY"
|
|
91
|
+
RETELL_API_KEY_ALIAS = "RETELL_API_KEY"
|
|
92
|
+
SIMULATOR_LLM_PROVIDER_ALIAS = "SIMULATOR_LLM_PROVIDER"
|
|
93
|
+
SIMULATOR_LLM_MODEL_ALIAS = "SIMULATOR_LLM_MODEL"
|
|
94
|
+
SIMULATOR_STT_PROVIDER_ALIAS = "SIMULATOR_STT_PROVIDER"
|
|
95
|
+
SIMULATOR_STT_MODEL_ALIAS = "SIMULATOR_STT_MODEL"
|
|
96
|
+
SIMULATOR_TTS_PROVIDER_ALIAS = "SIMULATOR_TTS_PROVIDER"
|
|
97
|
+
SIMULATOR_TTS_MODEL_ALIAS = "SIMULATOR_TTS_MODEL"
|
|
98
|
+
BACKGROUND_NOISE_ALIAS = "ALK_BACKGROUND_NOISE"
|
|
99
|
+
BACKGROUND_NOISE_CATALOG_ALIAS = "ALK_BACKGROUND_NOISE_CATALOG"
|
|
100
|
+
BACKGROUND_NOISE_VOLUME_ALIAS = "HARNESS_BACKGROUND_NOISE_VOLUME"
|
|
101
|
+
CALL_DIRECTION_ALIAS = "ALK_CALL_DIRECTION"
|
|
102
|
+
VOICEMAIL_CLIP_ALIAS = "HARNESS_VOICEMAIL_CLIP"
|
|
103
|
+
VOICEMAIL_CLIP_TONE_ALIAS = "HARNESS_VOICEMAIL_CLIP_HAS_TONE"
|
|
104
|
+
VOICEMAIL_CLIP_TEXT_ALIAS = "HARNESS_VOICEMAIL_CLIP_TRANSCRIPT"
|
|
105
|
+
LIVEKIT_URL_CONFIG_KEY = "livekit_url"
|
|
106
|
+
CALL_TIMEOUT_CONFIG_KEY = "voice_call_timeout_seconds"
|
|
107
|
+
|
|
108
|
+
_SIMULATOR_PLATFORM_ALIAS_MAP = {
|
|
109
|
+
"SIMULATOR_LIVEKIT_URL": LIVEKIT_URL_ALIAS,
|
|
110
|
+
"SIMULATOR_LIVEKIT_API_KEY": LIVEKIT_API_KEY_ALIAS,
|
|
111
|
+
"SIMULATOR_LIVEKIT_API_SECRET": LIVEKIT_API_SECRET_ALIAS,
|
|
112
|
+
"SIMULATOR_DEEPGRAM_API_KEY": DEEPGRAM_API_KEY_ALIAS,
|
|
113
|
+
"SIMULATOR_CARTESIA_API_KEY": CARTESIA_API_KEY_ALIAS,
|
|
114
|
+
"SIMULATOR_GEMINI_API_KEY": GEMINI_API_KEY_ALIAS,
|
|
115
|
+
"SIMULATOR_GOOGLE_API_KEY": GOOGLE_API_KEY_ALIAS,
|
|
116
|
+
"SIMULATOR_GOOGLE_APPLICATION_CREDENTIALS_JSON": (
|
|
117
|
+
GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS
|
|
118
|
+
),
|
|
119
|
+
"SIMULATOR_GOOGLE_CLOUD_PROJECT": GOOGLE_CLOUD_PROJECT_ALIAS,
|
|
120
|
+
"SIMULATOR_GOOGLE_CLOUD_LOCATION": GOOGLE_CLOUD_LOCATION_ALIAS,
|
|
121
|
+
"SIMULATOR_GOOGLE_GENAI_USE_VERTEXAI": GOOGLE_GENAI_USE_VERTEXAI_ALIAS,
|
|
122
|
+
"SIMULATOR_OPENAI_API_KEY": OPENAI_API_KEY_ALIAS,
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
_DEFAULT_CALL_TIMEOUT_SECONDS = 300.0
|
|
126
|
+
|
|
127
|
+
# sdk_voice.py::build_spec's own phase-overhead constants, reused verbatim so this runner's
|
|
128
|
+
# outer budget composes with the SDK's internal one the same way the local template does.
|
|
129
|
+
_RUN_SECONDS_PAD_SECONDS = 60.0
|
|
130
|
+
# Headroom beyond `spec.execution.timeout.run_seconds` -- SimulationRunner.run() already wraps
|
|
131
|
+
# `plugin.run(...)` in its OWN `asyncio.wait_for(..., timeout=spec.execution.timeout.run_seconds)`
|
|
132
|
+
# (runner.py) and catches that TimeoutError into a graceful `SimulationReport(status=TIMED_OUT)`.
|
|
133
|
+
# This runner's own outer wait_for must stay LARGER than that so the SDK's internal timeout fires
|
|
134
|
+
# first in the ordinary case; it only ever fires itself for a genuinely hung SDK (a real post-dial
|
|
135
|
+
# machinery failure) -- a runner-owned asyncio.wait_for as the last-resort bound.
|
|
136
|
+
_OUTER_WAIT_FOR_PAD_SECONDS = 60.0
|
|
137
|
+
|
|
138
|
+
# Unpinned by any contract and no producer exists yet. Isolated as one
|
|
139
|
+
# constant + two functions (`_clear_tool_trace_calls`, `_collect_tool_trace_calls`) so a real
|
|
140
|
+
# producer's disagreement on the name/shape is a one-line change.
|
|
141
|
+
_TOOL_TRACE_TABLE = "_alk_tool_trace"
|
|
142
|
+
|
|
143
|
+
_RESULT_TRUNCATE_CHARS = 2000
|
|
144
|
+
|
|
145
|
+
# Turns a timed-out call needs before it is worth grading rather than aborting. Low on purpose: the
|
|
146
|
+
# question is only whether a conversation happened at all.
|
|
147
|
+
_GRADEABLE_AFTER_TIMEOUT_TURNS = 4
|
|
148
|
+
|
|
149
|
+
# The real engine's zero-turn "agent joined but never spoke" failure codes (engines/livekit.py::
|
|
150
|
+
# _conversation_outcome) -- see `_translate_report`'s `is_silent_agent` gate for why these two, and
|
|
151
|
+
# only at zero turns, get mapped to a normal CallOutcome instead of a CallAborted.
|
|
152
|
+
_SILENT_AGENT_FAILURE_CODES = frozenset(
|
|
153
|
+
{"no_conversation", "conversation_silence_timeout"}
|
|
154
|
+
)
|
|
155
|
+
_CONVERSATION_STALL_FAILURE_CODES = frozenset(
|
|
156
|
+
{"conversation_silence_timeout", "conversation_stalled"}
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _attributed_stall(case: Any) -> tuple[str, str] | None:
|
|
161
|
+
"""Attribute a speech stall from committed transcript turns, without guessing."""
|
|
162
|
+
if (
|
|
163
|
+
case.failure is None
|
|
164
|
+
or case.failure.code not in _CONVERSATION_STALL_FAILURE_CODES
|
|
165
|
+
):
|
|
166
|
+
return None
|
|
167
|
+
if case.result is None or not case.result.messages:
|
|
168
|
+
return None
|
|
169
|
+
last = case.result.messages[-1]
|
|
170
|
+
if not isinstance(last, dict):
|
|
171
|
+
return None
|
|
172
|
+
role = str(last.get("role") or "").strip().lower()
|
|
173
|
+
content = str(last.get("content") or "").strip()
|
|
174
|
+
if role in {"user", "caller", "customer"}:
|
|
175
|
+
return (
|
|
176
|
+
"target_agent_stalled",
|
|
177
|
+
"Target agent produced no response after the caller's final transcribed turn",
|
|
178
|
+
)
|
|
179
|
+
if role in {"assistant", "agent"}:
|
|
180
|
+
if content and content[-1] not in ".?!":
|
|
181
|
+
return (
|
|
182
|
+
"target_agent_stalled",
|
|
183
|
+
"Target agent stopped mid-utterance and produced no further speech",
|
|
184
|
+
)
|
|
185
|
+
return (
|
|
186
|
+
"simulator_stalled",
|
|
187
|
+
"Simulated caller produced no response after the target agent's final turn",
|
|
188
|
+
)
|
|
189
|
+
return None
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# --- collaborator seams (named, injectable test boundaries) -----------------------------------
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
class ArtifactUploader(Protocol):
|
|
196
|
+
"""Narrow slice of `hosted_entrypoint.OutboundAdapter` -- avoids importing that module here
|
|
197
|
+
(it imports THIS module's factory to wire the real CallRunner; importing it back would be
|
|
198
|
+
circular)."""
|
|
199
|
+
|
|
200
|
+
async def upload_artifact(
|
|
201
|
+
self,
|
|
202
|
+
data: bytes,
|
|
203
|
+
*,
|
|
204
|
+
kind: ArtifactKind,
|
|
205
|
+
scenario_key: str | None = None,
|
|
206
|
+
deadline: float | None = None,
|
|
207
|
+
) -> str | None: ...
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
PlaceCall = Callable[[SimulationSpec], Awaitable[SimulationReport]]
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
async def _default_place_call(spec: SimulationSpec) -> SimulationReport:
|
|
214
|
+
return await SimulationRunner().run(spec)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
@dataclass(frozen=True)
|
|
218
|
+
class CallRunnerContext:
|
|
219
|
+
"""Everything `hosted_entrypoint.py`'s `run_job` already has in scope by the wiring point
|
|
220
|
+
(~1662) that the real `CallRunnerImpl` needs but the bare `CallRunner` protocol signature
|
|
221
|
+
(`run(scenario, runtime)`) has no room to carry. Threaded through the EXTENDED
|
|
222
|
+
`build_call_runner(adapter, context)` seam."""
|
|
223
|
+
|
|
224
|
+
job: HarnessJob
|
|
225
|
+
bundle_dir: Path
|
|
226
|
+
work_directory: Path
|
|
227
|
+
evidence_seam: EvidenceSeam | None
|
|
228
|
+
target_provider_secret_values: Mapping[str, str]
|
|
229
|
+
attempt_number: int
|
|
230
|
+
source_directory: Path | None = None
|
|
231
|
+
simulator_provider_secret_values: Mapping[str, str] = field(default_factory=dict)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# --- pre-dial validation -----------------------------------------------------------------------
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
@dataclass(frozen=True)
|
|
238
|
+
class _MissingVoiceConfig:
|
|
239
|
+
aliases: tuple[str, ...]
|
|
240
|
+
config_keys: tuple[str, ...]
|
|
241
|
+
|
|
242
|
+
def message(self) -> str:
|
|
243
|
+
parts = []
|
|
244
|
+
if self.aliases:
|
|
245
|
+
parts.append("secrets=" + ",".join(self.aliases))
|
|
246
|
+
if self.config_keys:
|
|
247
|
+
parts.append("config=" + ",".join(self.config_keys))
|
|
248
|
+
return "voice_capability_unavailable: missing " + "; ".join(parts)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _resolve_connector(
|
|
252
|
+
job: HarnessJob, target_provider_secret_values: Mapping[str, str]
|
|
253
|
+
) -> str:
|
|
254
|
+
"""Pin the job's transport connector from the credentials actually present.
|
|
255
|
+
|
|
256
|
+
A fresh one-shot ships ``job.json`` with ``connector="auto"``: the platform only writes the
|
|
257
|
+
authored connector back onto the job *after* authoring, by which time this guest has already
|
|
258
|
+
booted from the un-resolved payload. Mirror the platform rule so a LiveKit-credentialed
|
|
259
|
+
``auto`` job dispatches to the target agent instead of the simulator lane with no identity.
|
|
260
|
+
"""
|
|
261
|
+
connector = job.agent.connector.strip().lower()
|
|
262
|
+
if connector != "auto":
|
|
263
|
+
return connector
|
|
264
|
+
if job.agent.config.get(
|
|
265
|
+
LIVEKIT_URL_CONFIG_KEY
|
|
266
|
+
) or target_provider_secret_values.get(LIVEKIT_URL_ALIAS):
|
|
267
|
+
return "livekit"
|
|
268
|
+
if target_provider_secret_values.get(VAPI_API_KEY_ALIAS):
|
|
269
|
+
return "vapi"
|
|
270
|
+
if target_provider_secret_values.get(RETELL_API_KEY_ALIAS):
|
|
271
|
+
return "retell"
|
|
272
|
+
return connector
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _check_config(
|
|
276
|
+
job: HarnessJob,
|
|
277
|
+
target_provider_secret_values: Mapping[str, str],
|
|
278
|
+
simulator_values: Mapping[str, str] | None = None,
|
|
279
|
+
) -> _MissingVoiceConfig | None:
|
|
280
|
+
simulator_values = simulator_values or {}
|
|
281
|
+
|
|
282
|
+
def simulator_value(alias: str) -> str | None:
|
|
283
|
+
# Hosted runs supply platform-owned simulator credentials in the control process. The
|
|
284
|
+
# target-provider value remains a backwards-compatible fallback for local SDK callers.
|
|
285
|
+
return simulator_values.get(alias) or (
|
|
286
|
+
target_provider_secret_values.get(alias)
|
|
287
|
+
if job.execution is ExecutionMode.LOCAL
|
|
288
|
+
else None
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
config = job.agent.config
|
|
292
|
+
llm_provider = str(
|
|
293
|
+
config.get("simulator_llm_provider")
|
|
294
|
+
or simulator_value(SIMULATOR_LLM_PROVIDER_ALIAS)
|
|
295
|
+
or "google"
|
|
296
|
+
).lower()
|
|
297
|
+
stt_provider = str(
|
|
298
|
+
config.get("simulator_stt_provider")
|
|
299
|
+
or simulator_value(SIMULATOR_STT_PROVIDER_ALIAS)
|
|
300
|
+
or "deepgram"
|
|
301
|
+
).lower()
|
|
302
|
+
tts_provider = str(
|
|
303
|
+
config.get("simulator_tts_provider")
|
|
304
|
+
or simulator_value(SIMULATOR_TTS_PROVIDER_ALIAS)
|
|
305
|
+
or "deepgram"
|
|
306
|
+
).lower()
|
|
307
|
+
|
|
308
|
+
connector = _resolve_connector(job, target_provider_secret_values)
|
|
309
|
+
livekit_values = (
|
|
310
|
+
target_provider_secret_values if connector == "livekit" else simulator_values
|
|
311
|
+
)
|
|
312
|
+
required = [LIVEKIT_API_KEY_ALIAS, LIVEKIT_API_SECRET_ALIAS]
|
|
313
|
+
if connector == "vapi":
|
|
314
|
+
required.append(VAPI_API_KEY_ALIAS)
|
|
315
|
+
elif connector == "retell":
|
|
316
|
+
required.append(RETELL_API_KEY_ALIAS)
|
|
317
|
+
if "deepgram" in {stt_provider, tts_provider}:
|
|
318
|
+
if not simulator_value(DEEPGRAM_API_KEY_ALIAS):
|
|
319
|
+
required.append(DEEPGRAM_API_KEY_ALIAS)
|
|
320
|
+
|
|
321
|
+
def credential(alias: str) -> str | None:
|
|
322
|
+
if alias in {LIVEKIT_API_KEY_ALIAS, LIVEKIT_API_SECRET_ALIAS}:
|
|
323
|
+
return livekit_values.get(alias)
|
|
324
|
+
if alias in {VAPI_API_KEY_ALIAS, RETELL_API_KEY_ALIAS}:
|
|
325
|
+
return target_provider_secret_values.get(alias)
|
|
326
|
+
return simulator_value(alias)
|
|
327
|
+
|
|
328
|
+
missing_aliases = [alias for alias in required if not credential(alias)]
|
|
329
|
+
|
|
330
|
+
if llm_provider == "google":
|
|
331
|
+
has_api_key = bool(
|
|
332
|
+
simulator_value(GEMINI_API_KEY_ALIAS)
|
|
333
|
+
or simulator_value(GOOGLE_API_KEY_ALIAS)
|
|
334
|
+
)
|
|
335
|
+
has_vertex_adc = bool(
|
|
336
|
+
(
|
|
337
|
+
simulator_value(GOOGLE_APPLICATION_CREDENTIALS_ALIAS)
|
|
338
|
+
or simulator_value(GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS)
|
|
339
|
+
)
|
|
340
|
+
and simulator_value(GOOGLE_CLOUD_PROJECT_ALIAS)
|
|
341
|
+
)
|
|
342
|
+
if not has_api_key and not has_vertex_adc:
|
|
343
|
+
missing_aliases.append(
|
|
344
|
+
f"{GEMINI_API_KEY_ALIAS}_or_{GOOGLE_API_KEY_ALIAS}_or_VERTEX_ADC"
|
|
345
|
+
)
|
|
346
|
+
elif llm_provider == "openai" and not simulator_value(OPENAI_API_KEY_ALIAS):
|
|
347
|
+
missing_aliases.append(OPENAI_API_KEY_ALIAS)
|
|
348
|
+
|
|
349
|
+
has_livekit_url = bool(
|
|
350
|
+
config.get(LIVEKIT_URL_CONFIG_KEY) or livekit_values.get(LIVEKIT_URL_ALIAS)
|
|
351
|
+
)
|
|
352
|
+
missing_config_keys = [] if has_livekit_url else [LIVEKIT_URL_CONFIG_KEY]
|
|
353
|
+
if not missing_aliases and not missing_config_keys:
|
|
354
|
+
return None
|
|
355
|
+
return _MissingVoiceConfig(tuple(missing_aliases), tuple(missing_config_keys))
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _canonical_simulator_secrets(values: Mapping[str, str]) -> dict[str, str]:
|
|
359
|
+
"""Translate platform-only aliases into the names expected by simulator plugins."""
|
|
360
|
+
return {
|
|
361
|
+
_SIMULATOR_PLATFORM_ALIAS_MAP.get(alias, alias): value
|
|
362
|
+
for alias, value in values.items()
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _dispatch_agent_name(runtime: EnvironmentRuntime) -> str | None:
|
|
367
|
+
"""The ONLY place this repo reads the dispatch-identity metadata key, so a
|
|
368
|
+
change to the key name/convention is a one-line adapt. The provisioner
|
|
369
|
+
mirrors the agent process's rendered LIVEKIT_AGENT_NAME here; a bundle
|
|
370
|
+
that declares none (or an ambiguous set) leaves the key absent and the
|
|
371
|
+
caller's typed `CallAborted` below fires."""
|
|
372
|
+
value = runtime.metadata.get("livekit_agent_name")
|
|
373
|
+
return value.strip() if isinstance(value, str) and value.strip() else None
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
# --- scenario document re-read (the _CompiledScenario the scheduler hands over carries no
|
|
377
|
+
# persona/instruction -- scenario_source.py:170-184's deliberately narrow Scenario-protocol
|
|
378
|
+
# shape) ------------------------------------------------------------------------------------
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
class _ScenarioDocumentUnavailable(RuntimeError):
|
|
382
|
+
pass
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _read_scenario_document(bundle_dir: Path, scenario_key: str) -> dict[str, Any]:
|
|
386
|
+
"""Re-reads `scenarios/<folder>/scenario.json` from the bundle, matched by the document's OWN
|
|
387
|
+
`scenario_key` field -- never the folder name (`scenario_source.py`'s own convention; the two
|
|
388
|
+
are not guaranteed to match)."""
|
|
389
|
+
root = bundle_dir / "scenarios"
|
|
390
|
+
if not root.is_dir():
|
|
391
|
+
raise _ScenarioDocumentUnavailable(f"no {root} directory in this bundle")
|
|
392
|
+
try:
|
|
393
|
+
children = sorted(root.iterdir())
|
|
394
|
+
except OSError as exc:
|
|
395
|
+
raise _ScenarioDocumentUnavailable(f"cannot list {root}: {exc}") from exc
|
|
396
|
+
for child in children:
|
|
397
|
+
if not child.is_dir():
|
|
398
|
+
continue
|
|
399
|
+
doc_path = child / "scenario.json"
|
|
400
|
+
if not doc_path.is_file():
|
|
401
|
+
continue
|
|
402
|
+
try:
|
|
403
|
+
body = json.loads(doc_path.read_text(encoding="utf-8"))
|
|
404
|
+
except (OSError, ValueError):
|
|
405
|
+
continue
|
|
406
|
+
if isinstance(body, dict) and body.get("scenario_key") == scenario_key:
|
|
407
|
+
instruction = body.get("instruction")
|
|
408
|
+
if not isinstance(instruction, str) or not instruction.strip():
|
|
409
|
+
raise _ScenarioDocumentUnavailable(
|
|
410
|
+
f"{child.name}/scenario.json has no non-empty instruction"
|
|
411
|
+
)
|
|
412
|
+
return body
|
|
413
|
+
raise _ScenarioDocumentUnavailable(
|
|
414
|
+
f"no scenario.json under {root} carries scenario_key={scenario_key!r}"
|
|
415
|
+
)
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
# --- deterministic room naming (asserted verbatim by tests/harness/test_call_runner.py). WHY this
|
|
419
|
+
# is a PREFIX guarantee, not a full-match one: in managed room_mode, engines/livekit.py::
|
|
420
|
+
# _resolve_room_name appends its own `-{invocation_id}-{test_case_id[-12:]}` suffix unless
|
|
421
|
+
# `room_name_verbatim` is set (which this runner does not set) -- the scheme below still gives
|
|
422
|
+
# every call a unique, deterministic, greppable prefix; only the exact wire-level name is not this
|
|
423
|
+
# string verbatim. -------------------------------------------------------------------------------
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _room_name(
|
|
427
|
+
*, job_id: str, attempt_number: int, scenario_key: str, scenario_attempt: int
|
|
428
|
+
) -> str:
|
|
429
|
+
return f"harness-{job_id[:8]}-a{attempt_number}-{scenario_key}-s{scenario_attempt}"
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _duration_ms(started_at: datetime, ended_at: datetime) -> int:
|
|
433
|
+
return max(0, int((ended_at - started_at).total_seconds() * 1000))
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
# --- SimulationSpec construction. Shared with the local lane through simulator_voice; only the
|
|
437
|
+
# value lookup is lane-specific. ---------------------------------------------------------------
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _dials_the_person(doc: dict[str, Any]) -> bool:
|
|
441
|
+
"""Whether the agent places the call: scenario first, then the environment, then inbound."""
|
|
442
|
+
direction = str(
|
|
443
|
+
doc.get("call_direction") or os.environ.get(CALL_DIRECTION_ALIAS) or "inbound"
|
|
444
|
+
)
|
|
445
|
+
return direction.strip().lower() == "outbound"
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def _build_spec(
|
|
449
|
+
*,
|
|
450
|
+
run_id: str,
|
|
451
|
+
room_name: str,
|
|
452
|
+
agent_name: str | None,
|
|
453
|
+
doc: Mapping[str, Any],
|
|
454
|
+
livekit_url: str,
|
|
455
|
+
call_timeout_seconds: float,
|
|
456
|
+
run_seconds: float,
|
|
457
|
+
recordings_root: Path,
|
|
458
|
+
simulator_config: Mapping[str, Any],
|
|
459
|
+
environ: Mapping[str, str],
|
|
460
|
+
connector: str = "livekit",
|
|
461
|
+
provider_target_id: str | None = None,
|
|
462
|
+
) -> SimulationSpec:
|
|
463
|
+
"""The hosted lane: values come from job config and the bundle's scenario document.
|
|
464
|
+
|
|
465
|
+
Lowercase job config wins; provider environment aliases are accepted for compatibility.
|
|
466
|
+
"""
|
|
467
|
+
|
|
468
|
+
def setting(name: str) -> str:
|
|
469
|
+
return str(simulator_config.get(name.lower()) or environ.get(name) or "")
|
|
470
|
+
|
|
471
|
+
simulator = simulator_definition(setting, doc.get("persona"))
|
|
472
|
+
connector = connector.strip().lower()
|
|
473
|
+
provider_agent: simulate.AgentDefinition | None = None
|
|
474
|
+
if connector == "vapi":
|
|
475
|
+
if not provider_target_id:
|
|
476
|
+
raise ValueError("vapi_target_id_unavailable")
|
|
477
|
+
provider_agent = simulate.AgentDefinition(
|
|
478
|
+
name="harness-vapi-target",
|
|
479
|
+
system_prompt=str(
|
|
480
|
+
simulator_config.get("target_system_prompt")
|
|
481
|
+
or "Provider-hosted Vapi target under test."
|
|
482
|
+
),
|
|
483
|
+
target={
|
|
484
|
+
"provider": "vapi",
|
|
485
|
+
"assistant_id": provider_target_id,
|
|
486
|
+
"api_base_url": str(
|
|
487
|
+
simulator_config.get("vapi_api_base_url") or "https://api.vapi.ai"
|
|
488
|
+
),
|
|
489
|
+
"api_key_env": VAPI_API_KEY_ALIAS,
|
|
490
|
+
},
|
|
491
|
+
transport={"kind": "vapi_websocket"},
|
|
492
|
+
provider_evidence={
|
|
493
|
+
"provider": "vapi",
|
|
494
|
+
"call_id_source": "originator_response",
|
|
495
|
+
},
|
|
496
|
+
)
|
|
497
|
+
elif connector == "retell":
|
|
498
|
+
if not provider_target_id:
|
|
499
|
+
raise ValueError("retell_target_id_unavailable")
|
|
500
|
+
provider_agent = simulate.AgentDefinition(
|
|
501
|
+
name="harness-retell-target",
|
|
502
|
+
system_prompt=str(
|
|
503
|
+
simulator_config.get("target_system_prompt")
|
|
504
|
+
or "Provider-hosted Retell target under test."
|
|
505
|
+
),
|
|
506
|
+
target={
|
|
507
|
+
"provider": "retell",
|
|
508
|
+
"agent_id": provider_target_id,
|
|
509
|
+
"api_url": str(
|
|
510
|
+
simulator_config.get("retell_api_url")
|
|
511
|
+
or "https://api.retellai.com/v2/create-web-call"
|
|
512
|
+
),
|
|
513
|
+
"livekit_url": str(
|
|
514
|
+
simulator_config.get("retell_livekit_url")
|
|
515
|
+
or "wss://retell-ai-4ihahnq7.livekit.cloud"
|
|
516
|
+
),
|
|
517
|
+
"api_key_env": RETELL_API_KEY_ALIAS,
|
|
518
|
+
},
|
|
519
|
+
transport={"kind": "retell_webcall"},
|
|
520
|
+
provider_evidence={
|
|
521
|
+
"provider": "retell",
|
|
522
|
+
"call_id_source": "originator_response",
|
|
523
|
+
},
|
|
524
|
+
)
|
|
525
|
+
# A provider-hosted agent owns termination and can legitimately finish an agent-first call
|
|
526
|
+
# after the fifth message: agent greeting, caller request, agent clarification, caller answer,
|
|
527
|
+
# agent confirmation followed by the provider's end-call tool. Requiring the simulator's
|
|
528
|
+
# sixth acknowledgement after Retell/Vapi has already disconnected misclassifies a complete
|
|
529
|
+
# call as infrastructure failure and prevents the tool trace from being graded. Native
|
|
530
|
+
# LiveKit keeps the stricter six-message floor because our simulator owns that hang-up path.
|
|
531
|
+
min_turn_messages = 5 if connector in {"vapi", "retell"} else 6
|
|
532
|
+
return simulation_spec(
|
|
533
|
+
run_id=run_id,
|
|
534
|
+
room_name=room_name,
|
|
535
|
+
agent_name=agent_name,
|
|
536
|
+
system_prompt=doc["instruction"],
|
|
537
|
+
livekit_url=livekit_url,
|
|
538
|
+
recording_dir=recordings_root / run_id / "recordings",
|
|
539
|
+
scenario=caller_scenario(
|
|
540
|
+
name=str(doc.get("scenario_key") or doc.get("name") or "harness-voice"),
|
|
541
|
+
persona=doc.get("persona"),
|
|
542
|
+
situation=doc["instruction"],
|
|
543
|
+
fixture=doc.get("fixture"),
|
|
544
|
+
tts_provider=simulator.tts.provider,
|
|
545
|
+
),
|
|
546
|
+
simulator=simulator,
|
|
547
|
+
# An outbound agent dials; the person answers, so the caller opens.
|
|
548
|
+
direction="simulator_first" if _dials_the_person(doc) else "agent_first",
|
|
549
|
+
max_seconds=call_timeout_seconds,
|
|
550
|
+
min_turn_messages=min_turn_messages,
|
|
551
|
+
# Hosted targets can legitimately spend tens of seconds in a provider call or a tool
|
|
552
|
+
# round-trip after the conversation has begun. The previous 45-second value terminated
|
|
553
|
+
# an otherwise healthy LiveKit call at exactly the watchdog boundary. Keep a finite
|
|
554
|
+
# liveness guard, but align it with the engine's 60-second conversation-silence backstop.
|
|
555
|
+
agent_first_silence_seconds=60.0,
|
|
556
|
+
run_seconds=run_seconds,
|
|
557
|
+
agent_definition=provider_agent,
|
|
558
|
+
)
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
# --- evidence collection -----------------------------------------------------------------------
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _find_postgres_endpoint(runtime: EnvironmentRuntime) -> Any | None:
|
|
565
|
+
"""Protocol-based lookup, matching `hosted_entrypoint.py::_find_postgres_endpoint`'s own
|
|
566
|
+
already-correct convention -- capability slugs are bundle-author-chosen (`build_endpoints`,
|
|
567
|
+
process_runtime.py:318-339), never a fixed key, so a hardcoded `endpoints["database"]` would
|
|
568
|
+
break for any bundle that names its capability slug differently. Re-implemented locally rather
|
|
569
|
+
than imported: importing from `hosted_entrypoint.py` here would be circular (it imports this
|
|
570
|
+
module's factory)."""
|
|
571
|
+
for endpoint in runtime.endpoints.values():
|
|
572
|
+
if endpoint.protocol == "postgres":
|
|
573
|
+
return endpoint
|
|
574
|
+
return None
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _collect_http_tool_calls(runtime: EnvironmentRuntime) -> tuple[Call, ...]:
|
|
578
|
+
"""No guest-side capture surface exists anywhere in this repo for the
|
|
579
|
+
`http_tool` evidence seam. Verified, not assumed: `world/handle.py::HostedWorld.call()`
|
|
580
|
+
raises `WorldUnavailable` unconditionally with a docstring stating the wire format "is not
|
|
581
|
+
pinned anywhere in the contracts yet"; `process_runtime.py`'s own `provision()` signature
|
|
582
|
+
comment says "evidence-seam wiring is out of this phase's scope"; no `TOOLS_API_URL` wiring
|
|
583
|
+
exists in the hosted lane at all (the local lane's `ProvisionedWorld`/`TOOLS_API_URL` mechanism
|
|
584
|
+
lives in `provision.py`/`world/provisioned.py`, out of scope here and inapplicable to the guest
|
|
585
|
+
regardless). Deliberately stopped rather than inventing a capture proxy: a job whose bundle
|
|
586
|
+
declares `evidence_seam: http_tool` reads zero calls every time, which the scheduler's own
|
|
587
|
+
`evidence_missing` retry-once policy turns into the correct, honest outcome -- never a crash,
|
|
588
|
+
never fabricated evidence."""
|
|
589
|
+
del runtime
|
|
590
|
+
return ()
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
def _clear_tool_trace_calls(dsn: str) -> None:
|
|
594
|
+
"""world-handle-interface.md: "setup's tool calls are NOT evidence (the runner clears them
|
|
595
|
+
before the call starts, as the local runner does)" -- the local runner's analog is
|
|
596
|
+
`world.calls = []` right before dialing (`run/simulation.py`). Best-effort: a missing table (no
|
|
597
|
+
producer yet) or any connection error is swallowed, never raised. Clearing
|
|
598
|
+
is housekeeping, not a correctness requirement, while nothing writes this table yet; once a
|
|
599
|
+
real producer lands this stops being a no-op automatically."""
|
|
600
|
+
try:
|
|
601
|
+
import psycopg
|
|
602
|
+
|
|
603
|
+
with psycopg.connect(dsn, autocommit=True, connect_timeout=5) as connection:
|
|
604
|
+
connection.execute(f'DELETE FROM "{_TOOL_TRACE_TABLE}"') # noqa: S608 - fixed identifier, no interpolated user input
|
|
605
|
+
except Exception as exc: # noqa: BLE001 - best-effort housekeeping only, never a call-blocking failure
|
|
606
|
+
# WHY: never log exc_info / str(exc) here -- a psycopg connection failure embeds the raw
|
|
607
|
+
# DSN (including the world DB password) in its own exception message; only the exception
|
|
608
|
+
# TYPE is safe for a local log line.
|
|
609
|
+
logger.debug(
|
|
610
|
+
"tool_trace clear skipped (table likely absent): %s", type(exc).__name__
|
|
611
|
+
)
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def _collect_tool_trace_calls(runtime: EnvironmentRuntime) -> tuple[Call, ...]:
|
|
615
|
+
"""`_alk_tool_trace`'s name and column shape are an isolated local
|
|
616
|
+
convention -- unpinned by any contract (the only harness-reserved table anywhere in this
|
|
617
|
+
repo is `_alk_conformance`, unrelated), no producer exists yet. Isolated in this one function
|
|
618
|
+
(+ `_clear_tool_trace_calls`) so a real producer's disagreement on the name/shape is a one-line
|
|
619
|
+
change. Any failure (missing table, connection refused, malformed row) degrades to `()` --
|
|
620
|
+
never a crash, never fabricated evidence, matching `_collect_http_tool_calls`'s stopped
|
|
621
|
+
behavior above."""
|
|
622
|
+
endpoint = _find_postgres_endpoint(runtime)
|
|
623
|
+
if endpoint is None:
|
|
624
|
+
return ()
|
|
625
|
+
try:
|
|
626
|
+
import psycopg
|
|
627
|
+
|
|
628
|
+
with psycopg.connect(
|
|
629
|
+
endpoint.address,
|
|
630
|
+
autocommit=True,
|
|
631
|
+
connect_timeout=5,
|
|
632
|
+
options="-c default_transaction_read_only=on",
|
|
633
|
+
) as connection:
|
|
634
|
+
cursor = connection.execute(
|
|
635
|
+
f'SELECT name, arguments, result, ok, error, at FROM "{_TOOL_TRACE_TABLE}" ' # noqa: S608
|
|
636
|
+
"ORDER BY at ASC"
|
|
637
|
+
)
|
|
638
|
+
rows = cursor.fetchall()
|
|
639
|
+
columns = [description[0] for description in cursor.description or []]
|
|
640
|
+
except Exception as exc: # noqa: BLE001 - missing table / connection failure -> no evidence, not a crash
|
|
641
|
+
# WHY: same DSN-in-exception-message risk as `_clear_tool_trace_calls` above -- log only
|
|
642
|
+
# the exception TYPE, never exc_info/str(exc), which can carry the world DB password.
|
|
643
|
+
logger.debug(
|
|
644
|
+
"tool_trace read failed; treating as no evidence: %s", type(exc).__name__
|
|
645
|
+
)
|
|
646
|
+
return ()
|
|
647
|
+
|
|
648
|
+
calls: list[Call] = []
|
|
649
|
+
for row in rows:
|
|
650
|
+
record = dict(zip(columns, row, strict=True))
|
|
651
|
+
name = record.get("name")
|
|
652
|
+
if not isinstance(name, str) or not name:
|
|
653
|
+
continue
|
|
654
|
+
arguments = record.get("arguments")
|
|
655
|
+
if not isinstance(arguments, dict):
|
|
656
|
+
arguments = {}
|
|
657
|
+
ok = bool(record.get("ok", True))
|
|
658
|
+
raw_result = record.get("result")
|
|
659
|
+
if isinstance(raw_result, str):
|
|
660
|
+
result: Any = _truncate(raw_result)
|
|
661
|
+
else:
|
|
662
|
+
# Already parsed JSON (dict/list/etc, psycopg's own jsonb decoding) -- per
|
|
663
|
+
# world-handle-interface.md, only the STRING form is truncated at 2000 chars.
|
|
664
|
+
result = raw_result
|
|
665
|
+
error = _truncate(str(record.get("error") or ""))
|
|
666
|
+
raw_at = record.get("at")
|
|
667
|
+
at = float(raw_at) if isinstance(raw_at, (int, float)) else 0.0
|
|
668
|
+
calls.append(
|
|
669
|
+
Call(
|
|
670
|
+
name=name,
|
|
671
|
+
arguments=arguments,
|
|
672
|
+
result=result,
|
|
673
|
+
ok=ok,
|
|
674
|
+
error=error,
|
|
675
|
+
refused=not ok,
|
|
676
|
+
at=at,
|
|
677
|
+
)
|
|
678
|
+
)
|
|
679
|
+
return tuple(calls)
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
def _tool_trace_file(runtime: EnvironmentRuntime) -> Path | None:
|
|
683
|
+
raw = runtime.metadata.get("tool_trace_path")
|
|
684
|
+
return Path(raw) if isinstance(raw, str) and raw.strip() else None
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def _clear_file_tool_calls(runtime: EnvironmentRuntime) -> None:
|
|
688
|
+
path = _tool_trace_file(runtime)
|
|
689
|
+
if path is None:
|
|
690
|
+
return
|
|
691
|
+
try:
|
|
692
|
+
path.unlink(missing_ok=True)
|
|
693
|
+
except OSError:
|
|
694
|
+
logger.debug(
|
|
695
|
+
"file tool_trace clear failed; continuing without blocking the call"
|
|
696
|
+
)
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def _collect_file_tool_calls(runtime: EnvironmentRuntime) -> tuple[Call, ...]:
|
|
700
|
+
path = _tool_trace_file(runtime)
|
|
701
|
+
if path is None or not path.is_file():
|
|
702
|
+
return ()
|
|
703
|
+
calls: list[Call] = []
|
|
704
|
+
try:
|
|
705
|
+
lines = path.read_text(encoding="utf-8").splitlines()
|
|
706
|
+
except OSError:
|
|
707
|
+
return ()
|
|
708
|
+
for line in lines:
|
|
709
|
+
try:
|
|
710
|
+
record = json.loads(line)
|
|
711
|
+
except ValueError:
|
|
712
|
+
continue
|
|
713
|
+
if not isinstance(record, dict):
|
|
714
|
+
continue
|
|
715
|
+
name = record.get("name")
|
|
716
|
+
if not isinstance(name, str) or not name:
|
|
717
|
+
continue
|
|
718
|
+
arguments = record.get("arguments")
|
|
719
|
+
if isinstance(arguments, str):
|
|
720
|
+
try:
|
|
721
|
+
arguments = json.loads(arguments)
|
|
722
|
+
except ValueError:
|
|
723
|
+
arguments = {"raw": arguments}
|
|
724
|
+
if not isinstance(arguments, dict):
|
|
725
|
+
arguments = {}
|
|
726
|
+
is_error = bool(record.get("is_error", False))
|
|
727
|
+
output = record.get("output")
|
|
728
|
+
calls.append(
|
|
729
|
+
Call(
|
|
730
|
+
name=name,
|
|
731
|
+
arguments=arguments,
|
|
732
|
+
result=None if is_error else output,
|
|
733
|
+
ok=not is_error,
|
|
734
|
+
error=str(output) if is_error and output is not None else None,
|
|
735
|
+
)
|
|
736
|
+
)
|
|
737
|
+
return tuple(calls)
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
def _collect_provider_tool_calls(case: Any) -> tuple[Call, ...]:
|
|
741
|
+
"""Translate provider-reported tool evidence into scheduler calls.
|
|
742
|
+
|
|
743
|
+
LiveKit's legacy report conversion stores per-case evidence in
|
|
744
|
+
``result.metadata.evidence``; canonical reports may populate
|
|
745
|
+
``case.evidence`` directly. Accept both shapes so hosted execution is not
|
|
746
|
+
coupled to the report representation.
|
|
747
|
+
"""
|
|
748
|
+
sources: list[Any] = list(getattr(case, "evidence", None) or [])
|
|
749
|
+
result = getattr(case, "result", None)
|
|
750
|
+
result_metadata = getattr(result, "metadata", None)
|
|
751
|
+
if isinstance(result_metadata, Mapping):
|
|
752
|
+
embedded = result_metadata.get("evidence")
|
|
753
|
+
if isinstance(embedded, list):
|
|
754
|
+
sources.extend(embedded)
|
|
755
|
+
|
|
756
|
+
calls: list[Call] = []
|
|
757
|
+
for source in sources:
|
|
758
|
+
if hasattr(source, "model_dump"):
|
|
759
|
+
source = source.model_dump(mode="json", exclude_none=True)
|
|
760
|
+
if not isinstance(source, Mapping):
|
|
761
|
+
continue
|
|
762
|
+
metadata = source.get("metadata")
|
|
763
|
+
if not isinstance(metadata, Mapping):
|
|
764
|
+
continue
|
|
765
|
+
raw_calls = metadata.get("tool_calls")
|
|
766
|
+
if not isinstance(raw_calls, list):
|
|
767
|
+
continue
|
|
768
|
+
for raw in raw_calls:
|
|
769
|
+
if not isinstance(raw, Mapping):
|
|
770
|
+
continue
|
|
771
|
+
name = raw.get("name")
|
|
772
|
+
if not isinstance(name, str) or not name:
|
|
773
|
+
continue
|
|
774
|
+
arguments: Any = raw.get("arguments")
|
|
775
|
+
if isinstance(arguments, str):
|
|
776
|
+
try:
|
|
777
|
+
arguments = json.loads(arguments)
|
|
778
|
+
except ValueError:
|
|
779
|
+
arguments = {"raw": arguments}
|
|
780
|
+
if not isinstance(arguments, dict):
|
|
781
|
+
arguments = {}
|
|
782
|
+
ok = bool(raw.get("ok", True))
|
|
783
|
+
raw_at = raw.get("at")
|
|
784
|
+
at = float(raw_at) if isinstance(raw_at, (int, float)) else 0.0
|
|
785
|
+
calls.append(
|
|
786
|
+
Call(
|
|
787
|
+
name=name,
|
|
788
|
+
arguments=arguments,
|
|
789
|
+
result=raw.get("result") if ok else None,
|
|
790
|
+
ok=ok,
|
|
791
|
+
error=str(raw.get("error") or ""),
|
|
792
|
+
refused=not ok,
|
|
793
|
+
at=at,
|
|
794
|
+
)
|
|
795
|
+
)
|
|
796
|
+
return tuple(calls)
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
def _truncate(value: str, *, limit: int = _RESULT_TRUNCATE_CHARS) -> str:
|
|
800
|
+
return value if len(value) <= limit else value[:limit]
|
|
801
|
+
|
|
802
|
+
|
|
803
|
+
def _materialize_vertex_adc(
|
|
804
|
+
secret_values: Mapping[str, str],
|
|
805
|
+
work_directory: Path,
|
|
806
|
+
environ: dict[str, str],
|
|
807
|
+
) -> Path | None:
|
|
808
|
+
"""Materialize caller-lane Vertex credentials for Google ADC.
|
|
809
|
+
|
|
810
|
+
API-key auth needs no file. For Vertex, GOOGLE_APPLICATION_CREDENTIALS_JSON is resolved from
|
|
811
|
+
the platform vault and written mode-0600 under the job work directory; the sandbox is
|
|
812
|
+
ephemeral and the file is removed when the guest exits/deletes.
|
|
813
|
+
"""
|
|
814
|
+
raw = secret_values.get(GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS)
|
|
815
|
+
if not raw or environ.get(GOOGLE_APPLICATION_CREDENTIALS_ALIAS):
|
|
816
|
+
return None
|
|
817
|
+
try:
|
|
818
|
+
parsed = json.loads(raw)
|
|
819
|
+
except json.JSONDecodeError as exc:
|
|
820
|
+
raise CallAborted(
|
|
821
|
+
"voice_capability_unavailable: GOOGLE_APPLICATION_CREDENTIALS_JSON is invalid"
|
|
822
|
+
) from exc
|
|
823
|
+
if not isinstance(parsed, dict):
|
|
824
|
+
raise CallAborted(
|
|
825
|
+
"voice_capability_unavailable: GOOGLE_APPLICATION_CREDENTIALS_JSON must be an object"
|
|
826
|
+
)
|
|
827
|
+
credential_dir = work_directory / ".caller-credentials"
|
|
828
|
+
credential_dir.mkdir(parents=True, exist_ok=True)
|
|
829
|
+
fd, path_text = tempfile.mkstemp(
|
|
830
|
+
prefix="google-", suffix=".json", dir=credential_dir
|
|
831
|
+
)
|
|
832
|
+
path = Path(path_text)
|
|
833
|
+
try:
|
|
834
|
+
os.write(fd, raw.encode("utf-8"))
|
|
835
|
+
finally:
|
|
836
|
+
os.close(fd)
|
|
837
|
+
path.chmod(stat.S_IRUSR | stat.S_IWUSR)
|
|
838
|
+
environ[GOOGLE_APPLICATION_CREDENTIALS_ALIAS] = str(path)
|
|
839
|
+
return path
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
# --- the runner ----------------------------------------------------------------------------
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
class CallRunnerImpl:
|
|
846
|
+
"""Satisfies `hosted_scheduler.CallRunner`. See the module docstring for the three
|
|
847
|
+
sub-systems this class implements."""
|
|
848
|
+
|
|
849
|
+
def __init__(
|
|
850
|
+
self,
|
|
851
|
+
adapter: ArtifactUploader,
|
|
852
|
+
context: CallRunnerContext,
|
|
853
|
+
*,
|
|
854
|
+
place_call: PlaceCall | None = None,
|
|
855
|
+
environ: dict[str, str] | None = None,
|
|
856
|
+
) -> None:
|
|
857
|
+
self._adapter = adapter
|
|
858
|
+
self._context = context
|
|
859
|
+
self._place_call = place_call or _default_place_call
|
|
860
|
+
simulator_secret_values = _canonical_simulator_secrets(
|
|
861
|
+
context.simulator_provider_secret_values
|
|
862
|
+
)
|
|
863
|
+
# Local SDK runs remain BYOK and historically carry simulator keys in the one local
|
|
864
|
+
# target map. Hosted runs deliberately do not fall back: their simulator credentials must
|
|
865
|
+
# come from platform configuration and must not be confused with customer-agent keys.
|
|
866
|
+
if context.job.execution is ExecutionMode.LOCAL:
|
|
867
|
+
for alias in (
|
|
868
|
+
LIVEKIT_URL_ALIAS,
|
|
869
|
+
LIVEKIT_API_KEY_ALIAS,
|
|
870
|
+
LIVEKIT_API_SECRET_ALIAS,
|
|
871
|
+
DEEPGRAM_API_KEY_ALIAS,
|
|
872
|
+
CARTESIA_API_KEY_ALIAS,
|
|
873
|
+
GEMINI_API_KEY_ALIAS,
|
|
874
|
+
GOOGLE_API_KEY_ALIAS,
|
|
875
|
+
GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS,
|
|
876
|
+
GOOGLE_CLOUD_PROJECT_ALIAS,
|
|
877
|
+
GOOGLE_CLOUD_LOCATION_ALIAS,
|
|
878
|
+
GOOGLE_GENAI_USE_VERTEXAI_ALIAS,
|
|
879
|
+
OPENAI_API_KEY_ALIAS,
|
|
880
|
+
SIMULATOR_LLM_PROVIDER_ALIAS,
|
|
881
|
+
SIMULATOR_LLM_MODEL_ALIAS,
|
|
882
|
+
SIMULATOR_STT_PROVIDER_ALIAS,
|
|
883
|
+
SIMULATOR_STT_MODEL_ALIAS,
|
|
884
|
+
SIMULATOR_TTS_PROVIDER_ALIAS,
|
|
885
|
+
SIMULATOR_TTS_MODEL_ALIAS,
|
|
886
|
+
):
|
|
887
|
+
if alias not in simulator_secret_values:
|
|
888
|
+
value = context.target_provider_secret_values.get(alias)
|
|
889
|
+
if value:
|
|
890
|
+
simulator_secret_values[alias] = value
|
|
891
|
+
# WHY: the underlying LiveKit engine reads these directly via `os.environ.get(...)` deep
|
|
892
|
+
# inside `engines/livekit.py` / `livekit_models.py` -- they are NOT `SimulationSpec`
|
|
893
|
+
# fields, so there is no other way to hand them over. Exported ONCE here, at construction,
|
|
894
|
+
# not per-call: the values are job-level (the same secret for every scenario/attempt on
|
|
895
|
+
# this job) and W>1 means each world's CallRunner.run() executes inside this SAME guest
|
|
896
|
+
# process but against a per-world sandboxed agent process reached over the network; no
|
|
897
|
+
# other in-process worker races this job-level environment.
|
|
898
|
+
target_environ = os.environ if environ is None else environ
|
|
899
|
+
connector = _resolve_connector(
|
|
900
|
+
context.job, context.target_provider_secret_values
|
|
901
|
+
)
|
|
902
|
+
target_aliases = [VAPI_API_KEY_ALIAS, RETELL_API_KEY_ALIAS]
|
|
903
|
+
if connector == "livekit":
|
|
904
|
+
target_aliases.extend(
|
|
905
|
+
[LIVEKIT_API_KEY_ALIAS, LIVEKIT_API_SECRET_ALIAS, LIVEKIT_URL_ALIAS]
|
|
906
|
+
)
|
|
907
|
+
for alias in target_aliases:
|
|
908
|
+
value = context.target_provider_secret_values.get(alias)
|
|
909
|
+
if value:
|
|
910
|
+
target_environ[alias] = value
|
|
911
|
+
for alias in (
|
|
912
|
+
LIVEKIT_URL_ALIAS,
|
|
913
|
+
LIVEKIT_API_KEY_ALIAS,
|
|
914
|
+
LIVEKIT_API_SECRET_ALIAS,
|
|
915
|
+
DEEPGRAM_API_KEY_ALIAS,
|
|
916
|
+
CARTESIA_API_KEY_ALIAS,
|
|
917
|
+
GEMINI_API_KEY_ALIAS,
|
|
918
|
+
GOOGLE_API_KEY_ALIAS,
|
|
919
|
+
GOOGLE_APPLICATION_CREDENTIALS_JSON_ALIAS,
|
|
920
|
+
GOOGLE_CLOUD_PROJECT_ALIAS,
|
|
921
|
+
GOOGLE_CLOUD_LOCATION_ALIAS,
|
|
922
|
+
GOOGLE_GENAI_USE_VERTEXAI_ALIAS,
|
|
923
|
+
OPENAI_API_KEY_ALIAS,
|
|
924
|
+
SIMULATOR_LLM_PROVIDER_ALIAS,
|
|
925
|
+
SIMULATOR_LLM_MODEL_ALIAS,
|
|
926
|
+
SIMULATOR_STT_PROVIDER_ALIAS,
|
|
927
|
+
SIMULATOR_STT_MODEL_ALIAS,
|
|
928
|
+
SIMULATOR_TTS_PROVIDER_ALIAS,
|
|
929
|
+
SIMULATOR_TTS_MODEL_ALIAS,
|
|
930
|
+
BACKGROUND_NOISE_ALIAS,
|
|
931
|
+
BACKGROUND_NOISE_CATALOG_ALIAS,
|
|
932
|
+
BACKGROUND_NOISE_VOLUME_ALIAS,
|
|
933
|
+
):
|
|
934
|
+
value = simulator_secret_values.get(alias)
|
|
935
|
+
if value:
|
|
936
|
+
# Platform-owned simulator credentials already present in the hosted control
|
|
937
|
+
# process win. Target credentials are retained only as the local-SDK fallback.
|
|
938
|
+
target_environ.setdefault(alias, value)
|
|
939
|
+
self._environ = target_environ
|
|
940
|
+
self._adc_path = _materialize_vertex_adc(
|
|
941
|
+
simulator_secret_values,
|
|
942
|
+
context.work_directory,
|
|
943
|
+
target_environ,
|
|
944
|
+
)
|
|
945
|
+
atexit.register(self._cleanup_credentials)
|
|
946
|
+
self._livekit_url = str(
|
|
947
|
+
context.job.agent.config.get(LIVEKIT_URL_CONFIG_KEY)
|
|
948
|
+
or (
|
|
949
|
+
context.target_provider_secret_values.get(LIVEKIT_URL_ALIAS)
|
|
950
|
+
if connector == "livekit"
|
|
951
|
+
else simulator_secret_values.get(LIVEKIT_URL_ALIAS)
|
|
952
|
+
)
|
|
953
|
+
or ""
|
|
954
|
+
)
|
|
955
|
+
self._missing_config = _check_config(
|
|
956
|
+
context.job,
|
|
957
|
+
context.target_provider_secret_values,
|
|
958
|
+
simulator_secret_values,
|
|
959
|
+
)
|
|
960
|
+
self._scenario_attempt_counts: dict[str, int] = {}
|
|
961
|
+
self._closed = False
|
|
962
|
+
|
|
963
|
+
def _cleanup_credentials(self) -> None:
|
|
964
|
+
if self._adc_path is None:
|
|
965
|
+
return
|
|
966
|
+
try:
|
|
967
|
+
self._adc_path.unlink(missing_ok=True)
|
|
968
|
+
except OSError:
|
|
969
|
+
pass
|
|
970
|
+
if self._environ.get(GOOGLE_APPLICATION_CREDENTIALS_ALIAS) == str(
|
|
971
|
+
self._adc_path
|
|
972
|
+
):
|
|
973
|
+
self._environ.pop(GOOGLE_APPLICATION_CREDENTIALS_ALIAS, None)
|
|
974
|
+
self._adc_path = None
|
|
975
|
+
|
|
976
|
+
async def close(self) -> None:
|
|
977
|
+
"""Release job-scoped resources before ``asyncio.run`` closes its event loop.
|
|
978
|
+
|
|
979
|
+
LiveKit's Python objects own native FFI handles and several of them participate in
|
|
980
|
+
reference cycles. Leaving those cycles to interpreter shutdown lets their finalizers run
|
|
981
|
+
after LiveKit's callback loop has closed; sufficiently long jobs then abort in the native
|
|
982
|
+
FFI teardown even though every call and artifact already completed. Collect on the event
|
|
983
|
+
loop thread and yield twice so queued FFI callbacks drain while their loop is still valid.
|
|
984
|
+
"""
|
|
985
|
+
if self._closed:
|
|
986
|
+
return
|
|
987
|
+
self._closed = True
|
|
988
|
+
self._cleanup_credentials()
|
|
989
|
+
atexit.unregister(self._cleanup_credentials)
|
|
990
|
+
gc.collect()
|
|
991
|
+
# Native RTC shutdown is not synchronous with the Python objects that requested it.
|
|
992
|
+
# Give finalizers and already-enqueued disconnect/drop-handle callbacks real scheduling
|
|
993
|
+
# windows while the loop is still alive. Do not mutate LiveKit's private FFI subscriber
|
|
994
|
+
# list here: a subscriber is owned by its AudioStream task, and removing its queue behind
|
|
995
|
+
# that task's back produces stranded coroutines (observed after a 50-call soak as
|
|
996
|
+
# ``cannot reuse already awaited coroutine``). Deterministic collection on the live loop
|
|
997
|
+
# addresses the shutdown-order problem without violating stream ownership.
|
|
998
|
+
await asyncio.sleep(0.25)
|
|
999
|
+
gc.collect()
|
|
1000
|
+
await asyncio.sleep(0.25)
|
|
1001
|
+
|
|
1002
|
+
async def run(
|
|
1003
|
+
self,
|
|
1004
|
+
scenario: HostedScenario,
|
|
1005
|
+
runtime: EnvironmentRuntime,
|
|
1006
|
+
*,
|
|
1007
|
+
world: Any | None = None,
|
|
1008
|
+
) -> CallOutcome:
|
|
1009
|
+
del world # Voice tools cross the declared evidence seam; they are not response-carried.
|
|
1010
|
+
if self._missing_config is not None:
|
|
1011
|
+
# Pre-dial: dialing never starts, so no partial -- and never `WorldUnavailable` (that
|
|
1012
|
+
# code is reserved by the contract for a world-level capability mismatch, not a
|
|
1013
|
+
# job-level voice config gap).
|
|
1014
|
+
raise CallAborted(self._missing_config.message())
|
|
1015
|
+
|
|
1016
|
+
connector = _resolve_connector(
|
|
1017
|
+
self._context.job, self._context.target_provider_secret_values
|
|
1018
|
+
)
|
|
1019
|
+
agent_name = _dispatch_agent_name(runtime) if connector == "livekit" else None
|
|
1020
|
+
if connector == "livekit" and agent_name is None:
|
|
1021
|
+
raise CallAborted(
|
|
1022
|
+
"voice_dispatch_identity_unavailable: runtime.metadata['livekit_agent_name'] is "
|
|
1023
|
+
f"not set for world {runtime.world_index}"
|
|
1024
|
+
)
|
|
1025
|
+
|
|
1026
|
+
try:
|
|
1027
|
+
doc = _read_scenario_document(
|
|
1028
|
+
self._context.bundle_dir, scenario.scenario_key
|
|
1029
|
+
)
|
|
1030
|
+
except _ScenarioDocumentUnavailable as exc:
|
|
1031
|
+
raise CallAborted(f"voice_scenario_document_unavailable: {exc}") from exc
|
|
1032
|
+
|
|
1033
|
+
scenario_attempt = (
|
|
1034
|
+
self._scenario_attempt_counts.get(scenario.scenario_key, 0) + 1
|
|
1035
|
+
)
|
|
1036
|
+
self._scenario_attempt_counts[scenario.scenario_key] = scenario_attempt
|
|
1037
|
+
room_name = _room_name(
|
|
1038
|
+
job_id=self._context.job.job_id,
|
|
1039
|
+
attempt_number=self._context.attempt_number,
|
|
1040
|
+
scenario_key=scenario.scenario_key,
|
|
1041
|
+
scenario_attempt=scenario_attempt,
|
|
1042
|
+
)
|
|
1043
|
+
|
|
1044
|
+
raw_timeout = self._context.job.agent.config.get(CALL_TIMEOUT_CONFIG_KEY)
|
|
1045
|
+
call_timeout_seconds = (
|
|
1046
|
+
float(raw_timeout)
|
|
1047
|
+
if isinstance(raw_timeout, (int, float))
|
|
1048
|
+
else _DEFAULT_CALL_TIMEOUT_SECONDS
|
|
1049
|
+
)
|
|
1050
|
+
run_seconds = (
|
|
1051
|
+
call_timeout_seconds
|
|
1052
|
+
+ CONNECT_TIMEOUT_SECONDS
|
|
1053
|
+
+ READINESS_TIMEOUT_SECONDS
|
|
1054
|
+
+ CLEANUP_TIMEOUT_SECONDS
|
|
1055
|
+
+ _RUN_SECONDS_PAD_SECONDS
|
|
1056
|
+
)
|
|
1057
|
+
|
|
1058
|
+
# The engine reads this from the environment at call time, so it is set per
|
|
1059
|
+
# scenario and cleared otherwise rather than leaking into the next call.
|
|
1060
|
+
noise = scenario_source(
|
|
1061
|
+
doc.get("background_noise"),
|
|
1062
|
+
doc.get("fixture"),
|
|
1063
|
+
seed=str(doc.get("name") or ""),
|
|
1064
|
+
)
|
|
1065
|
+
if noise:
|
|
1066
|
+
self._environ["HARNESS_BACKGROUND_NOISE"] = noise
|
|
1067
|
+
else:
|
|
1068
|
+
self._environ.pop("HARNESS_BACKGROUND_NOISE", None)
|
|
1069
|
+
|
|
1070
|
+
# Read the same way and for the same reason as the noise source above: the simulator's
|
|
1071
|
+
# instructions are built deep inside simulator_definition, which sees the environment and
|
|
1072
|
+
# not this scenario. Set per scenario and cleared otherwise so one outbound scenario cannot
|
|
1073
|
+
# frame the next inbound one.
|
|
1074
|
+
# A scenario that names its own direction wins. Otherwise the contract's, which the
|
|
1075
|
+
# understand stage read off the agent's own instructions and `hosted_entrypoint` puts here
|
|
1076
|
+
# for this process. Not an operator setting: whether an agent places calls or answers them
|
|
1077
|
+
# is a fact about the agent, so there is nothing for a run to choose.
|
|
1078
|
+
direction = (
|
|
1079
|
+
str(
|
|
1080
|
+
doc.get("call_direction")
|
|
1081
|
+
or os.environ.get(CALL_DIRECTION_ALIAS)
|
|
1082
|
+
or "inbound"
|
|
1083
|
+
)
|
|
1084
|
+
.strip()
|
|
1085
|
+
.lower()
|
|
1086
|
+
)
|
|
1087
|
+
if direction == "outbound":
|
|
1088
|
+
self._environ["HARNESS_CALL_DIRECTION"] = direction
|
|
1089
|
+
awareness = str(doc.get("caller_awareness") or "").strip().lower()
|
|
1090
|
+
if awareness:
|
|
1091
|
+
self._environ["HARNESS_CALLER_AWARENESS"] = awareness
|
|
1092
|
+
else:
|
|
1093
|
+
self._environ.pop("HARNESS_CALLER_AWARENESS", None)
|
|
1094
|
+
# Cleared otherwise, so one voicemail scenario cannot silence the next caller.
|
|
1095
|
+
if (
|
|
1096
|
+
voicemail_enabled()
|
|
1097
|
+
and str(doc.get("answered_by") or "").strip().lower() == "voicemail"
|
|
1098
|
+
):
|
|
1099
|
+
self._environ["HARNESS_ANSWERED_BY"] = "voicemail"
|
|
1100
|
+
# Which kind of mailbox, which decides the greeting and whether a tone follows it.
|
|
1101
|
+
style = str(doc.get("voicemail_style") or "").strip().lower()
|
|
1102
|
+
if style:
|
|
1103
|
+
self._environ["HARNESS_VOICEMAIL_STYLE"] = style
|
|
1104
|
+
else:
|
|
1105
|
+
self._environ.pop("HARNESS_VOICEMAIL_STYLE", None)
|
|
1106
|
+
# A recorded greeting where the catalogue has one for this style AND language. It
|
|
1107
|
+
# replaces the spoken greeting rather than joining it.
|
|
1108
|
+
languages = doc.get("languages") or []
|
|
1109
|
+
chosen = clip_for(
|
|
1110
|
+
style or DEFAULT_VOICEMAIL_STYLE,
|
|
1111
|
+
str(languages[0]) if languages else "",
|
|
1112
|
+
)
|
|
1113
|
+
if chosen:
|
|
1114
|
+
self._environ[VOICEMAIL_CLIP_ALIAS] = chosen["source"]
|
|
1115
|
+
self._environ[VOICEMAIL_CLIP_TONE_ALIAS] = (
|
|
1116
|
+
"1" if chosen["has_tone"] else "0"
|
|
1117
|
+
)
|
|
1118
|
+
if chosen.get("transcript"):
|
|
1119
|
+
self._environ[VOICEMAIL_CLIP_TEXT_ALIAS] = chosen["transcript"]
|
|
1120
|
+
else:
|
|
1121
|
+
self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
|
|
1122
|
+
else:
|
|
1123
|
+
self._environ.pop(VOICEMAIL_CLIP_ALIAS, None)
|
|
1124
|
+
self._environ.pop(VOICEMAIL_CLIP_TONE_ALIAS, None)
|
|
1125
|
+
self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
|
|
1126
|
+
else:
|
|
1127
|
+
self._environ.pop("HARNESS_ANSWERED_BY", None)
|
|
1128
|
+
self._environ.pop("HARNESS_VOICEMAIL_STYLE", None)
|
|
1129
|
+
self._environ.pop(VOICEMAIL_CLIP_ALIAS, None)
|
|
1130
|
+
self._environ.pop(VOICEMAIL_CLIP_TONE_ALIAS, None)
|
|
1131
|
+
self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
|
|
1132
|
+
else:
|
|
1133
|
+
self._environ.pop("HARNESS_CALL_DIRECTION", None)
|
|
1134
|
+
self._environ.pop("HARNESS_CALLER_AWARENESS", None)
|
|
1135
|
+
self._environ.pop("HARNESS_ANSWERED_BY", None)
|
|
1136
|
+
self._environ.pop("HARNESS_VOICEMAIL_STYLE", None)
|
|
1137
|
+
self._environ.pop(VOICEMAIL_CLIP_ALIAS, None)
|
|
1138
|
+
self._environ.pop(VOICEMAIL_CLIP_TONE_ALIAS, None)
|
|
1139
|
+
self._environ.pop(VOICEMAIL_CLIP_TEXT_ALIAS, None)
|
|
1140
|
+
|
|
1141
|
+
provider_target_key = {"vapi": "assistant_id", "retell": "agent_id"}.get(
|
|
1142
|
+
connector
|
|
1143
|
+
)
|
|
1144
|
+
provider_target_id: str | None = None
|
|
1145
|
+
if provider_target_key and self._context.job.agent.mode in {
|
|
1146
|
+
None,
|
|
1147
|
+
ProviderExecutionMode.CONNECT_ONLY,
|
|
1148
|
+
}:
|
|
1149
|
+
provider_target_id = str(
|
|
1150
|
+
self._context.job.agent.config.get(provider_target_key) or ""
|
|
1151
|
+
).strip()
|
|
1152
|
+
if provider_target_key and self._context.job.agent.mode not in {
|
|
1153
|
+
None,
|
|
1154
|
+
ProviderExecutionMode.CONNECT_ONLY,
|
|
1155
|
+
}:
|
|
1156
|
+
dynamic_target = runtime.metadata.get("provider_target_id")
|
|
1157
|
+
provider_target_id = (
|
|
1158
|
+
dynamic_target.strip()
|
|
1159
|
+
if isinstance(dynamic_target, str) and dynamic_target.strip()
|
|
1160
|
+
else None
|
|
1161
|
+
)
|
|
1162
|
+
|
|
1163
|
+
spec = _build_spec(
|
|
1164
|
+
run_id=new_run_id(),
|
|
1165
|
+
room_name=room_name,
|
|
1166
|
+
connector=connector,
|
|
1167
|
+
agent_name=agent_name,
|
|
1168
|
+
provider_target_id=provider_target_id,
|
|
1169
|
+
doc=doc,
|
|
1170
|
+
simulator_config=self._context.job.agent.config,
|
|
1171
|
+
environ=self._environ,
|
|
1172
|
+
livekit_url=self._livekit_url,
|
|
1173
|
+
call_timeout_seconds=call_timeout_seconds,
|
|
1174
|
+
run_seconds=run_seconds,
|
|
1175
|
+
recordings_root=self._context.work_directory / "voice-calls",
|
|
1176
|
+
)
|
|
1177
|
+
|
|
1178
|
+
if self._context.evidence_seam is EvidenceSeam.TOOL_TRACE:
|
|
1179
|
+
_clear_file_tool_calls(runtime)
|
|
1180
|
+
endpoint = _find_postgres_endpoint(runtime)
|
|
1181
|
+
if endpoint is not None:
|
|
1182
|
+
_clear_tool_trace_calls(endpoint.address)
|
|
1183
|
+
|
|
1184
|
+
started_at = datetime.now(timezone.utc)
|
|
1185
|
+
outer_timeout = run_seconds + _OUTER_WAIT_FOR_PAD_SECONDS
|
|
1186
|
+
try:
|
|
1187
|
+
report = await asyncio.wait_for(
|
|
1188
|
+
self._place_call(spec), timeout=outer_timeout
|
|
1189
|
+
)
|
|
1190
|
+
except asyncio.CancelledError:
|
|
1191
|
+
raise
|
|
1192
|
+
except asyncio.TimeoutError as exc:
|
|
1193
|
+
raise CallAborted(
|
|
1194
|
+
"voice_call_runner_timeout: place_call exceeded its outer budget "
|
|
1195
|
+
f"({outer_timeout:.0f}s)",
|
|
1196
|
+
partial=self._timing_only_outcome(started_at),
|
|
1197
|
+
) from exc
|
|
1198
|
+
except Exception as exc: # noqa: BLE001 - post-dial machinery failure, never let it escape raw
|
|
1199
|
+
raise CallAborted(
|
|
1200
|
+
f"voice_call_runner_crashed: {type(exc).__name__}: {exc}",
|
|
1201
|
+
partial=self._timing_only_outcome(started_at),
|
|
1202
|
+
) from exc
|
|
1203
|
+
|
|
1204
|
+
try:
|
|
1205
|
+
return await self._translate_report(
|
|
1206
|
+
report,
|
|
1207
|
+
runtime=runtime,
|
|
1208
|
+
scenario_key=scenario.scenario_key,
|
|
1209
|
+
started_at=started_at,
|
|
1210
|
+
)
|
|
1211
|
+
except (CallAborted, WorldUnavailable):
|
|
1212
|
+
# `_translate_report`'s own typed control-flow (non-completed status, no test case,
|
|
1213
|
+
# agent-never-joined) -- never re-wrap an intentional abort.
|
|
1214
|
+
raise
|
|
1215
|
+
except Exception as exc: # noqa: BLE001 - a transcript/recording read or upload surprise
|
|
1216
|
+
# must never lose the timing this call already measured (the receipt's `call` field
|
|
1217
|
+
# must not be null once the call has genuinely started) by escaping run() raw.
|
|
1218
|
+
raise CallAborted(
|
|
1219
|
+
f"voice_call_translate_crashed: {type(exc).__name__}: {exc}",
|
|
1220
|
+
partial=self._timing_only_outcome(started_at),
|
|
1221
|
+
) from exc
|
|
1222
|
+
|
|
1223
|
+
def _timing_only_outcome(self, started_at: datetime) -> CallOutcome:
|
|
1224
|
+
ended_at = datetime.now(timezone.utc)
|
|
1225
|
+
return CallOutcome(
|
|
1226
|
+
calls=(),
|
|
1227
|
+
turns=0,
|
|
1228
|
+
started_at=format_rfc3339_millis(started_at),
|
|
1229
|
+
ended_at=format_rfc3339_millis(ended_at),
|
|
1230
|
+
duration_ms=_duration_ms(started_at, ended_at),
|
|
1231
|
+
)
|
|
1232
|
+
|
|
1233
|
+
async def _translate_report(
|
|
1234
|
+
self,
|
|
1235
|
+
report: SimulationReport,
|
|
1236
|
+
*,
|
|
1237
|
+
runtime: EnvironmentRuntime,
|
|
1238
|
+
scenario_key: str,
|
|
1239
|
+
started_at: datetime,
|
|
1240
|
+
) -> CallOutcome:
|
|
1241
|
+
case = report.test_cases[0] if report.test_cases else None
|
|
1242
|
+
case_started_at = (
|
|
1243
|
+
case.started_at
|
|
1244
|
+
if case is not None and case.started_at is not None
|
|
1245
|
+
else started_at
|
|
1246
|
+
)
|
|
1247
|
+
ended_at = (
|
|
1248
|
+
case.ended_at
|
|
1249
|
+
if case is not None and case.ended_at is not None
|
|
1250
|
+
else datetime.now(timezone.utc)
|
|
1251
|
+
)
|
|
1252
|
+
turns = (
|
|
1253
|
+
len(case.result.messages)
|
|
1254
|
+
if case is not None and case.result is not None
|
|
1255
|
+
else 0
|
|
1256
|
+
)
|
|
1257
|
+
|
|
1258
|
+
transcript_artifact: str | None = None
|
|
1259
|
+
recording_artifacts: list[str] = []
|
|
1260
|
+
# Evidence belongs to the call attempt, not only to successful calls. Collect it before
|
|
1261
|
+
# interpreting the simulator status so a timeout/agent failure still carries the exact
|
|
1262
|
+
# tool activity in its partial receipt. Previously the early CallAborted below discarded
|
|
1263
|
+
# every tool call from failed calls, making a real upstream tool error indistinguishable
|
|
1264
|
+
# from a proxy/transport failure.
|
|
1265
|
+
calls = self._collect_calls(runtime) if case is not None else ()
|
|
1266
|
+
# Provider-hosted agents execute tools outside the guest process, so their
|
|
1267
|
+
# authoritative call evidence is returned by Vapi/Retell after the call.
|
|
1268
|
+
# Preserve provider-native controls such as ``end_call`` because scenario
|
|
1269
|
+
# checks may verify termination ordering. Fall back to that observed stream
|
|
1270
|
+
# when the submitted backend exposes no local trace seam; never infer calls
|
|
1271
|
+
# from transcript prose.
|
|
1272
|
+
if not calls and case is not None:
|
|
1273
|
+
calls = _collect_provider_tool_calls(case)
|
|
1274
|
+
if calls:
|
|
1275
|
+
tool_trace = "\n".join(
|
|
1276
|
+
json.dumps(
|
|
1277
|
+
{
|
|
1278
|
+
"name": call.name,
|
|
1279
|
+
"arguments": call.arguments,
|
|
1280
|
+
"result": call.result,
|
|
1281
|
+
"ok": call.ok,
|
|
1282
|
+
"error": call.error,
|
|
1283
|
+
"refused": call.refused,
|
|
1284
|
+
"at": call.at,
|
|
1285
|
+
},
|
|
1286
|
+
sort_keys=True,
|
|
1287
|
+
default=str,
|
|
1288
|
+
)
|
|
1289
|
+
for call in calls
|
|
1290
|
+
).encode("utf-8")
|
|
1291
|
+
await self._adapter.upload_artifact(
|
|
1292
|
+
tool_trace,
|
|
1293
|
+
kind=ArtifactKind.TOOL_TRACE,
|
|
1294
|
+
scenario_key=scenario_key,
|
|
1295
|
+
)
|
|
1296
|
+
if case is not None and case.result is not None:
|
|
1297
|
+
result = case.result
|
|
1298
|
+
if result.transcript:
|
|
1299
|
+
transcript_payload = json.dumps(
|
|
1300
|
+
{
|
|
1301
|
+
"schema_version": "futureagi.call-transcript.v1",
|
|
1302
|
+
"transcript": result.transcript,
|
|
1303
|
+
"messages": result.messages,
|
|
1304
|
+
},
|
|
1305
|
+
sort_keys=True,
|
|
1306
|
+
default=str,
|
|
1307
|
+
).encode("utf-8")
|
|
1308
|
+
transcript_artifact = await self._adapter.upload_artifact(
|
|
1309
|
+
transcript_payload,
|
|
1310
|
+
kind=ArtifactKind.TRANSCRIPT,
|
|
1311
|
+
scenario_key=scenario_key,
|
|
1312
|
+
)
|
|
1313
|
+
for path_str, kind in (
|
|
1314
|
+
(result.audio_combined_path, ArtifactKind.RECORDING_COMBINED),
|
|
1315
|
+
(result.audio_stereo_path, ArtifactKind.RECORDING_STEREO),
|
|
1316
|
+
(result.audio_input_path, ArtifactKind.RECORDING_CUSTOMER),
|
|
1317
|
+
(result.audio_output_path, ArtifactKind.RECORDING_ASSISTANT),
|
|
1318
|
+
):
|
|
1319
|
+
if not path_str:
|
|
1320
|
+
continue
|
|
1321
|
+
path = Path(path_str)
|
|
1322
|
+
if not path.is_file():
|
|
1323
|
+
continue
|
|
1324
|
+
artifact_id = await self._adapter.upload_artifact(
|
|
1325
|
+
path.read_bytes(),
|
|
1326
|
+
kind=kind,
|
|
1327
|
+
scenario_key=scenario_key,
|
|
1328
|
+
)
|
|
1329
|
+
if artifact_id is not None:
|
|
1330
|
+
recording_artifacts.append(artifact_id)
|
|
1331
|
+
|
|
1332
|
+
base = CallOutcome(
|
|
1333
|
+
calls=calls,
|
|
1334
|
+
turns=turns,
|
|
1335
|
+
started_at=format_rfc3339_millis(case_started_at),
|
|
1336
|
+
ended_at=format_rfc3339_millis(ended_at),
|
|
1337
|
+
duration_ms=_duration_ms(case_started_at, ended_at),
|
|
1338
|
+
transcript_artifact=transcript_artifact,
|
|
1339
|
+
recording_artifacts=tuple(recording_artifacts),
|
|
1340
|
+
messages=tuple(
|
|
1341
|
+
case.result.messages or ()
|
|
1342
|
+
if case is not None and case.result is not None
|
|
1343
|
+
else ()
|
|
1344
|
+
),
|
|
1345
|
+
stop_reason=(
|
|
1346
|
+
str(case.result.metadata.get("stop_reason"))
|
|
1347
|
+
if case is not None
|
|
1348
|
+
and case.result is not None
|
|
1349
|
+
and case.result.metadata.get("stop_reason")
|
|
1350
|
+
else None
|
|
1351
|
+
),
|
|
1352
|
+
)
|
|
1353
|
+
|
|
1354
|
+
if case is None:
|
|
1355
|
+
raise CallAborted(
|
|
1356
|
+
"voice_call_no_test_case: SimulationReport carried no test case",
|
|
1357
|
+
partial=base,
|
|
1358
|
+
)
|
|
1359
|
+
|
|
1360
|
+
if case.status is TestCaseStatus.AGENT_UNAVAILABLE:
|
|
1361
|
+
# world-handle-interface.md: "the agent never joined" is a WORLD failure, not a
|
|
1362
|
+
# scenario one -- the agent is part of the world, so the scheduler retires it and
|
|
1363
|
+
# retries elsewhere. Verified against the engine's own source (engines/livekit.py):
|
|
1364
|
+
# this status fires ONLY on a readiness-stage timeout with a session already started
|
|
1365
|
+
# but no target dispatched -- exactly "dispatch fails, agent never joins," never a
|
|
1366
|
+
# mid-call condition.
|
|
1367
|
+
reason = (
|
|
1368
|
+
case.failure.message
|
|
1369
|
+
if case.failure is not None
|
|
1370
|
+
else "agent_unavailable"
|
|
1371
|
+
)
|
|
1372
|
+
raise WorldUnavailable(f"target agent never joined the room: {reason}")
|
|
1373
|
+
|
|
1374
|
+
# A genuinely silent agent-first call (agent joined, zero conversational turns) reaches
|
|
1375
|
+
# the real engine (engines/livekit.py::_conversation_outcome) as FAILED with code
|
|
1376
|
+
# "no_conversation" or "conversation_silence_timeout" and zero messages -- never as a
|
|
1377
|
+
# COMPLETED case with zero turns (COMPLETED requires >= min_turn_messages AND role
|
|
1378
|
+
# alternation, so the engine cannot produce that shape). Scoped to zero turns only: a
|
|
1379
|
+
# short-but-nonzero conversation on either code still failed the completion bar for a real
|
|
1380
|
+
# reason and must stay a CallAborted below.
|
|
1381
|
+
is_silent_agent = (
|
|
1382
|
+
case.status is TestCaseStatus.FAILED
|
|
1383
|
+
and turns == 0
|
|
1384
|
+
and case.failure is not None
|
|
1385
|
+
and case.failure.code in _SILENT_AGENT_FAILURE_CODES
|
|
1386
|
+
)
|
|
1387
|
+
|
|
1388
|
+
# An intake agent may ask thirty to fifty questions, so a deadline is an ordinary outcome.
|
|
1389
|
+
ran_out_of_time = (
|
|
1390
|
+
case.status is TestCaseStatus.TIMED_OUT
|
|
1391
|
+
and turns >= _GRADEABLE_AFTER_TIMEOUT_TURNS
|
|
1392
|
+
)
|
|
1393
|
+
|
|
1394
|
+
if (
|
|
1395
|
+
case.status is not TestCaseStatus.COMPLETED
|
|
1396
|
+
and not is_silent_agent
|
|
1397
|
+
and not ran_out_of_time
|
|
1398
|
+
):
|
|
1399
|
+
reason = (
|
|
1400
|
+
case.failure.message if case.failure is not None else case.status.value
|
|
1401
|
+
)
|
|
1402
|
+
attributed = _attributed_stall(case)
|
|
1403
|
+
if attributed is not None:
|
|
1404
|
+
code, reason = attributed
|
|
1405
|
+
raise CallAborted(reason, partial=base, code=code)
|
|
1406
|
+
if (
|
|
1407
|
+
case.failure is not None
|
|
1408
|
+
and case.failure.code == "target_agent_tool_failed"
|
|
1409
|
+
):
|
|
1410
|
+
raise CallAborted(
|
|
1411
|
+
case.failure.message,
|
|
1412
|
+
partial=base,
|
|
1413
|
+
code="target_agent_tool_failed",
|
|
1414
|
+
)
|
|
1415
|
+
raise CallAborted(
|
|
1416
|
+
f"voice_call_not_completed: {case.status.value}: {reason}", partial=base
|
|
1417
|
+
)
|
|
1418
|
+
|
|
1419
|
+
# Never fabricate calls for a call that produced no conversation -- the scheduler's own
|
|
1420
|
+
# coverage guarantee turns an empty `calls` tuple into evidence_missing/simulator
|
|
1421
|
+
# regardless of turns (hosted_scheduler.py's own unconditioned-on-turns rule).
|
|
1422
|
+
calls = () if is_silent_agent else base.calls
|
|
1423
|
+
# Copied rather than rebuilt field by field: relisting them dropped `messages` silently,
|
|
1424
|
+
# and the judge then had no transcript to settle a spoken claim against.
|
|
1425
|
+
return replace(base, calls=calls)
|
|
1426
|
+
|
|
1427
|
+
def _collect_calls(self, runtime: EnvironmentRuntime) -> tuple[Call, ...]:
|
|
1428
|
+
seam = self._context.evidence_seam
|
|
1429
|
+
file_calls = _collect_file_tool_calls(runtime)
|
|
1430
|
+
if file_calls:
|
|
1431
|
+
return file_calls
|
|
1432
|
+
if seam is EvidenceSeam.HTTP_TOOL:
|
|
1433
|
+
return _collect_http_tool_calls(runtime)
|
|
1434
|
+
if seam is EvidenceSeam.TOOL_TRACE:
|
|
1435
|
+
return _collect_tool_trace_calls(runtime)
|
|
1436
|
+
# Unrecognized/None (should not happen for a `kind: process` bundle past preflight --
|
|
1437
|
+
# bundle_v2.py requires `evidence_seam` whenever `kind is PROCESS` -- but degrading rather
|
|
1438
|
+
# than crashing keeps this on the scheduler's own evidence_missing path, never a raw
|
|
1439
|
+
# exception).
|
|
1440
|
+
return ()
|