agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,1340 @@
|
|
|
1
|
+
"""FutureAGIResultSink — local write + real platform submission.
|
|
2
|
+
|
|
3
|
+
Composes ``LocalFilesystemResultSink`` and adds ``submit(...)`` that POSTs
|
|
4
|
+
report data to the Future AGI platform using the ALK ingestion endpoints:
|
|
5
|
+
|
|
6
|
+
POST /simulate/alk-simulate/run-tests/{run_test_id}/test-executions/
|
|
7
|
+
POST /simulate/alk-simulate/test-executions/{test_execution_id}/batch/
|
|
8
|
+
PATCH /simulate/alk-simulate/call-executions/{call_execution_id}/result/
|
|
9
|
+
|
|
10
|
+
Configuration is env-driven so local runs stay unaffected when the platform
|
|
11
|
+
target is not set. Submission accepts either the internal service bearer or
|
|
12
|
+
the external API-key pair:
|
|
13
|
+
|
|
14
|
+
FI_BASE_URL / FUTURE_AGI_API_URL / AGENT_LEARNING_API_URL — base URL
|
|
15
|
+
FI_API_KEY / FUTURE_AGI_API_KEY / AGENT_LEARNING_API_KEY — x-api-key
|
|
16
|
+
FI_SECRET_KEY / FUTURE_AGI_SECRET_KEY / AGENT_LEARNING_SECRET_KEY — x-secret-key
|
|
17
|
+
FI_RUN_TEST_ID / FUTURE_AGI_RUN_TEST_ID / AGENT_LEARNING_RUN_TEST_ID — target run test
|
|
18
|
+
FI_TEST_EXECUTION_ID / … — optional pre-created TestExecution (hosted runs); when
|
|
19
|
+
set the sink submits into it instead of creating one from the run test.
|
|
20
|
+
|
|
21
|
+
When any of those are absent the sink records ``status: "not_configured"``
|
|
22
|
+
in ``submission.json`` and returns cleanly — no HTTP is attempted.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import hashlib
|
|
28
|
+
import json
|
|
29
|
+
import logging
|
|
30
|
+
import os
|
|
31
|
+
from datetime import datetime, timezone
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Any
|
|
34
|
+
|
|
35
|
+
import httpx
|
|
36
|
+
|
|
37
|
+
from fi.simulate.runtime import (
|
|
38
|
+
CanonicalEvent,
|
|
39
|
+
SimulationPlan,
|
|
40
|
+
SimulationReport,
|
|
41
|
+
SimulationSpec,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
from .filesystem import LocalFilesystemResultSink
|
|
45
|
+
|
|
46
|
+
logger = logging.getLogger("fi.simulate.results.futureagi")
|
|
47
|
+
|
|
48
|
+
_STATUS_MAP = {
|
|
49
|
+
"completed": "completed",
|
|
50
|
+
"failed": "failed",
|
|
51
|
+
"cancelled": "cancelled",
|
|
52
|
+
"timed_out": "failed",
|
|
53
|
+
"agent_unavailable": "failed",
|
|
54
|
+
}
|
|
55
|
+
_API_KEY_ENV = ("FI_API_KEY", "FUTURE_AGI_API_KEY", "AGENT_LEARNING_API_KEY")
|
|
56
|
+
_SECRET_KEY_ENV = (
|
|
57
|
+
"FI_SECRET_KEY",
|
|
58
|
+
"FUTURE_AGI_SECRET_KEY",
|
|
59
|
+
"AGENT_LEARNING_SECRET_KEY",
|
|
60
|
+
)
|
|
61
|
+
_INTERNAL_SECRET_ENV = (
|
|
62
|
+
"FI_INTERNAL_SUBMIT_SECRET",
|
|
63
|
+
"ALK_RUNNER_INTERNAL_SECRET",
|
|
64
|
+
"INTERNAL_API_SECRET",
|
|
65
|
+
)
|
|
66
|
+
_API_URL_ENV = ("FI_BASE_URL", "FUTURE_AGI_API_URL", "AGENT_LEARNING_API_URL")
|
|
67
|
+
_RUN_TEST_ID_ENV = (
|
|
68
|
+
"FI_RUN_TEST_ID",
|
|
69
|
+
"FUTURE_AGI_RUN_TEST_ID",
|
|
70
|
+
"AGENT_LEARNING_RUN_TEST_ID",
|
|
71
|
+
)
|
|
72
|
+
_TEST_EXECUTION_ID_ENV = (
|
|
73
|
+
"FI_TEST_EXECUTION_ID",
|
|
74
|
+
"FUTURE_AGI_TEST_EXECUTION_ID",
|
|
75
|
+
"AGENT_LEARNING_TEST_EXECUTION_ID",
|
|
76
|
+
)
|
|
77
|
+
_HTTP_TIMEOUT_SECONDS = 60.0
|
|
78
|
+
_RECORDING_UPLOAD_TIMEOUT_SECONDS = 300.0
|
|
79
|
+
_CONTENT_TYPE_BY_EXT = {
|
|
80
|
+
".wav": "audio/wav",
|
|
81
|
+
".mp3": "audio/mpeg",
|
|
82
|
+
".ogg": "audio/ogg",
|
|
83
|
+
".webm": "audio/webm",
|
|
84
|
+
".m4a": "audio/mp4",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class FutureAGIResultSink:
|
|
89
|
+
"""Local sink + platform submission over HTTP."""
|
|
90
|
+
|
|
91
|
+
def __init__(
|
|
92
|
+
self,
|
|
93
|
+
*,
|
|
94
|
+
root: str | Path = ".fagi/runs",
|
|
95
|
+
api_url: str | None = None,
|
|
96
|
+
api_key_env: tuple[str, ...] = _API_KEY_ENV,
|
|
97
|
+
secret_key_env: tuple[str, ...] = _SECRET_KEY_ENV,
|
|
98
|
+
run_test_id: str | None = None,
|
|
99
|
+
test_execution_id: str | None = None,
|
|
100
|
+
) -> None:
|
|
101
|
+
self._local = LocalFilesystemResultSink(root=root)
|
|
102
|
+
self._api_url = api_url or _first_env(_API_URL_ENV)
|
|
103
|
+
self._api_key_env = api_key_env
|
|
104
|
+
self._secret_key_env = secret_key_env
|
|
105
|
+
self._run_test_id = run_test_id or _first_env(_RUN_TEST_ID_ENV)
|
|
106
|
+
self._test_execution_id = test_execution_id or _first_env(
|
|
107
|
+
_TEST_EXECUTION_ID_ENV
|
|
108
|
+
)
|
|
109
|
+
self._event_count = 0
|
|
110
|
+
self._spec: SimulationSpec | None = None
|
|
111
|
+
self._plan: SimulationPlan | None = None
|
|
112
|
+
# Streaming state (hosted runs only). ``begin_stream`` opens the client
|
|
113
|
+
# and allocates rows up front; ``submit_case`` PATCHes one row by index as
|
|
114
|
+
# its case finishes; ``finalize_stream`` reconciles + closes.
|
|
115
|
+
self._streaming = False
|
|
116
|
+
self._stream_client: httpx.Client | None = None
|
|
117
|
+
self._stream_call_ids: list[str] = []
|
|
118
|
+
self._streamed_indices: set[int] = set()
|
|
119
|
+
self._stream_failures: dict[int, dict[str, Any]] = {}
|
|
120
|
+
|
|
121
|
+
@property
|
|
122
|
+
def run_directory(self) -> Path | None:
|
|
123
|
+
return self._local.run_directory
|
|
124
|
+
|
|
125
|
+
def prepare(
|
|
126
|
+
self,
|
|
127
|
+
spec: SimulationSpec,
|
|
128
|
+
plan: SimulationPlan | None = None,
|
|
129
|
+
) -> Path:
|
|
130
|
+
self._spec = spec
|
|
131
|
+
self._plan = plan
|
|
132
|
+
self._event_count = 0
|
|
133
|
+
return self._local.prepare(spec, plan)
|
|
134
|
+
|
|
135
|
+
def write_event(self, event: CanonicalEvent) -> None:
|
|
136
|
+
self._event_count += 1
|
|
137
|
+
self._local.write_event(event)
|
|
138
|
+
|
|
139
|
+
def write_report(self, report: SimulationReport) -> Path:
|
|
140
|
+
report_path = self._local.write_report(report)
|
|
141
|
+
# When streaming, cases were already PATCHed one-by-one as they finished;
|
|
142
|
+
# finalize only reconciles the stragglers and writes submission.json.
|
|
143
|
+
# Otherwise the whole report is submitted here in one batch (local/chat).
|
|
144
|
+
if self._streaming:
|
|
145
|
+
self.finalize_stream(report)
|
|
146
|
+
else:
|
|
147
|
+
self.submit(report)
|
|
148
|
+
return report_path
|
|
149
|
+
|
|
150
|
+
def begin_stream(
|
|
151
|
+
self,
|
|
152
|
+
spec: SimulationSpec,
|
|
153
|
+
plan: SimulationPlan | None = None,
|
|
154
|
+
) -> bool:
|
|
155
|
+
"""Open a run-scoped submission session for per-case streaming.
|
|
156
|
+
|
|
157
|
+
Hosted only: a pre-created ``test_execution_id`` is the signal. Local and
|
|
158
|
+
chat runs (no pre-created execution) return ``False`` and keep the
|
|
159
|
+
batch-at-end path untouched. On any setup error we also return ``False``
|
|
160
|
+
and fall back to that path — streaming setup must never break the run.
|
|
161
|
+
"""
|
|
162
|
+
if not self._test_execution_id:
|
|
163
|
+
return False
|
|
164
|
+
api_key = _first_env(self._api_key_env)
|
|
165
|
+
secret_key = _first_env(self._secret_key_env)
|
|
166
|
+
internal_secret = _first_env(_INTERNAL_SECRET_ENV)
|
|
167
|
+
if _missing_config(
|
|
168
|
+
api_url=self._api_url,
|
|
169
|
+
api_key=api_key,
|
|
170
|
+
secret_key=secret_key,
|
|
171
|
+
internal_secret=internal_secret,
|
|
172
|
+
run_test_id=self._run_test_id,
|
|
173
|
+
):
|
|
174
|
+
return False
|
|
175
|
+
try:
|
|
176
|
+
client = _open_client(self._api_url, api_key, secret_key, internal_secret)
|
|
177
|
+
call_ids = _allocate_call_ids(client, self._test_execution_id)
|
|
178
|
+
except Exception:
|
|
179
|
+
if self._stream_client is not None:
|
|
180
|
+
self._stream_client.close()
|
|
181
|
+
self._stream_client = None
|
|
182
|
+
return False
|
|
183
|
+
|
|
184
|
+
self._stream_client = client
|
|
185
|
+
self._stream_call_ids = call_ids
|
|
186
|
+
self._streamed_indices = set()
|
|
187
|
+
self._stream_failures = {}
|
|
188
|
+
self._streaming = True
|
|
189
|
+
return True
|
|
190
|
+
|
|
191
|
+
def submit_case(self, index: int, case: Any) -> None:
|
|
192
|
+
"""PATCH one finished case into its pre-allocated CallExecution row.
|
|
193
|
+
|
|
194
|
+
Called off the event loop (``asyncio.to_thread``) from the runner's
|
|
195
|
+
per-case callback; ``httpx.Client`` is thread-safe, so the run-scoped
|
|
196
|
+
client is shared across concurrent case submissions. Failures are logged
|
|
197
|
+
and left for ``finalize_stream`` to reconcile — never raised.
|
|
198
|
+
"""
|
|
199
|
+
if not self._streaming or self._stream_client is None:
|
|
200
|
+
return
|
|
201
|
+
if index >= len(self._stream_call_ids):
|
|
202
|
+
# More results than allocated rows: the platform under-provisioned.
|
|
203
|
+
# The batch path drops these silently via ``zip``; record it here so
|
|
204
|
+
# submission.json shows the drop.
|
|
205
|
+
self._stream_failures[index] = {
|
|
206
|
+
"index": index,
|
|
207
|
+
"reason": "no_allocated_row",
|
|
208
|
+
}
|
|
209
|
+
return
|
|
210
|
+
call_id = self._stream_call_ids[index]
|
|
211
|
+
try:
|
|
212
|
+
payload = _build_result_payload(case)
|
|
213
|
+
recording_url = _maybe_upload_recording(self._stream_client, call_id, case)
|
|
214
|
+
if recording_url:
|
|
215
|
+
payload["recording_url"] = recording_url
|
|
216
|
+
stereo_url = _maybe_upload_stereo_recording(
|
|
217
|
+
self._stream_client, call_id, case
|
|
218
|
+
)
|
|
219
|
+
if stereo_url:
|
|
220
|
+
payload["stereo_recording_url"] = stereo_url
|
|
221
|
+
_attach_channel_recordings(self._stream_client, call_id, case, payload)
|
|
222
|
+
_stamp_result_digest(payload)
|
|
223
|
+
resp = self._stream_client.patch(
|
|
224
|
+
f"/simulate/api/alk-simulate/call-executions/{call_id}/result/",
|
|
225
|
+
json=payload,
|
|
226
|
+
)
|
|
227
|
+
if resp.is_error:
|
|
228
|
+
self._stream_failures[index] = {
|
|
229
|
+
"index": index,
|
|
230
|
+
"call_execution_id": call_id,
|
|
231
|
+
"status_code": resp.status_code,
|
|
232
|
+
"body": _safe_body(resp),
|
|
233
|
+
}
|
|
234
|
+
logger.warning(
|
|
235
|
+
"case submission http error",
|
|
236
|
+
extra={
|
|
237
|
+
"case_index": index,
|
|
238
|
+
"call_execution_id": call_id,
|
|
239
|
+
"status_code": resp.status_code,
|
|
240
|
+
},
|
|
241
|
+
)
|
|
242
|
+
return
|
|
243
|
+
self._streamed_indices.add(index)
|
|
244
|
+
self._stream_failures.pop(index, None)
|
|
245
|
+
except Exception as exc:
|
|
246
|
+
self._stream_failures[index] = {
|
|
247
|
+
"index": index,
|
|
248
|
+
"call_execution_id": call_id,
|
|
249
|
+
"error": f"{type(exc).__name__}: {exc}",
|
|
250
|
+
}
|
|
251
|
+
logger.warning(
|
|
252
|
+
"case submission failed",
|
|
253
|
+
extra={
|
|
254
|
+
"case_index": index,
|
|
255
|
+
"call_execution_id": call_id,
|
|
256
|
+
"error": f"{type(exc).__name__}: {exc}",
|
|
257
|
+
},
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
def case_started(self, index: int) -> None:
|
|
261
|
+
"""PATCH a pre-allocated CallExecution row to ONGOING the moment its case
|
|
262
|
+
starts, so the platform shows progress instead of PENDING → terminal.
|
|
263
|
+
|
|
264
|
+
Purely cosmetic and best-effort. Unlike ``submit_case`` this must NOT
|
|
265
|
+
record into ``_stream_failures`` / ``_streamed_indices`` — those drive
|
|
266
|
+
``finalize_stream`` result reconciliation, and a missed status ping is not
|
|
267
|
+
a missed result. The backend gates the update on PENDING, so a lost, late,
|
|
268
|
+
or duplicate ping can never overwrite a terminal result; failures are
|
|
269
|
+
swallowed.
|
|
270
|
+
"""
|
|
271
|
+
if not self._streaming or self._stream_client is None:
|
|
272
|
+
return
|
|
273
|
+
if index >= len(self._stream_call_ids):
|
|
274
|
+
return
|
|
275
|
+
call_id = self._stream_call_ids[index]
|
|
276
|
+
try:
|
|
277
|
+
self._stream_client.patch(
|
|
278
|
+
f"/simulate/api/alk-simulate/call-executions/{call_id}/status/",
|
|
279
|
+
json={"status": "ongoing"},
|
|
280
|
+
)
|
|
281
|
+
except Exception:
|
|
282
|
+
# Non-authoritative: never let a status ping disturb the run or the
|
|
283
|
+
# reconciliation bookkeeping.
|
|
284
|
+
pass
|
|
285
|
+
|
|
286
|
+
def finalize_stream(self, report: SimulationReport) -> dict[str, Any]:
|
|
287
|
+
"""Reconcile any case the stream missed, then close the session.
|
|
288
|
+
|
|
289
|
+
Runs after the engine returns, so no cases are in flight. Any index not
|
|
290
|
+
already streamed (PATCH failure, or a whole-report failure that never fired
|
|
291
|
+
callbacks) is retried here. ``submission.json`` records ``status:
|
|
292
|
+
submitted`` when at least one case landed and ``failed`` when none did —
|
|
293
|
+
``child_entrypoint`` gates job success on that field.
|
|
294
|
+
"""
|
|
295
|
+
run_directory = self._local.run_directory
|
|
296
|
+
for index, case in enumerate(report.test_cases):
|
|
297
|
+
if index in self._streamed_indices:
|
|
298
|
+
continue
|
|
299
|
+
self.submit_case(index, case)
|
|
300
|
+
|
|
301
|
+
submitted = [self._stream_call_ids[i] for i in sorted(self._streamed_indices)]
|
|
302
|
+
failed = [detail for _, detail in sorted(self._stream_failures.items())]
|
|
303
|
+
# Every allocated row failing to submit is a failed submission — not a
|
|
304
|
+
# green job with zero results landed (the batch path signalled this by
|
|
305
|
+
# letting the exception propagate to ``submit``). Partial failures stay
|
|
306
|
+
# "submitted": the cases that did land are real.
|
|
307
|
+
all_failed = (
|
|
308
|
+
bool(report.test_cases)
|
|
309
|
+
and bool(self._stream_call_ids)
|
|
310
|
+
and not self._streamed_indices
|
|
311
|
+
)
|
|
312
|
+
submission: dict[str, Any] = {
|
|
313
|
+
"schema_version": "futureagi.submission.v1",
|
|
314
|
+
"run_id": report.run_id,
|
|
315
|
+
"report_hash": report.report_hash,
|
|
316
|
+
"test_cases": len(report.test_cases),
|
|
317
|
+
"events_recorded": self._event_count,
|
|
318
|
+
"api_url": self._api_url,
|
|
319
|
+
"run_test_id": self._run_test_id,
|
|
320
|
+
"test_execution_id": self._test_execution_id,
|
|
321
|
+
"streamed": True,
|
|
322
|
+
"status": "failed" if all_failed else "submitted",
|
|
323
|
+
"allocated_call_executions": list(self._stream_call_ids),
|
|
324
|
+
"submitted_call_executions": submitted,
|
|
325
|
+
"failed_call_executions": failed,
|
|
326
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
327
|
+
}
|
|
328
|
+
if all_failed:
|
|
329
|
+
submission["reason"] = "stream_all_cases_failed"
|
|
330
|
+
if self._stream_client is not None:
|
|
331
|
+
self._stream_client.close()
|
|
332
|
+
self._stream_client = None
|
|
333
|
+
self._streaming = False
|
|
334
|
+
if run_directory is not None:
|
|
335
|
+
_write_submission(run_directory, submission)
|
|
336
|
+
return submission
|
|
337
|
+
|
|
338
|
+
def submit(self, report: SimulationReport) -> dict[str, Any]:
|
|
339
|
+
run_directory = self._local.run_directory
|
|
340
|
+
if run_directory is None:
|
|
341
|
+
raise RuntimeError("result_sink_not_prepared")
|
|
342
|
+
|
|
343
|
+
api_key = _first_env(self._api_key_env)
|
|
344
|
+
secret_key = _first_env(self._secret_key_env)
|
|
345
|
+
internal_secret = _first_env(_INTERNAL_SECRET_ENV)
|
|
346
|
+
|
|
347
|
+
submission: dict[str, Any] = {
|
|
348
|
+
"schema_version": "futureagi.submission.v1",
|
|
349
|
+
"run_id": report.run_id,
|
|
350
|
+
"report_hash": report.report_hash,
|
|
351
|
+
"test_cases": len(report.test_cases),
|
|
352
|
+
"artifact_count": len(report.artifacts.entries),
|
|
353
|
+
"events_recorded": self._event_count,
|
|
354
|
+
"api_url": self._api_url,
|
|
355
|
+
"run_test_id": self._run_test_id,
|
|
356
|
+
"test_execution_id": self._test_execution_id,
|
|
357
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
missing = _missing_config(
|
|
361
|
+
api_url=self._api_url,
|
|
362
|
+
api_key=api_key,
|
|
363
|
+
secret_key=secret_key,
|
|
364
|
+
internal_secret=internal_secret,
|
|
365
|
+
run_test_id=self._run_test_id,
|
|
366
|
+
)
|
|
367
|
+
if missing:
|
|
368
|
+
submission["status"] = "not_configured"
|
|
369
|
+
submission["reason"] = "missing_config: " + ",".join(missing)
|
|
370
|
+
logger.warning(
|
|
371
|
+
"submission not configured", extra={"missing": ",".join(missing)}
|
|
372
|
+
)
|
|
373
|
+
_write_submission(run_directory, submission)
|
|
374
|
+
return submission
|
|
375
|
+
|
|
376
|
+
try:
|
|
377
|
+
outcome = _submit_via_http(
|
|
378
|
+
report=report,
|
|
379
|
+
base_url=self._api_url,
|
|
380
|
+
api_key=api_key,
|
|
381
|
+
secret_key=secret_key,
|
|
382
|
+
internal_secret=internal_secret,
|
|
383
|
+
run_test_id=self._run_test_id,
|
|
384
|
+
test_execution_id=self._test_execution_id,
|
|
385
|
+
)
|
|
386
|
+
submission.update(outcome)
|
|
387
|
+
submission["status"] = "submitted"
|
|
388
|
+
logger.info(
|
|
389
|
+
"submission ok",
|
|
390
|
+
extra={
|
|
391
|
+
"run_test_id": self._run_test_id,
|
|
392
|
+
"test_execution_id": self._test_execution_id,
|
|
393
|
+
},
|
|
394
|
+
)
|
|
395
|
+
except Exception as exc:
|
|
396
|
+
submission["status"] = "failed"
|
|
397
|
+
submission["reason"] = f"submission_error: {exc.__class__.__name__}: {exc}"
|
|
398
|
+
logger.error(
|
|
399
|
+
"submission failed",
|
|
400
|
+
extra={
|
|
401
|
+
"run_test_id": self._run_test_id,
|
|
402
|
+
"test_execution_id": self._test_execution_id,
|
|
403
|
+
"error": f"{exc.__class__.__name__}: {exc}",
|
|
404
|
+
},
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
_write_submission(run_directory, submission)
|
|
408
|
+
return submission
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def _first_env(names: tuple[str, ...]) -> str | None:
|
|
412
|
+
for name in names:
|
|
413
|
+
value = os.environ.get(name)
|
|
414
|
+
if value:
|
|
415
|
+
return value
|
|
416
|
+
return None
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _missing_config(
|
|
420
|
+
*,
|
|
421
|
+
api_url: str | None,
|
|
422
|
+
api_key: str | None,
|
|
423
|
+
secret_key: str | None,
|
|
424
|
+
internal_secret: str | None,
|
|
425
|
+
run_test_id: str | None,
|
|
426
|
+
) -> list[str]:
|
|
427
|
+
missing: list[str] = []
|
|
428
|
+
if not api_url:
|
|
429
|
+
missing.append("api_url")
|
|
430
|
+
if not internal_secret:
|
|
431
|
+
if not api_key:
|
|
432
|
+
missing.append("api_key")
|
|
433
|
+
if not secret_key:
|
|
434
|
+
missing.append("secret_key")
|
|
435
|
+
if not run_test_id:
|
|
436
|
+
missing.append("run_test_id")
|
|
437
|
+
return missing
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _open_client(
|
|
441
|
+
base_url: str,
|
|
442
|
+
api_key: str | None,
|
|
443
|
+
secret_key: str | None,
|
|
444
|
+
internal_secret: str | None = None,
|
|
445
|
+
) -> httpx.Client:
|
|
446
|
+
"""Build the ALK ingestion HTTP client (shared by batch + streaming paths).
|
|
447
|
+
|
|
448
|
+
No client-level Content-Type: httpx sets application/json for ``json=`` calls
|
|
449
|
+
and multipart/form-data (with boundary) for the ``files=`` recording upload.
|
|
450
|
+
A fixed application/json here silently breaks the multipart upload.
|
|
451
|
+
"""
|
|
452
|
+
headers: dict[str, str] = {}
|
|
453
|
+
if api_key and secret_key:
|
|
454
|
+
headers.update({"x-api-key": api_key, "x-secret-key": secret_key})
|
|
455
|
+
# These are alternative authentication modes. A developer/hosted-runner
|
|
456
|
+
# environment can legitimately contain both sets of variables; sending
|
|
457
|
+
# both lets the backend's bearer authenticator win and loses the customer
|
|
458
|
+
# organization carried by the API-key pair. Prefer the explicit customer
|
|
459
|
+
# identity whenever it is complete, and use the internal token only as the
|
|
460
|
+
# service-to-service fallback.
|
|
461
|
+
elif internal_secret:
|
|
462
|
+
headers["Authorization"] = f"Bearer {internal_secret}"
|
|
463
|
+
return httpx.Client(
|
|
464
|
+
base_url=base_url.rstrip("/"),
|
|
465
|
+
headers=headers,
|
|
466
|
+
timeout=_HTTP_TIMEOUT_SECONDS,
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def _ensure_test_execution(client: httpx.Client, run_test_id: str) -> str:
|
|
471
|
+
"""Create a TestExecution from the run test (local runs only)."""
|
|
472
|
+
start = client.post(
|
|
473
|
+
f"/simulate/api/alk-simulate/run-tests/{run_test_id}/test-executions/",
|
|
474
|
+
json={},
|
|
475
|
+
)
|
|
476
|
+
start.raise_for_status()
|
|
477
|
+
return _unwrap(start.json())["test_execution_id"]
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _allocate_call_ids(
|
|
481
|
+
client: httpx.Client,
|
|
482
|
+
test_execution_id: str,
|
|
483
|
+
count: int | None = None,
|
|
484
|
+
) -> list[str]:
|
|
485
|
+
"""Claim exact local-report rows, or all pre-created hosted rows."""
|
|
486
|
+
if count is not None and count <= 0:
|
|
487
|
+
return []
|
|
488
|
+
|
|
489
|
+
call_execution_ids: list[str] = []
|
|
490
|
+
for _ in range(64): # hard cap to prevent runaway
|
|
491
|
+
remaining = count - len(call_execution_ids) if count is not None else None
|
|
492
|
+
resp = client.post(
|
|
493
|
+
f"/simulate/api/alk-simulate/test-executions/{test_execution_id}/batch/",
|
|
494
|
+
json={"count": remaining} if remaining is not None else {},
|
|
495
|
+
)
|
|
496
|
+
resp.raise_for_status()
|
|
497
|
+
body = _unwrap(resp.json())
|
|
498
|
+
allocated = body["call_execution_ids"]
|
|
499
|
+
if remaining is not None and len(allocated) > remaining:
|
|
500
|
+
raise RuntimeError("backend allocated more call executions than requested")
|
|
501
|
+
call_execution_ids.extend(allocated)
|
|
502
|
+
if count is not None and len(call_execution_ids) == count:
|
|
503
|
+
break
|
|
504
|
+
if not allocated or not body.get("has_more"):
|
|
505
|
+
if count is None:
|
|
506
|
+
break
|
|
507
|
+
raise RuntimeError(
|
|
508
|
+
f"backend allocated {len(call_execution_ids)} of {count} required call executions"
|
|
509
|
+
)
|
|
510
|
+
|
|
511
|
+
if count is not None and len(call_execution_ids) != count:
|
|
512
|
+
raise RuntimeError(
|
|
513
|
+
f"backend allocated {len(call_execution_ids)} of {count} required call executions"
|
|
514
|
+
)
|
|
515
|
+
return call_execution_ids
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _submit_via_http(
|
|
519
|
+
*,
|
|
520
|
+
report: SimulationReport,
|
|
521
|
+
base_url: str,
|
|
522
|
+
api_key: str | None,
|
|
523
|
+
secret_key: str | None,
|
|
524
|
+
run_test_id: str,
|
|
525
|
+
internal_secret: str | None = None,
|
|
526
|
+
test_execution_id: str | None = None,
|
|
527
|
+
) -> dict[str, Any]:
|
|
528
|
+
with _open_client(base_url, api_key, secret_key, internal_secret) as client:
|
|
529
|
+
# Hosted runs submit into a TestExecution the platform pre-created; local
|
|
530
|
+
# runs create one here from the run test.
|
|
531
|
+
if not test_execution_id:
|
|
532
|
+
test_execution_id = _ensure_test_execution(client, run_test_id)
|
|
533
|
+
|
|
534
|
+
call_execution_ids = _allocate_call_ids(
|
|
535
|
+
client,
|
|
536
|
+
test_execution_id,
|
|
537
|
+
len(report.test_cases),
|
|
538
|
+
)
|
|
539
|
+
|
|
540
|
+
submitted_ids: list[str] = []
|
|
541
|
+
failed: list[dict[str, Any]] = []
|
|
542
|
+
for call_id, case in zip(call_execution_ids, report.test_cases):
|
|
543
|
+
payload = _build_result_payload(case)
|
|
544
|
+
recording_url = _maybe_upload_recording(client, call_id, case)
|
|
545
|
+
if recording_url:
|
|
546
|
+
payload["recording_url"] = recording_url
|
|
547
|
+
stereo_url = _maybe_upload_stereo_recording(client, call_id, case)
|
|
548
|
+
if stereo_url:
|
|
549
|
+
payload["stereo_recording_url"] = stereo_url
|
|
550
|
+
_attach_channel_recordings(client, call_id, case, payload)
|
|
551
|
+
_stamp_result_digest(payload)
|
|
552
|
+
resp = client.patch(
|
|
553
|
+
f"/simulate/api/alk-simulate/call-executions/{call_id}/result/",
|
|
554
|
+
json=payload,
|
|
555
|
+
)
|
|
556
|
+
if resp.is_error:
|
|
557
|
+
failed.append(
|
|
558
|
+
{
|
|
559
|
+
"call_execution_id": call_id,
|
|
560
|
+
"status_code": resp.status_code,
|
|
561
|
+
"body": _safe_body(resp),
|
|
562
|
+
}
|
|
563
|
+
)
|
|
564
|
+
else:
|
|
565
|
+
submitted_ids.append(call_id)
|
|
566
|
+
|
|
567
|
+
return {
|
|
568
|
+
"test_execution_id": test_execution_id,
|
|
569
|
+
"allocated_call_executions": call_execution_ids,
|
|
570
|
+
"submitted_call_executions": submitted_ids,
|
|
571
|
+
"failed_call_executions": failed,
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
def _unwrap(body: Any) -> dict[str, Any]:
|
|
576
|
+
if isinstance(body, dict) and "result" in body and isinstance(body["result"], dict):
|
|
577
|
+
return body["result"]
|
|
578
|
+
if isinstance(body, dict):
|
|
579
|
+
return body
|
|
580
|
+
raise ValueError(f"unexpected_response_shape: {body!r}")
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def _safe_body(response: httpx.Response) -> Any:
|
|
584
|
+
try:
|
|
585
|
+
return response.json()
|
|
586
|
+
except Exception:
|
|
587
|
+
return response.text[:500]
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def _build_result_payload(case) -> dict[str, Any]:
|
|
591
|
+
"""Map a SimulationTestCaseResult into the ALK ingestion PATCH body.
|
|
592
|
+
|
|
593
|
+
Backend derives conversation metrics and CSAT from the transcript, so
|
|
594
|
+
the SDK only ships what it directly observed.
|
|
595
|
+
"""
|
|
596
|
+
payload: dict[str, Any] = {
|
|
597
|
+
"status": _STATUS_MAP.get(case.status.value, "failed"),
|
|
598
|
+
}
|
|
599
|
+
|
|
600
|
+
started_at = case.started_at
|
|
601
|
+
ended_at = case.ended_at
|
|
602
|
+
# Case-level timestamps are unset for LiveKit runs (the engine does not
|
|
603
|
+
# stamp them). Fall back to the observed speech timing carried on each
|
|
604
|
+
# message so duration/start-time populate on the platform.
|
|
605
|
+
if started_at is None or ended_at is None:
|
|
606
|
+
speech_start, speech_end = _speech_bounds(case)
|
|
607
|
+
started_at = started_at or speech_start
|
|
608
|
+
ended_at = ended_at or speech_end
|
|
609
|
+
|
|
610
|
+
if started_at is not None:
|
|
611
|
+
payload["started_at"] = started_at.isoformat()
|
|
612
|
+
if ended_at is not None:
|
|
613
|
+
payload["ended_at"] = ended_at.isoformat()
|
|
614
|
+
if started_at is not None and ended_at is not None:
|
|
615
|
+
payload["duration_seconds"] = max(
|
|
616
|
+
int((ended_at - started_at).total_seconds()), 0
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
if case.failure is not None:
|
|
620
|
+
payload["ended_reason"] = case.failure.code
|
|
621
|
+
payload["error_message"] = case.failure.message or ""
|
|
622
|
+
|
|
623
|
+
result = case.result
|
|
624
|
+
transcript_segments: list[dict[str, Any]] = []
|
|
625
|
+
if result is not None:
|
|
626
|
+
transcript_segments = _extract_transcript_segments(result)
|
|
627
|
+
if transcript_segments:
|
|
628
|
+
payload["transcript"] = transcript_segments
|
|
629
|
+
if "ended_reason" not in payload:
|
|
630
|
+
stop_reason = result.metadata.get("stop_reason")
|
|
631
|
+
if isinstance(stop_reason, str) and stop_reason:
|
|
632
|
+
payload["ended_reason"] = stop_reason
|
|
633
|
+
|
|
634
|
+
recording_uri = _extract_recording_uri(result)
|
|
635
|
+
if recording_uri:
|
|
636
|
+
payload["recording_url"] = recording_uri
|
|
637
|
+
|
|
638
|
+
provider_call_data: dict[str, Any] = {}
|
|
639
|
+
existing_pcd = result.metadata.get("provider_call_data")
|
|
640
|
+
if isinstance(existing_pcd, dict):
|
|
641
|
+
provider_call_data = dict(existing_pcd)
|
|
642
|
+
|
|
643
|
+
# Fold the target agent's provider-reported usage/cost (captured by the
|
|
644
|
+
# SDK evidence layer — Vapi costBreakdown, Retell call_cost, LiveKit
|
|
645
|
+
# usage) into provider_call_data under the normalized ``usage.llm``
|
|
646
|
+
# shape the platform already reads for native voice. This is the
|
|
647
|
+
# agent-under-test's real usage — not the FutureAGI simulator's.
|
|
648
|
+
target = _target_provider_usage(case)
|
|
649
|
+
if target is not None:
|
|
650
|
+
provider_bucket = dict(provider_call_data.get(target.provider) or {})
|
|
651
|
+
if target.usage:
|
|
652
|
+
provider_bucket["usage"] = {
|
|
653
|
+
**(provider_bucket.get("usage") or {}),
|
|
654
|
+
"llm": target.usage,
|
|
655
|
+
}
|
|
656
|
+
if target.raw:
|
|
657
|
+
provider_bucket.setdefault("costBreakdown", target.raw)
|
|
658
|
+
if provider_bucket:
|
|
659
|
+
provider_call_data[target.provider] = provider_bucket
|
|
660
|
+
if target.cost_cents is not None:
|
|
661
|
+
payload["costs"] = {"cost_cents": target.cost_cents}
|
|
662
|
+
|
|
663
|
+
# Every LiveKit-engine run carries a truthy ``livekit`` marker so the
|
|
664
|
+
# platform's SpeakerRoleResolver detects the provider as LiveKit — its
|
|
665
|
+
# role map is direction-independent and already matches the SDK's
|
|
666
|
+
# tested-agent-perspective transcript. Without this a black-box target
|
|
667
|
+
# (no usage evidence) leaves ``provider_call_data`` empty, the platform
|
|
668
|
+
# falls back to VAPI, and an inbound default swaps agent/customer labels.
|
|
669
|
+
# A falsy ``{}`` is not enough — ``detect_provider`` treats it as absent.
|
|
670
|
+
if str(
|
|
671
|
+
result.metadata.get("engine")
|
|
672
|
+
) == "livekit" and not provider_call_data.get("livekit"):
|
|
673
|
+
provider_call_data["livekit"] = {"engine": "livekit"}
|
|
674
|
+
|
|
675
|
+
if provider_call_data:
|
|
676
|
+
payload["provider_call_data"] = provider_call_data
|
|
677
|
+
|
|
678
|
+
summary = result.metadata.get("call_summary") or result.metadata.get("summary")
|
|
679
|
+
if isinstance(summary, str) and summary:
|
|
680
|
+
payload["call_summary"] = summary
|
|
681
|
+
|
|
682
|
+
call_metadata = {
|
|
683
|
+
k: v
|
|
684
|
+
for k, v in result.metadata.items()
|
|
685
|
+
if k
|
|
686
|
+
not in {
|
|
687
|
+
"provider_call_data",
|
|
688
|
+
"call_summary",
|
|
689
|
+
"summary",
|
|
690
|
+
"failure",
|
|
691
|
+
"status",
|
|
692
|
+
"test_case_id",
|
|
693
|
+
"run_id",
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
if call_metadata:
|
|
697
|
+
payload["call_metadata"] = _json_safe(call_metadata)
|
|
698
|
+
|
|
699
|
+
return payload
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
def _stamp_result_digest(payload: dict[str, Any]) -> None:
|
|
703
|
+
"""Bind an idempotency key to exactly the canonical payload being submitted."""
|
|
704
|
+
payload.pop("result_digest", None)
|
|
705
|
+
payload["result_digest"] = (
|
|
706
|
+
"sha256:"
|
|
707
|
+
+ hashlib.sha256(
|
|
708
|
+
json.dumps(
|
|
709
|
+
payload,
|
|
710
|
+
sort_keys=True,
|
|
711
|
+
separators=(",", ":"),
|
|
712
|
+
default=str,
|
|
713
|
+
).encode()
|
|
714
|
+
).hexdigest()
|
|
715
|
+
)
|
|
716
|
+
|
|
717
|
+
|
|
718
|
+
def _extract_transcript_segments(result) -> list[dict[str, Any]]:
|
|
719
|
+
"""Convert TestCaseResult.messages into ALK transcript segments.
|
|
720
|
+
|
|
721
|
+
LiveKit engine emits each message with ``started_speaking_at`` and
|
|
722
|
+
``stopped_speaking_at`` (seconds since epoch, from ``ChatMessage.metrics``).
|
|
723
|
+
We convert to millisecond offsets relative to the first speech timestamp
|
|
724
|
+
so ``ConversationMetricsCalculator`` can compute overlap-based interruption
|
|
725
|
+
counts, WPM and talk-ratio on the backend.
|
|
726
|
+
"""
|
|
727
|
+
segments: list[dict[str, Any]] = []
|
|
728
|
+
typed_messages = [msg for msg in result.messages if isinstance(msg, dict)]
|
|
729
|
+
|
|
730
|
+
anchor = _first_speech_anchor(typed_messages)
|
|
731
|
+
for msg in typed_messages:
|
|
732
|
+
role = msg.get("role")
|
|
733
|
+
content = msg.get("content")
|
|
734
|
+
tool_calls = _normalize_tool_calls(msg.get("tool_calls"))
|
|
735
|
+
start_ms, end_ms = _resolve_message_timing_ms(msg, anchor)
|
|
736
|
+
latency_ms = _message_latency_ms(msg)
|
|
737
|
+
|
|
738
|
+
if role == "assistant":
|
|
739
|
+
# An assistant turn can carry text, tool calls, or both. Tool-call
|
|
740
|
+
# turns usually have empty content — emit them anyway as a
|
|
741
|
+
# ``tool_calls`` segment so the agent's real tool activity survives
|
|
742
|
+
# ingestion instead of being dropped by the empty-content guard.
|
|
743
|
+
if isinstance(content, str) and content:
|
|
744
|
+
segments.append(
|
|
745
|
+
_segment("assistant", content, start_ms, end_ms, latency_ms)
|
|
746
|
+
)
|
|
747
|
+
if tool_calls:
|
|
748
|
+
segments.append(
|
|
749
|
+
_segment(
|
|
750
|
+
"tool_calls",
|
|
751
|
+
_render_tool_calls(tool_calls),
|
|
752
|
+
start_ms,
|
|
753
|
+
end_ms,
|
|
754
|
+
latency_ms,
|
|
755
|
+
tool_calls=tool_calls,
|
|
756
|
+
)
|
|
757
|
+
)
|
|
758
|
+
continue
|
|
759
|
+
|
|
760
|
+
if role == "tool":
|
|
761
|
+
if isinstance(content, str) and content:
|
|
762
|
+
segments.append(
|
|
763
|
+
_segment(
|
|
764
|
+
"tool_call_result",
|
|
765
|
+
content,
|
|
766
|
+
start_ms,
|
|
767
|
+
end_ms,
|
|
768
|
+
None,
|
|
769
|
+
tool_call_id=msg.get("tool_call_id") or msg.get("id"),
|
|
770
|
+
)
|
|
771
|
+
)
|
|
772
|
+
continue
|
|
773
|
+
|
|
774
|
+
if not isinstance(content, str) or not content:
|
|
775
|
+
continue
|
|
776
|
+
if role in {"user", "customer"}:
|
|
777
|
+
speaker_role = "user"
|
|
778
|
+
elif role == "system":
|
|
779
|
+
speaker_role = "system"
|
|
780
|
+
else:
|
|
781
|
+
speaker_role = "unknown"
|
|
782
|
+
segments.append(_segment(speaker_role, content, start_ms, end_ms, None))
|
|
783
|
+
|
|
784
|
+
if segments:
|
|
785
|
+
return segments
|
|
786
|
+
|
|
787
|
+
if not result.transcript:
|
|
788
|
+
return []
|
|
789
|
+
for line in result.transcript.splitlines():
|
|
790
|
+
if ":" not in line:
|
|
791
|
+
continue
|
|
792
|
+
role_label, content = line.split(":", 1)
|
|
793
|
+
role_label = role_label.strip().lower()
|
|
794
|
+
content = content.strip()
|
|
795
|
+
if not content:
|
|
796
|
+
continue
|
|
797
|
+
if role_label in {"assistant", "agent", "bot"}:
|
|
798
|
+
speaker_role = "assistant"
|
|
799
|
+
elif role_label in {"customer", "user", "simulator", "caller"}:
|
|
800
|
+
speaker_role = "user"
|
|
801
|
+
else:
|
|
802
|
+
speaker_role = "unknown"
|
|
803
|
+
segments.append(
|
|
804
|
+
{
|
|
805
|
+
"speaker_role": speaker_role,
|
|
806
|
+
"content": content,
|
|
807
|
+
"start_time_ms": 0,
|
|
808
|
+
"end_time_ms": 0,
|
|
809
|
+
}
|
|
810
|
+
)
|
|
811
|
+
return segments
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _segment(
|
|
815
|
+
speaker_role: str,
|
|
816
|
+
content: str,
|
|
817
|
+
start_ms: int,
|
|
818
|
+
end_ms: int,
|
|
819
|
+
latency_ms: int | None,
|
|
820
|
+
*,
|
|
821
|
+
tool_calls: list[dict[str, Any]] | None = None,
|
|
822
|
+
tool_call_id: str | None = None,
|
|
823
|
+
) -> dict[str, Any]:
|
|
824
|
+
seg: dict[str, Any] = {
|
|
825
|
+
"speaker_role": speaker_role,
|
|
826
|
+
"content": content,
|
|
827
|
+
"start_time_ms": start_ms,
|
|
828
|
+
"end_time_ms": end_ms,
|
|
829
|
+
}
|
|
830
|
+
if latency_ms is not None:
|
|
831
|
+
seg["latency_ms"] = latency_ms
|
|
832
|
+
if tool_calls:
|
|
833
|
+
seg["tool_calls"] = tool_calls
|
|
834
|
+
if tool_call_id:
|
|
835
|
+
seg["tool_call_id"] = tool_call_id
|
|
836
|
+
return seg
|
|
837
|
+
|
|
838
|
+
|
|
839
|
+
def _normalize_tool_calls(raw: Any) -> list[dict[str, Any]] | None:
|
|
840
|
+
"""Coerce a message's tool_calls into a stable [{id, name, arguments}] shape.
|
|
841
|
+
|
|
842
|
+
Accepts both the flat SDK shape (``{"name", "arguments", "id"}``) and the
|
|
843
|
+
OpenAI/LiteLLM nested shape (``{"function": {"name", "arguments"}}``);
|
|
844
|
+
``arguments`` is JSON-decoded when the provider ships it as a string.
|
|
845
|
+
"""
|
|
846
|
+
if not raw or not isinstance(raw, (list, tuple)):
|
|
847
|
+
return None
|
|
848
|
+
calls: list[dict[str, Any]] = []
|
|
849
|
+
for tc in raw:
|
|
850
|
+
if not isinstance(tc, dict):
|
|
851
|
+
continue
|
|
852
|
+
fn = tc.get("function") if isinstance(tc.get("function"), dict) else {}
|
|
853
|
+
name = tc.get("name") or fn.get("name")
|
|
854
|
+
if not name:
|
|
855
|
+
continue
|
|
856
|
+
arguments = tc.get("arguments")
|
|
857
|
+
if arguments is None:
|
|
858
|
+
arguments = fn.get("arguments")
|
|
859
|
+
if isinstance(arguments, str):
|
|
860
|
+
try:
|
|
861
|
+
arguments = json.loads(arguments) if arguments.strip() else {}
|
|
862
|
+
except (ValueError, TypeError):
|
|
863
|
+
pass
|
|
864
|
+
calls.append(
|
|
865
|
+
{
|
|
866
|
+
"id": tc.get("id") or name,
|
|
867
|
+
"name": name,
|
|
868
|
+
"arguments": arguments if arguments is not None else {},
|
|
869
|
+
}
|
|
870
|
+
)
|
|
871
|
+
return calls or None
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def _render_tool_calls(calls: list[dict[str, Any]]) -> str:
|
|
875
|
+
lines: list[str] = []
|
|
876
|
+
for call in calls:
|
|
877
|
+
args = call.get("arguments")
|
|
878
|
+
try:
|
|
879
|
+
rendered = json.dumps(args, ensure_ascii=False, sort_keys=True)
|
|
880
|
+
except (TypeError, ValueError):
|
|
881
|
+
rendered = str(args)
|
|
882
|
+
lines.append(f"{call['name']}({rendered})")
|
|
883
|
+
return "\n".join(lines)
|
|
884
|
+
|
|
885
|
+
|
|
886
|
+
def _message_latency_ms(msg: dict[str, Any]) -> int | None:
|
|
887
|
+
for key in ("latency_ms", "latency"):
|
|
888
|
+
value = msg.get(key)
|
|
889
|
+
if isinstance(value, (int, float)) and value > 0:
|
|
890
|
+
return int(value)
|
|
891
|
+
metrics = msg.get("metrics")
|
|
892
|
+
if isinstance(metrics, dict):
|
|
893
|
+
for key in ("latency_ms", "latency"):
|
|
894
|
+
value = metrics.get(key)
|
|
895
|
+
if isinstance(value, (int, float)) and value > 0:
|
|
896
|
+
return int(value)
|
|
897
|
+
return None
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
def _maybe_upload_recording(
|
|
901
|
+
client: httpx.Client, call_execution_id: str, case
|
|
902
|
+
) -> str | None:
|
|
903
|
+
"""Upload the case's audio file (if any) via a multipart POST.
|
|
904
|
+
|
|
905
|
+
Prefers a combined/mixed WAV, falls back to output-only then input-only.
|
|
906
|
+
Skips silently when no on-disk audio exists (e.g. ``record_audio=False``
|
|
907
|
+
on the runner, or the SDK already surfaced an HTTPS URL via
|
|
908
|
+
``result.artifacts``). Returns the persisted ``recording_url`` to attach
|
|
909
|
+
to the ingestion PATCH, or None.
|
|
910
|
+
"""
|
|
911
|
+
if case.result is None:
|
|
912
|
+
return None
|
|
913
|
+
audio_path = _select_audio_path(case.result)
|
|
914
|
+
if audio_path is None:
|
|
915
|
+
return None
|
|
916
|
+
|
|
917
|
+
filename = audio_path.name
|
|
918
|
+
content_type = _CONTENT_TYPE_BY_EXT.get(
|
|
919
|
+
audio_path.suffix.lower(), "application/octet-stream"
|
|
920
|
+
)
|
|
921
|
+
with audio_path.open("rb") as fh:
|
|
922
|
+
files = {"file": (filename, fh, content_type)}
|
|
923
|
+
data = {
|
|
924
|
+
"filename": filename,
|
|
925
|
+
"sha256": _sha256_file(audio_path),
|
|
926
|
+
"kind": _recording_kind(case.result, audio_path),
|
|
927
|
+
}
|
|
928
|
+
resp = client.post(
|
|
929
|
+
f"/simulate/api/alk-simulate/call-executions/{call_execution_id}/recording/",
|
|
930
|
+
files=files,
|
|
931
|
+
data=data,
|
|
932
|
+
timeout=_RECORDING_UPLOAD_TIMEOUT_SECONDS,
|
|
933
|
+
)
|
|
934
|
+
if resp.is_error:
|
|
935
|
+
return None
|
|
936
|
+
body = _unwrap(resp.json())
|
|
937
|
+
return body.get("recording_url")
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
def _maybe_upload_stereo_recording(
|
|
941
|
+
client: httpx.Client, call_execution_id: str, case
|
|
942
|
+
) -> str | None:
|
|
943
|
+
"""Upload the case's 2-channel stereo WAV (ch0 customer, ch1 assistant).
|
|
944
|
+
|
|
945
|
+
Uses the same multipart endpoint as ``_maybe_upload_recording`` and returns
|
|
946
|
+
the persisted URL for ``stereo_recording_url``, or None when absent.
|
|
947
|
+
"""
|
|
948
|
+
if case.result is None:
|
|
949
|
+
return None
|
|
950
|
+
stereo_path = getattr(case.result, "audio_stereo_path", None)
|
|
951
|
+
if not stereo_path:
|
|
952
|
+
return None
|
|
953
|
+
path = Path(str(stereo_path)).expanduser()
|
|
954
|
+
if not (path.exists() and path.is_file() and path.stat().st_size > 0):
|
|
955
|
+
return None
|
|
956
|
+
|
|
957
|
+
filename = path.name
|
|
958
|
+
content_type = _CONTENT_TYPE_BY_EXT.get(
|
|
959
|
+
path.suffix.lower(), "application/octet-stream"
|
|
960
|
+
)
|
|
961
|
+
with path.open("rb") as fh:
|
|
962
|
+
files = {"file": (filename, fh, content_type)}
|
|
963
|
+
data = {
|
|
964
|
+
"filename": filename,
|
|
965
|
+
"sha256": _sha256_file(path),
|
|
966
|
+
"kind": "stereo",
|
|
967
|
+
}
|
|
968
|
+
resp = client.post(
|
|
969
|
+
f"/simulate/api/alk-simulate/call-executions/{call_execution_id}/recording/",
|
|
970
|
+
files=files,
|
|
971
|
+
data=data,
|
|
972
|
+
timeout=_RECORDING_UPLOAD_TIMEOUT_SECONDS,
|
|
973
|
+
)
|
|
974
|
+
if resp.is_error:
|
|
975
|
+
return None
|
|
976
|
+
body = _unwrap(resp.json())
|
|
977
|
+
return body.get("recording_url")
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
def _upload_audio_file(
|
|
981
|
+
client: httpx.Client, call_execution_id: str, path: Path, *, kind: str
|
|
982
|
+
) -> str | None:
|
|
983
|
+
"""POST a single on-disk WAV to the recording endpoint; return its URL."""
|
|
984
|
+
filename = path.name
|
|
985
|
+
content_type = _CONTENT_TYPE_BY_EXT.get(
|
|
986
|
+
path.suffix.lower(), "application/octet-stream"
|
|
987
|
+
)
|
|
988
|
+
with path.open("rb") as fh:
|
|
989
|
+
files = {"file": (filename, fh, content_type)}
|
|
990
|
+
data = {"filename": filename, "sha256": _sha256_file(path), "kind": kind}
|
|
991
|
+
resp = client.post(
|
|
992
|
+
f"/simulate/api/alk-simulate/call-executions/{call_execution_id}/recording/",
|
|
993
|
+
files=files,
|
|
994
|
+
data=data,
|
|
995
|
+
timeout=_RECORDING_UPLOAD_TIMEOUT_SECONDS,
|
|
996
|
+
)
|
|
997
|
+
if resp.is_error:
|
|
998
|
+
return None
|
|
999
|
+
return _unwrap(resp.json()).get("recording_url")
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def _sha256_file(path: Path) -> str:
|
|
1003
|
+
digest = hashlib.sha256()
|
|
1004
|
+
with path.open("rb") as stream:
|
|
1005
|
+
while chunk := stream.read(1024 * 1024):
|
|
1006
|
+
digest.update(chunk)
|
|
1007
|
+
return digest.hexdigest()
|
|
1008
|
+
|
|
1009
|
+
|
|
1010
|
+
def _upload_channel_recording(
|
|
1011
|
+
client: httpx.Client, call_execution_id: str, case, attr: str
|
|
1012
|
+
) -> str | None:
|
|
1013
|
+
"""Upload one per-speaker mono WAV named by ``attr`` on the case result."""
|
|
1014
|
+
if case.result is None:
|
|
1015
|
+
return None
|
|
1016
|
+
value = getattr(case.result, attr, None)
|
|
1017
|
+
if not value:
|
|
1018
|
+
return None
|
|
1019
|
+
path = Path(str(value)).expanduser()
|
|
1020
|
+
if not (path.exists() and path.is_file() and path.stat().st_size > 0):
|
|
1021
|
+
return None
|
|
1022
|
+
kind = "assistant" if attr == "audio_output_path" else "customer"
|
|
1023
|
+
return _upload_audio_file(client, call_execution_id, path, kind=kind)
|
|
1024
|
+
|
|
1025
|
+
|
|
1026
|
+
def _attach_channel_recordings(
|
|
1027
|
+
client: httpx.Client, call_execution_id: str, case, payload: dict[str, Any]
|
|
1028
|
+
) -> None:
|
|
1029
|
+
"""Upload the per-speaker assistant/customer mono tracks and fold their URLs
|
|
1030
|
+
into ``provider_call_data.livekit.recording`` so evals mapped to
|
|
1031
|
+
``call.assistant_recording`` / ``call.customer_recording`` resolve. LiveKit
|
|
1032
|
+
runs otherwise only produce combined + stereo, leaving the per-channel
|
|
1033
|
+
variables empty. ``audio_output_path`` is the target/assistant track,
|
|
1034
|
+
``audio_input_path`` is the simulator/customer track (matching the stereo
|
|
1035
|
+
channel order ch0 customer, ch1 assistant)."""
|
|
1036
|
+
assistant_url = _upload_channel_recording(
|
|
1037
|
+
client, call_execution_id, case, "audio_output_path"
|
|
1038
|
+
)
|
|
1039
|
+
customer_url = _upload_channel_recording(
|
|
1040
|
+
client, call_execution_id, case, "audio_input_path"
|
|
1041
|
+
)
|
|
1042
|
+
recording: dict[str, str] = {}
|
|
1043
|
+
if assistant_url:
|
|
1044
|
+
recording["assistant"] = assistant_url
|
|
1045
|
+
if customer_url:
|
|
1046
|
+
recording["customer"] = customer_url
|
|
1047
|
+
if not recording:
|
|
1048
|
+
return
|
|
1049
|
+
provider_call_data = payload.setdefault("provider_call_data", {})
|
|
1050
|
+
livekit = provider_call_data.setdefault("livekit", {})
|
|
1051
|
+
if not isinstance(livekit, dict):
|
|
1052
|
+
return
|
|
1053
|
+
livekit["recording"] = {**(livekit.get("recording") or {}), **recording}
|
|
1054
|
+
|
|
1055
|
+
|
|
1056
|
+
def _select_audio_path(result) -> Path | None:
|
|
1057
|
+
for candidate in (
|
|
1058
|
+
result.audio_combined_path,
|
|
1059
|
+
result.audio_output_path,
|
|
1060
|
+
result.audio_input_path,
|
|
1061
|
+
):
|
|
1062
|
+
if not candidate:
|
|
1063
|
+
continue
|
|
1064
|
+
path = Path(str(candidate)).expanduser()
|
|
1065
|
+
if path.exists() and path.is_file() and path.stat().st_size > 0:
|
|
1066
|
+
return path
|
|
1067
|
+
return None
|
|
1068
|
+
|
|
1069
|
+
|
|
1070
|
+
def _recording_kind(result, path: Path) -> str:
|
|
1071
|
+
if (
|
|
1072
|
+
result.audio_combined_path
|
|
1073
|
+
and path == Path(str(result.audio_combined_path)).expanduser()
|
|
1074
|
+
):
|
|
1075
|
+
return "combined"
|
|
1076
|
+
if (
|
|
1077
|
+
result.audio_output_path
|
|
1078
|
+
and path == Path(str(result.audio_output_path)).expanduser()
|
|
1079
|
+
):
|
|
1080
|
+
return "assistant"
|
|
1081
|
+
return "customer"
|
|
1082
|
+
|
|
1083
|
+
|
|
1084
|
+
_TARGET_PROVIDERS = ("vapi", "retell", "livekit")
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
class _TargetUsage:
|
|
1088
|
+
"""Normalized target-agent usage extracted from one provider's evidence."""
|
|
1089
|
+
|
|
1090
|
+
__slots__ = ("provider", "usage", "cost_cents", "raw")
|
|
1091
|
+
|
|
1092
|
+
def __init__(
|
|
1093
|
+
self,
|
|
1094
|
+
provider: str,
|
|
1095
|
+
usage: dict[str, int] | None,
|
|
1096
|
+
cost_cents: int | None,
|
|
1097
|
+
raw: dict[str, Any] | None,
|
|
1098
|
+
) -> None:
|
|
1099
|
+
self.provider = provider
|
|
1100
|
+
self.usage = usage
|
|
1101
|
+
self.cost_cents = cost_cents
|
|
1102
|
+
self.raw = raw
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _target_provider_usage(case) -> _TargetUsage | None:
|
|
1106
|
+
"""Pull the target agent's provider-reported usage from case evidence.
|
|
1107
|
+
|
|
1108
|
+
Provider-agnostic: dispatches to a per-provider extractor because each
|
|
1109
|
+
provider reports cost/tokens in a different shape (Vapi costBreakdown,
|
|
1110
|
+
Retell call_cost + llm_token_usage, LiveKit normalized usage). Returns a
|
|
1111
|
+
``_TargetUsage`` with a normalized ``usage`` (``prompt_tokens`` /
|
|
1112
|
+
``completion_tokens`` / ``total_tokens``) and ``cost_cents``, or None when
|
|
1113
|
+
no target evidence surfaced usage (e.g. a black-box self-hosted target).
|
|
1114
|
+
"""
|
|
1115
|
+
evidence = getattr(case, "evidence", None) or []
|
|
1116
|
+
for source in evidence:
|
|
1117
|
+
metadata = getattr(source, "metadata", None) or {}
|
|
1118
|
+
provider = metadata.get("provider")
|
|
1119
|
+
if provider not in _TARGET_PROVIDERS:
|
|
1120
|
+
continue
|
|
1121
|
+
extractor = _PROVIDER_USAGE_EXTRACTORS.get(provider)
|
|
1122
|
+
if extractor is None:
|
|
1123
|
+
continue
|
|
1124
|
+
result = extractor(metadata)
|
|
1125
|
+
if result is not None:
|
|
1126
|
+
return result
|
|
1127
|
+
return None
|
|
1128
|
+
|
|
1129
|
+
|
|
1130
|
+
def _vapi_usage(metadata: dict[str, Any]) -> _TargetUsage | None:
|
|
1131
|
+
cost = metadata.get("cost") if isinstance(metadata.get("cost"), dict) else {}
|
|
1132
|
+
breakdown = (
|
|
1133
|
+
cost.get("breakdown") if isinstance(cost.get("breakdown"), dict) else None
|
|
1134
|
+
)
|
|
1135
|
+
usage = None
|
|
1136
|
+
if breakdown:
|
|
1137
|
+
prompt = breakdown.get("llmPromptTokens", breakdown.get("promptTokens"))
|
|
1138
|
+
completion = breakdown.get(
|
|
1139
|
+
"llmCompletionTokens", breakdown.get("completionTokens")
|
|
1140
|
+
)
|
|
1141
|
+
usage = _normalized_usage(prompt, completion)
|
|
1142
|
+
cost_cents = _dollars_to_cents(cost.get("total"))
|
|
1143
|
+
if usage is None and cost_cents is None:
|
|
1144
|
+
return None
|
|
1145
|
+
return _TargetUsage("vapi", usage, cost_cents, breakdown)
|
|
1146
|
+
|
|
1147
|
+
|
|
1148
|
+
def _retell_usage(metadata: dict[str, Any]) -> _TargetUsage | None:
|
|
1149
|
+
token_usage = metadata.get("usage")
|
|
1150
|
+
usage = None
|
|
1151
|
+
if isinstance(token_usage, dict):
|
|
1152
|
+
# Retell may report prompt/completion directly, or per-request `values`
|
|
1153
|
+
# (total tokens only, no split).
|
|
1154
|
+
prompt = token_usage.get("num_input_tokens", token_usage.get("prompt_tokens"))
|
|
1155
|
+
completion = token_usage.get(
|
|
1156
|
+
"num_output_tokens", token_usage.get("completion_tokens")
|
|
1157
|
+
)
|
|
1158
|
+
if prompt is not None or completion is not None:
|
|
1159
|
+
usage = _normalized_usage(prompt, completion)
|
|
1160
|
+
else:
|
|
1161
|
+
values = token_usage.get("values")
|
|
1162
|
+
if isinstance(values, list) and values:
|
|
1163
|
+
total = sum(_coerce_int(v) for v in values)
|
|
1164
|
+
if total:
|
|
1165
|
+
usage = {"total_tokens": total}
|
|
1166
|
+
call_cost = metadata.get("cost") if isinstance(metadata.get("cost"), dict) else {}
|
|
1167
|
+
# Retell reports combined_cost already in cents.
|
|
1168
|
+
cost_cents = _coerce_int_or_none(call_cost.get("combined_cost"))
|
|
1169
|
+
if usage is None and cost_cents is None:
|
|
1170
|
+
return None
|
|
1171
|
+
return _TargetUsage("retell", usage, cost_cents, call_cost or None)
|
|
1172
|
+
|
|
1173
|
+
|
|
1174
|
+
def _livekit_usage(metadata: dict[str, Any]) -> _TargetUsage | None:
|
|
1175
|
+
# A LiveKit target that reports a normalized usage blob back through the
|
|
1176
|
+
# evidence layer (self-hosted worker). Absent for black-box targets.
|
|
1177
|
+
usage_blob = metadata.get("usage")
|
|
1178
|
+
if not isinstance(usage_blob, dict):
|
|
1179
|
+
return None
|
|
1180
|
+
llm = (
|
|
1181
|
+
usage_blob.get("llm") if isinstance(usage_blob.get("llm"), dict) else usage_blob
|
|
1182
|
+
)
|
|
1183
|
+
prompt = llm.get("prompt_tokens", llm.get("promptTokens"))
|
|
1184
|
+
completion = llm.get("completion_tokens", llm.get("completionTokens"))
|
|
1185
|
+
usage = _normalized_usage(prompt, completion)
|
|
1186
|
+
cost_cents = _dollars_to_cents(
|
|
1187
|
+
(metadata.get("cost") or {}).get("total")
|
|
1188
|
+
if isinstance(metadata.get("cost"), dict)
|
|
1189
|
+
else None
|
|
1190
|
+
)
|
|
1191
|
+
if usage is None and cost_cents is None:
|
|
1192
|
+
return None
|
|
1193
|
+
return _TargetUsage("livekit", usage, cost_cents, None)
|
|
1194
|
+
|
|
1195
|
+
|
|
1196
|
+
_PROVIDER_USAGE_EXTRACTORS = {
|
|
1197
|
+
"vapi": _vapi_usage,
|
|
1198
|
+
"retell": _retell_usage,
|
|
1199
|
+
"livekit": _livekit_usage,
|
|
1200
|
+
}
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
def _normalized_usage(prompt: Any, completion: Any) -> dict[str, int] | None:
|
|
1204
|
+
if prompt is None and completion is None:
|
|
1205
|
+
return None
|
|
1206
|
+
prompt_i = _coerce_int(prompt)
|
|
1207
|
+
completion_i = _coerce_int(completion)
|
|
1208
|
+
return {
|
|
1209
|
+
"prompt_tokens": prompt_i,
|
|
1210
|
+
"completion_tokens": completion_i,
|
|
1211
|
+
"total_tokens": prompt_i + completion_i,
|
|
1212
|
+
}
|
|
1213
|
+
|
|
1214
|
+
|
|
1215
|
+
def _dollars_to_cents(value: Any) -> int | None:
|
|
1216
|
+
dollars = _coerce_float(value)
|
|
1217
|
+
return int(round(dollars * 100)) if dollars is not None else None
|
|
1218
|
+
|
|
1219
|
+
|
|
1220
|
+
def _coerce_int(value: Any) -> int:
|
|
1221
|
+
try:
|
|
1222
|
+
return int(value or 0)
|
|
1223
|
+
except (TypeError, ValueError):
|
|
1224
|
+
return 0
|
|
1225
|
+
|
|
1226
|
+
|
|
1227
|
+
def _coerce_int_or_none(value: Any) -> int | None:
|
|
1228
|
+
if value is None:
|
|
1229
|
+
return None
|
|
1230
|
+
try:
|
|
1231
|
+
return int(round(float(value)))
|
|
1232
|
+
except (TypeError, ValueError):
|
|
1233
|
+
return None
|
|
1234
|
+
|
|
1235
|
+
|
|
1236
|
+
def _coerce_float(value: Any) -> float | None:
|
|
1237
|
+
try:
|
|
1238
|
+
return float(value) if value is not None else None
|
|
1239
|
+
except (TypeError, ValueError):
|
|
1240
|
+
return None
|
|
1241
|
+
|
|
1242
|
+
|
|
1243
|
+
def _speech_bounds(case) -> tuple[Any, Any]:
|
|
1244
|
+
"""Return (start, end) datetimes from a case's message speech timing.
|
|
1245
|
+
|
|
1246
|
+
LiveKit messages carry ``started_speaking_at`` / ``stopped_speaking_at``
|
|
1247
|
+
as epoch seconds; the earliest start and latest stop bound the actual
|
|
1248
|
+
conversation. Returns (None, None) when no timing is available.
|
|
1249
|
+
"""
|
|
1250
|
+
if case.result is None:
|
|
1251
|
+
return None, None
|
|
1252
|
+
starts: list[float] = []
|
|
1253
|
+
ends: list[float] = []
|
|
1254
|
+
for msg in case.result.messages:
|
|
1255
|
+
if not isinstance(msg, dict):
|
|
1256
|
+
continue
|
|
1257
|
+
start = msg.get("started_speaking_at") or msg.get("created_at")
|
|
1258
|
+
stop = msg.get("stopped_speaking_at") or msg.get("created_at")
|
|
1259
|
+
if isinstance(start, (int, float)) and start > 0:
|
|
1260
|
+
starts.append(float(start))
|
|
1261
|
+
if isinstance(stop, (int, float)) and stop > 0:
|
|
1262
|
+
ends.append(float(stop))
|
|
1263
|
+
if not starts or not ends:
|
|
1264
|
+
return None, None
|
|
1265
|
+
start_dt = datetime.fromtimestamp(min(starts), tz=timezone.utc)
|
|
1266
|
+
end_dt = datetime.fromtimestamp(max(ends), tz=timezone.utc)
|
|
1267
|
+
if end_dt < start_dt:
|
|
1268
|
+
end_dt = start_dt
|
|
1269
|
+
return start_dt, end_dt
|
|
1270
|
+
|
|
1271
|
+
|
|
1272
|
+
def _first_speech_anchor(messages: list[dict[str, Any]]) -> float | None:
|
|
1273
|
+
for msg in messages:
|
|
1274
|
+
for key in ("started_speaking_at", "created_at"):
|
|
1275
|
+
value = msg.get(key)
|
|
1276
|
+
if isinstance(value, (int, float)) and value > 0:
|
|
1277
|
+
return float(value)
|
|
1278
|
+
return None
|
|
1279
|
+
|
|
1280
|
+
|
|
1281
|
+
def _resolve_message_timing_ms(
|
|
1282
|
+
msg: dict[str, Any], anchor: float | None
|
|
1283
|
+
) -> tuple[int, int]:
|
|
1284
|
+
"""Return (start_ms, end_ms) relative to the first-speech anchor.
|
|
1285
|
+
|
|
1286
|
+
Falls back to ``created_at`` when speech-timing metrics are missing (text
|
|
1287
|
+
turns, providers that don't report the metric). Zero is used as the last
|
|
1288
|
+
resort — the backend metrics calculator degrades gracefully when timings
|
|
1289
|
+
collapse to zero-duration.
|
|
1290
|
+
"""
|
|
1291
|
+
if anchor is None:
|
|
1292
|
+
return 0, 0
|
|
1293
|
+
|
|
1294
|
+
start_raw = msg.get("started_speaking_at") or msg.get("created_at") or 0.0
|
|
1295
|
+
stop_raw = (
|
|
1296
|
+
msg.get("stopped_speaking_at") or msg.get("created_at") or start_raw or 0.0
|
|
1297
|
+
)
|
|
1298
|
+
start_ms = (
|
|
1299
|
+
max(int(round((float(start_raw) - anchor) * 1000)), 0) if start_raw else 0
|
|
1300
|
+
)
|
|
1301
|
+
end_ms = (
|
|
1302
|
+
max(int(round((float(stop_raw) - anchor) * 1000)), start_ms)
|
|
1303
|
+
if stop_raw
|
|
1304
|
+
else start_ms
|
|
1305
|
+
)
|
|
1306
|
+
return start_ms, end_ms
|
|
1307
|
+
|
|
1308
|
+
|
|
1309
|
+
def _extract_recording_uri(result) -> str | None:
|
|
1310
|
+
for artifact in result.artifacts:
|
|
1311
|
+
artifact_type = getattr(artifact, "type", None)
|
|
1312
|
+
if artifact_type == "audio" and getattr(artifact, "uri", None):
|
|
1313
|
+
return artifact.uri
|
|
1314
|
+
for candidate in (
|
|
1315
|
+
result.audio_combined_path,
|
|
1316
|
+
result.audio_output_path,
|
|
1317
|
+
result.audio_input_path,
|
|
1318
|
+
):
|
|
1319
|
+
if candidate and str(candidate).startswith(("http://", "https://")):
|
|
1320
|
+
return str(candidate)
|
|
1321
|
+
return None
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
def _json_safe(value: Any) -> Any:
|
|
1325
|
+
try:
|
|
1326
|
+
json.dumps(value)
|
|
1327
|
+
return value
|
|
1328
|
+
except (TypeError, ValueError):
|
|
1329
|
+
return json.loads(json.dumps(value, default=str))
|
|
1330
|
+
|
|
1331
|
+
|
|
1332
|
+
def _write_submission(run_directory: Path, payload: dict[str, Any]) -> None:
|
|
1333
|
+
submission_path = run_directory / "submission.json"
|
|
1334
|
+
submission_path.write_text(
|
|
1335
|
+
json.dumps(payload, indent=2, sort_keys=True, default=str) + "\n",
|
|
1336
|
+
encoding="utf-8",
|
|
1337
|
+
)
|
|
1338
|
+
|
|
1339
|
+
|
|
1340
|
+
__all__ = ["FutureAGIResultSink"]
|