agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,1209 @@
|
|
|
1
|
+
"""The one thing the harness hands a suite to.
|
|
2
|
+
|
|
3
|
+
The harness does not run scenarios. It builds a world, writes scenarios against it, and calls
|
|
4
|
+
`simulate` once. Everything after that belongs to ALK: how many run at a time, whether the person
|
|
5
|
+
is typed to or phoned, where the audio goes, what a report looks like.
|
|
6
|
+
|
|
7
|
+
That split matters more than it looks. While the harness ran scenarios itself, one at a time,
|
|
8
|
+
through its own conversation loop, a suite was only as good as the harness's patience: a run took
|
|
9
|
+
as many turns of the chat as it had scenarios, and the simulator driving it was not the one the
|
|
10
|
+
product ships. Handing over means the suite runs the same way whether a person triggered it from
|
|
11
|
+
the UI, a script did, or nobody did.
|
|
12
|
+
|
|
13
|
+
Chat and voice are one path here. A contract-only chat spec may run as an in-process target; a
|
|
14
|
+
repository-backed chat agent runs its submitted service and is reached through its declared HTTP
|
|
15
|
+
or WebSocket ingress; a voice agent is reached through its declared realtime transport. All three
|
|
16
|
+
receive the same isolated world, setup, checks and report. Only the target adapter differs, and a
|
|
17
|
+
repository-backed agent is never reconstructed from its extracted prompt.
|
|
18
|
+
|
|
19
|
+
A run is a folder. One simulation over a suite is one run, kept whole, so a session accumulates
|
|
20
|
+
runs that can be compared rather than one result file that the next run overwrites.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import asyncio
|
|
26
|
+
import inspect
|
|
27
|
+
import json
|
|
28
|
+
import logging
|
|
29
|
+
import re
|
|
30
|
+
import time
|
|
31
|
+
from collections.abc import Callable
|
|
32
|
+
from dataclasses import asdict
|
|
33
|
+
from datetime import UTC, datetime
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import TYPE_CHECKING, Any
|
|
36
|
+
|
|
37
|
+
from ..contract import AgentContract
|
|
38
|
+
from ..scenario import Scenario
|
|
39
|
+
from ..world.runtime import Call
|
|
40
|
+
from .grade import Judgement, Result
|
|
41
|
+
|
|
42
|
+
if TYPE_CHECKING:
|
|
43
|
+
from .conversation import Exchange
|
|
44
|
+
|
|
45
|
+
RUNS = "runs"
|
|
46
|
+
RUN = "run.json"
|
|
47
|
+
RESULT = "result.json"
|
|
48
|
+
TRANSCRIPT = "transcript.txt"
|
|
49
|
+
CALLS = "calls.json"
|
|
50
|
+
logger = logging.getLogger(__name__)
|
|
51
|
+
|
|
52
|
+
# How many scenarios run at once by default. One, because the shipped default should be the one
|
|
53
|
+
# that cannot surprise anybody: a voice suite places real calls that cost real money, and fanning
|
|
54
|
+
# out to twenty is a bad thing to learn from a bill.
|
|
55
|
+
CONCURRENCY = 1
|
|
56
|
+
|
|
57
|
+
# What ALK calls the world and the person, per modality. Both are registry names it validates
|
|
58
|
+
# against the plugin's own manifest, so a typo is an error here rather than a confusing run.
|
|
59
|
+
WORLDS = {"text": ("chat", "chat"), "voice": ("voice", "voice")}
|
|
60
|
+
SIMULATORS = {"text": "synthetic_user", "voice": "livekit_simulator"}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def spoken_to(contract: AgentContract) -> bool:
|
|
64
|
+
"""Whether this agent is spoken to rather than typed to."""
|
|
65
|
+
return (contract.modality or "text").strip().lower() == "voice"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def new_run_id() -> str:
|
|
69
|
+
return datetime.now(UTC).strftime("run-%Y%m%d-%H%M%S")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def run_root(destination: Path, run_id: str) -> Path:
|
|
73
|
+
return Path(destination) / RUNS / run_id
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def every_run(destination: Path) -> list[dict[str, Any]]:
|
|
77
|
+
"""Every run in this session, newest first, finished or not.
|
|
78
|
+
|
|
79
|
+
A run that is still going is reported too, from the results already written. `run.json` is
|
|
80
|
+
written once, at the end, so requiring it meant an hour-long suite showed nothing at all
|
|
81
|
+
while its results sat on disk: the scenario that finished forty minutes ago was as invisible
|
|
82
|
+
as the one that had not started. `finished` says which kind each is.
|
|
83
|
+
"""
|
|
84
|
+
root = Path(destination) / RUNS
|
|
85
|
+
if not root.exists():
|
|
86
|
+
return []
|
|
87
|
+
found: list[dict[str, Any]] = []
|
|
88
|
+
for folder in sorted(root.iterdir(), reverse=True):
|
|
89
|
+
if not folder.is_dir():
|
|
90
|
+
continue
|
|
91
|
+
kept = folder / RUN
|
|
92
|
+
if kept.exists():
|
|
93
|
+
try:
|
|
94
|
+
summary = json.loads(kept.read_text(encoding="utf-8"))
|
|
95
|
+
except Exception: # noqa: BLE001 - one unreadable run never hides the rest
|
|
96
|
+
continue
|
|
97
|
+
summary["finished"] = True
|
|
98
|
+
found.append(summary)
|
|
99
|
+
continue
|
|
100
|
+
done = _cases_so_far(folder)
|
|
101
|
+
if done:
|
|
102
|
+
found.append(
|
|
103
|
+
{
|
|
104
|
+
"run_id": folder.name,
|
|
105
|
+
"finished": False,
|
|
106
|
+
"scenarios": len(done),
|
|
107
|
+
"passed": sum(1 for one in done if one.get("passed")),
|
|
108
|
+
"seconds": round(sum(one.get("seconds") or 0 for one in done), 1),
|
|
109
|
+
"results": done,
|
|
110
|
+
}
|
|
111
|
+
)
|
|
112
|
+
return found
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _cases_so_far(folder: Path) -> list[dict[str, Any]]:
|
|
116
|
+
"""The scenarios of an unfinished run that have already been written."""
|
|
117
|
+
done: list[dict[str, Any]] = []
|
|
118
|
+
for case in sorted(folder.iterdir()):
|
|
119
|
+
kept = case / RESULT
|
|
120
|
+
if not case.is_dir() or not kept.exists():
|
|
121
|
+
continue
|
|
122
|
+
try:
|
|
123
|
+
one = json.loads(kept.read_text(encoding="utf-8"))
|
|
124
|
+
except Exception: # noqa: BLE001 - a result being written this instant is not an error
|
|
125
|
+
continue
|
|
126
|
+
done.append(
|
|
127
|
+
{
|
|
128
|
+
"scenario": one.get("scenario", case.name),
|
|
129
|
+
"passed": bool(one.get("passed")),
|
|
130
|
+
"met": one.get("met"),
|
|
131
|
+
"of": len(one.get("checkpoints") or []),
|
|
132
|
+
"seconds": one.get("seconds"),
|
|
133
|
+
"recording": one.get("recording", ""),
|
|
134
|
+
"problems": one.get("problems") or [],
|
|
135
|
+
}
|
|
136
|
+
)
|
|
137
|
+
return done
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def read_run(destination: Path, run_id: str) -> dict[str, Any]:
|
|
141
|
+
"""One run in full: its summary, and every scenario's result, transcript and calls.
|
|
142
|
+
|
|
143
|
+
Read from the folder rather than held in memory, so the harness can be asked about a run
|
|
144
|
+
that happened before it was started, and about any single call inside one.
|
|
145
|
+
"""
|
|
146
|
+
root = run_root(destination, run_id)
|
|
147
|
+
kept = root / RUN
|
|
148
|
+
if not root.exists():
|
|
149
|
+
raise FileNotFoundError(f"no run {run_id} in {destination}")
|
|
150
|
+
# A run still going has no summary yet, but the scenarios it has finished are readable and
|
|
151
|
+
# worth reading. Only a folder that is not there at all is an error.
|
|
152
|
+
summary = (
|
|
153
|
+
json.loads(kept.read_text(encoding="utf-8"))
|
|
154
|
+
if kept.exists()
|
|
155
|
+
else {"run_id": run_id, "finished": False, "passed": 0}
|
|
156
|
+
)
|
|
157
|
+
summary.setdefault("finished", kept.exists())
|
|
158
|
+
scenarios: list[dict[str, Any]] = []
|
|
159
|
+
for folder in sorted(root.iterdir()):
|
|
160
|
+
if not folder.is_dir() or not (folder / RESULT).exists():
|
|
161
|
+
continue
|
|
162
|
+
one = json.loads((folder / RESULT).read_text(encoding="utf-8"))
|
|
163
|
+
one["transcript"] = _text(folder / TRANSCRIPT)
|
|
164
|
+
one["calls_detail"] = _json(folder / CALLS)
|
|
165
|
+
scenarios.append(one)
|
|
166
|
+
summary["scenarios"] = scenarios
|
|
167
|
+
return summary
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _text(path: Path) -> str:
|
|
171
|
+
return path.read_text(encoding="utf-8") if path.exists() else ""
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _json(path: Path) -> list[dict[str, Any]]:
|
|
175
|
+
return json.loads(path.read_text(encoding="utf-8")) if path.exists() else []
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
async def simulate(
|
|
179
|
+
scenarios: list[Scenario],
|
|
180
|
+
contract: AgentContract,
|
|
181
|
+
world_root: Path,
|
|
182
|
+
*,
|
|
183
|
+
destination: Path | None = None,
|
|
184
|
+
model: str | None = None,
|
|
185
|
+
concurrency: int = CONCURRENCY,
|
|
186
|
+
run_id: str = "",
|
|
187
|
+
on_case_start: Callable[[Scenario], Any] | None = None,
|
|
188
|
+
on_case_done: Callable[[Result], Any] | None = None,
|
|
189
|
+
on_exchange: Callable[[str, dict[str, Any]], Any] | None = None,
|
|
190
|
+
) -> dict[str, Any]:
|
|
191
|
+
"""Run a whole suite through ALK and write it out as one run.
|
|
192
|
+
|
|
193
|
+
Returns the run's summary. Results are in the order they were asked for, not the order they
|
|
194
|
+
finished, so a report reads the same however it was scheduled.
|
|
195
|
+
"""
|
|
196
|
+
from .models import for_roles
|
|
197
|
+
|
|
198
|
+
destination = Path(destination or world_root)
|
|
199
|
+
if (Path(world_root) / "environment.json").exists() and concurrency != 1:
|
|
200
|
+
# Restored source worlds point at the submitted Compose project's one real datastore.
|
|
201
|
+
# Until a provisioner can clone that entire project per case, parallel cases would reset
|
|
202
|
+
# and mutate the same database underneath each other. Serialize explicitly rather than
|
|
203
|
+
# offering fast but invalid isolation.
|
|
204
|
+
logger.warning(
|
|
205
|
+
"source-provisioned scenarios share one isolated Compose project; forcing "
|
|
206
|
+
"concurrency from %s to 1",
|
|
207
|
+
concurrency,
|
|
208
|
+
)
|
|
209
|
+
concurrency = 1
|
|
210
|
+
run_id = run_id or new_run_id()
|
|
211
|
+
root = run_root(destination, run_id)
|
|
212
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
213
|
+
roles = for_roles(model)
|
|
214
|
+
|
|
215
|
+
# Durations must not jump when the host clock is corrected (common on laptops/VMs). Keep the
|
|
216
|
+
# human timestamp separately and measure elapsed time with the monotonic clock.
|
|
217
|
+
started_at = datetime.now(UTC).isoformat(timespec="seconds")
|
|
218
|
+
started = time.monotonic()
|
|
219
|
+
room = asyncio.Semaphore(max(1, concurrency))
|
|
220
|
+
ordered: list[Result | None] = [None] * len(scenarios)
|
|
221
|
+
|
|
222
|
+
async def one(index: int, scenario: Scenario) -> None:
|
|
223
|
+
async with room:
|
|
224
|
+
if on_case_start:
|
|
225
|
+
notified = on_case_start(scenario)
|
|
226
|
+
if inspect.isawaitable(notified):
|
|
227
|
+
await notified
|
|
228
|
+
began = time.monotonic()
|
|
229
|
+
folder = root / scenario.name
|
|
230
|
+
folder.mkdir(parents=True, exist_ok=True)
|
|
231
|
+
try:
|
|
232
|
+
result = await _run_one(
|
|
233
|
+
scenario,
|
|
234
|
+
contract,
|
|
235
|
+
world_root,
|
|
236
|
+
folder,
|
|
237
|
+
roles=roles,
|
|
238
|
+
on_exchange=(
|
|
239
|
+
(lambda turn: on_exchange(scenario.name, turn))
|
|
240
|
+
if on_exchange
|
|
241
|
+
else None
|
|
242
|
+
),
|
|
243
|
+
)
|
|
244
|
+
except Exception as failed: # noqa: BLE001 - one bad scenario never stops the suite
|
|
245
|
+
result = Result(
|
|
246
|
+
scenario=scenario.name,
|
|
247
|
+
tests=scenario.tests,
|
|
248
|
+
problems=[f"{type(failed).__name__}: {failed}"],
|
|
249
|
+
# This is a terminal outcome for the attempted scenario, but it is not an
|
|
250
|
+
# agent result. Keeping an explicit ending prevents downstream artifact
|
|
251
|
+
# readers from confusing an exception-shaped partial record with a call
|
|
252
|
+
# that is still in progress.
|
|
253
|
+
ended="failed",
|
|
254
|
+
)
|
|
255
|
+
result.seconds = round(time.monotonic() - began, 1)
|
|
256
|
+
_write_case(folder, result)
|
|
257
|
+
ordered[index] = result
|
|
258
|
+
if on_case_done:
|
|
259
|
+
notified = on_case_done(result)
|
|
260
|
+
if inspect.isawaitable(notified):
|
|
261
|
+
await notified
|
|
262
|
+
|
|
263
|
+
await asyncio.gather(
|
|
264
|
+
*(one(index, scenario) for index, scenario in enumerate(scenarios))
|
|
265
|
+
)
|
|
266
|
+
results = [one for one in ordered if one is not None]
|
|
267
|
+
|
|
268
|
+
summary = {
|
|
269
|
+
"run_id": run_id,
|
|
270
|
+
"agent": contract.agent,
|
|
271
|
+
"modality": contract.modality or "text",
|
|
272
|
+
"started": started_at,
|
|
273
|
+
"seconds": round(time.monotonic() - started, 1),
|
|
274
|
+
"concurrency": concurrency,
|
|
275
|
+
"models": roles,
|
|
276
|
+
"scenarios": len(results),
|
|
277
|
+
"passed": sum(1 for one in results if one.passed),
|
|
278
|
+
# A scenario that could not be executed is an infrastructure/harness outcome, not a
|
|
279
|
+
# weak-agent grade. The CLI and hosted worker use this count to keep those two result
|
|
280
|
+
# classes distinct all the way to the platform.
|
|
281
|
+
"unrunnable": sum(1 for one in results if one.problems),
|
|
282
|
+
"spent_usd": round(sum(one.spent_usd for one in results), 4),
|
|
283
|
+
# Averaged across the scenarios that reported them, so a suite has one line per metric
|
|
284
|
+
# rather than a number nobody compares. Only over the runs that actually measured it:
|
|
285
|
+
# averaging a missing metric as zero would make a suite look worse the more of it failed
|
|
286
|
+
# to run, which is the opposite of informative.
|
|
287
|
+
"metrics": _averaged([one.measured for one in results]),
|
|
288
|
+
"results": [
|
|
289
|
+
{
|
|
290
|
+
"scenario": one.scenario,
|
|
291
|
+
"passed": one.passed,
|
|
292
|
+
"met": one.met,
|
|
293
|
+
"of": len(one.checkpoints),
|
|
294
|
+
"seconds": one.seconds,
|
|
295
|
+
"recording": one.recording,
|
|
296
|
+
"problems": one.problems,
|
|
297
|
+
}
|
|
298
|
+
for one in results
|
|
299
|
+
],
|
|
300
|
+
}
|
|
301
|
+
(root / RUN).write_text(
|
|
302
|
+
json.dumps(summary, indent=2, default=str), encoding="utf-8"
|
|
303
|
+
)
|
|
304
|
+
return summary
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _write_case(folder: Path, result: Result) -> None:
|
|
308
|
+
"""One scenario's result, transcript and calls, each in the form it is read in.
|
|
309
|
+
|
|
310
|
+
The transcript is written as text because it is read by people, and the calls as JSON
|
|
311
|
+
because they are read by the UI and by the harness looking into a single call.
|
|
312
|
+
"""
|
|
313
|
+
body = asdict(result)
|
|
314
|
+
body["passed"] = result.passed
|
|
315
|
+
body["met"] = result.met
|
|
316
|
+
detail = body.pop("calls_detail", None) or []
|
|
317
|
+
(folder / RESULT).write_text(
|
|
318
|
+
json.dumps(body, indent=2, default=str), encoding="utf-8"
|
|
319
|
+
)
|
|
320
|
+
(folder / TRANSCRIPT).write_text(result.transcript or "", encoding="utf-8")
|
|
321
|
+
(folder / CALLS).write_text(
|
|
322
|
+
json.dumps(detail, indent=2, default=str), encoding="utf-8"
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
async def _run_one(
|
|
327
|
+
scenario: Scenario,
|
|
328
|
+
contract: AgentContract,
|
|
329
|
+
world_root: Path,
|
|
330
|
+
folder: Path,
|
|
331
|
+
*,
|
|
332
|
+
roles: dict[str, str],
|
|
333
|
+
on_exchange: Callable[[dict[str, Any]], Any] | None = None,
|
|
334
|
+
) -> Result:
|
|
335
|
+
"""One scenario, in its own world, through ALK's runner.
|
|
336
|
+
|
|
337
|
+
The world is prepared here and handed in, rather than named in the spec, because isolation
|
|
338
|
+
is ours to guarantee: every scenario starts from the same frozen base with only its own
|
|
339
|
+
setup applied, and a world shared between cases would let the first one decide what the
|
|
340
|
+
second is graded against.
|
|
341
|
+
"""
|
|
342
|
+
|
|
343
|
+
from ..folder import apply_setup, check_ready
|
|
344
|
+
from ..world.snapshot import restore
|
|
345
|
+
|
|
346
|
+
spoken = spoken_to(contract)
|
|
347
|
+
kind = "voice" if spoken else "text"
|
|
348
|
+
adapter, world_kind = WORLDS[kind]
|
|
349
|
+
|
|
350
|
+
world = restore(world_root)
|
|
351
|
+
try:
|
|
352
|
+
world.reset()
|
|
353
|
+
applied = apply_setup(scenario, world)
|
|
354
|
+
if not applied.ok:
|
|
355
|
+
raise RuntimeError(f"the scenario's setup did not run: {applied.said}")
|
|
356
|
+
ready = check_ready(scenario, world)
|
|
357
|
+
if not ready.ok:
|
|
358
|
+
raise RuntimeError(
|
|
359
|
+
f"the world is not ready for this scenario: {ready.said}. Running it would "
|
|
360
|
+
"test us rather than the agent."
|
|
361
|
+
)
|
|
362
|
+
# The setup's own calls are not the agent's.
|
|
363
|
+
world.calls = []
|
|
364
|
+
|
|
365
|
+
if not spoken:
|
|
366
|
+
# Typed, and driven by a model rather than by ALK's chat simulator.
|
|
367
|
+
#
|
|
368
|
+
# That simulator is deterministic on purpose: an untyped persona gets three fixed
|
|
369
|
+
# lines ("Can you give me the exact next step…"), and a typed one renders utterances
|
|
370
|
+
# from a compiled behaviour policy. Reproducible, and not a simulation of a person.
|
|
371
|
+
# A suite whose user says the same three things to every agent tests one path and
|
|
372
|
+
# calls it coverage.
|
|
373
|
+
#
|
|
374
|
+
# So the conversation is driven here, by a model reading the simulator prompt the
|
|
375
|
+
# build stage wrote for this agent. Everything around it is unchanged: same world,
|
|
376
|
+
# same setup, same checks, same run folder.
|
|
377
|
+
return await _typed_to(
|
|
378
|
+
scenario, contract, world, world_root, folder, roles=roles
|
|
379
|
+
)
|
|
380
|
+
|
|
381
|
+
# Spoken. The agent is not here: it runs in Vapi, with its own prompt, its own model
|
|
382
|
+
# and its own voice, and the only thing that changes is where its tools are answered.
|
|
383
|
+
# ALK places the call and drives a simulated caller that is a real model over STT and
|
|
384
|
+
# TTS, so this half was never deterministic.
|
|
385
|
+
return await _spoken_to(
|
|
386
|
+
scenario,
|
|
387
|
+
contract,
|
|
388
|
+
world,
|
|
389
|
+
world_root,
|
|
390
|
+
folder,
|
|
391
|
+
roles=roles,
|
|
392
|
+
on_exchange=on_exchange,
|
|
393
|
+
)
|
|
394
|
+
finally:
|
|
395
|
+
try:
|
|
396
|
+
world.close()
|
|
397
|
+
except Exception: # cleanup must never replace a completed scenario result
|
|
398
|
+
logger.exception("world cleanup failed after scenario %s", scenario.name)
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _found_audio(directory: Path) -> Path | None:
|
|
402
|
+
"""The recording a run left behind, if it left one.
|
|
403
|
+
|
|
404
|
+
Asked of the directory rather than taken on trust from whatever placed the call: a runner
|
|
405
|
+
that exits badly still returns a path, and a path is not a file.
|
|
406
|
+
"""
|
|
407
|
+
if not directory.exists():
|
|
408
|
+
return None
|
|
409
|
+
for path in sorted(directory.rglob("*")):
|
|
410
|
+
if path.is_file() and path.suffix.lower() in (".wav", ".mp3", ".ogg", ".m4a"):
|
|
411
|
+
return path
|
|
412
|
+
return None
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
async def _typed_to(
|
|
416
|
+
scenario: Scenario,
|
|
417
|
+
contract: AgentContract,
|
|
418
|
+
world: Any,
|
|
419
|
+
world_root: Path,
|
|
420
|
+
folder: Path,
|
|
421
|
+
*,
|
|
422
|
+
roles: dict[str, str],
|
|
423
|
+
) -> Result:
|
|
424
|
+
"""A typed conversation, with a model on both sides.
|
|
425
|
+
|
|
426
|
+
The same grading as every other run: the world it is handed is already set up, and what it
|
|
427
|
+
leaves behind is what the checks read.
|
|
428
|
+
"""
|
|
429
|
+
from ..catalogue import load_catalogue
|
|
430
|
+
from . import converse
|
|
431
|
+
from .grade import (
|
|
432
|
+
checkpoints,
|
|
433
|
+
grade_sub_goals,
|
|
434
|
+
judge,
|
|
435
|
+
judge_suite_evals,
|
|
436
|
+
reconcile_task_completion,
|
|
437
|
+
ungraded_sub_goals,
|
|
438
|
+
)
|
|
439
|
+
from .targets import resolve
|
|
440
|
+
|
|
441
|
+
repository_backed = bool(
|
|
442
|
+
contract.runtime or contract.tool_entrypoints or contract.implementation
|
|
443
|
+
)
|
|
444
|
+
if repository_backed:
|
|
445
|
+
agent = resolve("repository")(
|
|
446
|
+
contract,
|
|
447
|
+
world,
|
|
448
|
+
world_root=world_root,
|
|
449
|
+
trace_path=folder / "agent-tool-calls.jsonl",
|
|
450
|
+
scenario_name=scenario.name,
|
|
451
|
+
)
|
|
452
|
+
else:
|
|
453
|
+
agent = resolve("local")(contract, world, model=roles["agent"])
|
|
454
|
+
transcript = await converse(
|
|
455
|
+
agent, scenario, contract, world_root=world_root, model=roles["user"]
|
|
456
|
+
)
|
|
457
|
+
catalogue = load_catalogue(world_root)
|
|
458
|
+
settled = grade_sub_goals(world, scenario, catalogue, transcript.calls)
|
|
459
|
+
ending = ", ".join(
|
|
460
|
+
f"{name}: {len(rows)} rows"
|
|
461
|
+
for name, rows in sorted(world.observe().state.items())
|
|
462
|
+
)
|
|
463
|
+
judgements, judged_cost = await judge(
|
|
464
|
+
scenario, transcript, contract, catalogue, model=roles["judge"], ending=ending
|
|
465
|
+
)
|
|
466
|
+
suite_judgements = judge_suite_evals(
|
|
467
|
+
catalogue.suite_evals, scenario, transcript, contract, ending=ending
|
|
468
|
+
)
|
|
469
|
+
judgements += reconcile_task_completion(suite_judgements, settled, judgements)
|
|
470
|
+
result = Result(
|
|
471
|
+
scenario=scenario.name,
|
|
472
|
+
tests=scenario.tests,
|
|
473
|
+
problems=[
|
|
474
|
+
f"{name} is not in this catalogue, so nothing graded it"
|
|
475
|
+
for name in ungraded_sub_goals(scenario, catalogue)
|
|
476
|
+
],
|
|
477
|
+
state_failures=[f"{one.name}: {one.said}" for one in settled if not one.held],
|
|
478
|
+
conduct=judgements,
|
|
479
|
+
checkpoints=checkpoints(settled, judgements),
|
|
480
|
+
crashes=[f"{call.name}: {call.error}" for call in transcript.crashed()],
|
|
481
|
+
ended=transcript.ended,
|
|
482
|
+
turns=len(transcript.exchanges),
|
|
483
|
+
calls=len(transcript.calls),
|
|
484
|
+
spent_usd=transcript.spent_usd + judged_cost,
|
|
485
|
+
transcript=transcript.spoken(),
|
|
486
|
+
exchanges=[
|
|
487
|
+
{"speaker": turn.speaker, "text": turn.text}
|
|
488
|
+
for turn in transcript.exchanges
|
|
489
|
+
],
|
|
490
|
+
actions=transcript.actions(),
|
|
491
|
+
)
|
|
492
|
+
result.calls_detail = _calls_of(transcript.calls)
|
|
493
|
+
return result
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def _calls_of(calls: Any) -> list[dict[str, Any]]:
|
|
497
|
+
"""Every call in full, for the timeline and for anyone asking what one call did."""
|
|
498
|
+
return [
|
|
499
|
+
{
|
|
500
|
+
"name": call.name,
|
|
501
|
+
"arguments": call.arguments,
|
|
502
|
+
"result": str(call.result)[:2000],
|
|
503
|
+
"ok": call.ok,
|
|
504
|
+
"refused": call.refused,
|
|
505
|
+
"error": call.error,
|
|
506
|
+
"at": getattr(call, "at", 0.0),
|
|
507
|
+
}
|
|
508
|
+
for call in calls
|
|
509
|
+
]
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _said(line: str) -> Exchange:
|
|
513
|
+
"""One transcript line as a turn, with its role read off rather than left in the text.
|
|
514
|
+
|
|
515
|
+
The line arrives already labelled ("assistant: ..."). Keeping that label in the text made the
|
|
516
|
+
judge read ``agent: assistant: ...``, two speakers deep for every turn.
|
|
517
|
+
"""
|
|
518
|
+
from .conversation import Exchange
|
|
519
|
+
|
|
520
|
+
role, _, text = line.partition(":")
|
|
521
|
+
named = role.strip().lower()
|
|
522
|
+
if named in ("assistant", "agent"):
|
|
523
|
+
return Exchange("agent", text.strip())
|
|
524
|
+
if named in ("user", "customer"):
|
|
525
|
+
return Exchange("customer", text.strip())
|
|
526
|
+
return Exchange("customer", line.strip())
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
async def _spoken_to(
|
|
530
|
+
scenario: Scenario,
|
|
531
|
+
contract: AgentContract,
|
|
532
|
+
world: Any,
|
|
533
|
+
world_root: Path,
|
|
534
|
+
folder: Path,
|
|
535
|
+
*,
|
|
536
|
+
roles: dict[str, str],
|
|
537
|
+
on_exchange: Callable[[dict[str, Any]], Any] | None = None,
|
|
538
|
+
) -> Result:
|
|
539
|
+
"""A real call, with the agent's own tools answered by this world.
|
|
540
|
+
|
|
541
|
+
The agent under test is not reconstructed here and is not running in this process. It is the
|
|
542
|
+
hosted assistant, with its own prompt, model and voice; the only thing that changes for the
|
|
543
|
+
duration is where its tool calls are sent. That makes this the more faithful of the two
|
|
544
|
+
paths, and the reason a spoken suite is worth more than a typed one.
|
|
545
|
+
|
|
546
|
+
The call itself belongs to ALK, which drives a simulated caller through speech: a real model
|
|
547
|
+
behind STT and TTS, not a script.
|
|
548
|
+
"""
|
|
549
|
+
import os
|
|
550
|
+
import time
|
|
551
|
+
|
|
552
|
+
from ..catalogue import load_catalogue
|
|
553
|
+
from .call import place_the_call
|
|
554
|
+
from .conversation import Transcript
|
|
555
|
+
from .evidence import measured, newest_report, spoken_times, tracks_in
|
|
556
|
+
from .grade import (
|
|
557
|
+
checkpoints,
|
|
558
|
+
grade_sub_goals,
|
|
559
|
+
judge,
|
|
560
|
+
judge_suite_evals,
|
|
561
|
+
reconcile_task_completion,
|
|
562
|
+
ungraded_sub_goals,
|
|
563
|
+
)
|
|
564
|
+
from .live import wire
|
|
565
|
+
from .tools import configure_source_voice, missing_prerequisites
|
|
566
|
+
|
|
567
|
+
configure_source_voice(world_root, contract)
|
|
568
|
+
stopping = missing_prerequisites(world_root, contract)
|
|
569
|
+
if stopping:
|
|
570
|
+
raise RuntimeError("cannot place a call:\n - " + "\n - ".join(stopping))
|
|
571
|
+
|
|
572
|
+
loop = asyncio.get_running_loop()
|
|
573
|
+
|
|
574
|
+
def live_exchange(turn: dict[str, Any]) -> None:
|
|
575
|
+
if on_exchange:
|
|
576
|
+
normalized = _normalize_live_exchange(turn)
|
|
577
|
+
loop.call_soon_threadsafe(on_exchange, normalized)
|
|
578
|
+
|
|
579
|
+
def placed_once() -> tuple[int, dict[str, Any], str]:
|
|
580
|
+
"""Everything about the call, off the event loop.
|
|
581
|
+
|
|
582
|
+
Wiring reads a subprocess's stdout and the call itself blocks for minutes. Run inline
|
|
583
|
+
they freeze whatever loop is hosting this, which for the UI means the stream, the status
|
|
584
|
+
endpoint and the stop button all stop with it.
|
|
585
|
+
"""
|
|
586
|
+
_world, instruction, webhook, tunnel, _url, _moved = wire(
|
|
587
|
+
scenario,
|
|
588
|
+
world_root,
|
|
589
|
+
world=world,
|
|
590
|
+
trace_path=folder / "agent-tool-calls.jsonl",
|
|
591
|
+
)
|
|
592
|
+
started = time.time()
|
|
593
|
+
runtime_output = ""
|
|
594
|
+
try:
|
|
595
|
+
sdk_output = folder / "sdk"
|
|
596
|
+
os.environ["HARNESS_VOICE_OUTPUT_ROOT"] = str(sdk_output.resolve())
|
|
597
|
+
os.environ["HARNESS_INSTRUCTION"] = instruction
|
|
598
|
+
os.environ["HARNESS_SCENARIO"] = scenario.name
|
|
599
|
+
# The caller is never handed the grader's pass question: `tests` is written about
|
|
600
|
+
# the agent in the third person, so as an objective it reads as a rubric rather
|
|
601
|
+
# than a motive. What this person wants is already in the instruction.
|
|
602
|
+
os.environ.pop("HARNESS_OUTCOME", None)
|
|
603
|
+
os.environ["HARNESS_PERSONA"] = json.dumps(
|
|
604
|
+
scenario.persona.model_dump(exclude_none=True)
|
|
605
|
+
if scenario.persona is not None
|
|
606
|
+
else {"name": "customer"}
|
|
607
|
+
)
|
|
608
|
+
os.environ["HARNESS_INITIAL_MESSAGE"] = (
|
|
609
|
+
scenario.persona.initial_message if scenario.persona is not None else ""
|
|
610
|
+
)
|
|
611
|
+
os.environ["HARNESS_SCRIPTED_CALLER"] = json.dumps(
|
|
612
|
+
scenario.persona.scripted_caller
|
|
613
|
+
if scenario.persona is not None
|
|
614
|
+
and scenario.persona.scripted_caller is not None
|
|
615
|
+
else {}
|
|
616
|
+
)
|
|
617
|
+
os.environ["HARNESS_FIXTURE"] = json.dumps(
|
|
618
|
+
scenario.fixture, ensure_ascii=False, default=str
|
|
619
|
+
)
|
|
620
|
+
# A scenario that asks to be heard through background noise selects a clip for the
|
|
621
|
+
# caller's environment; the voice engine mixes it under the caller. Cleared otherwise so
|
|
622
|
+
# a previous call's noise never leaks into a quiet one.
|
|
623
|
+
from ..background_noise import scenario_source
|
|
624
|
+
|
|
625
|
+
source = scenario_source(
|
|
626
|
+
getattr(scenario, "background_noise", False),
|
|
627
|
+
scenario.fixture,
|
|
628
|
+
seed=scenario.name,
|
|
629
|
+
)
|
|
630
|
+
if source:
|
|
631
|
+
os.environ["HARNESS_BACKGROUND_NOISE"] = source
|
|
632
|
+
else:
|
|
633
|
+
os.environ.pop("HARNESS_BACKGROUND_NOISE", None)
|
|
634
|
+
code = place_the_call(
|
|
635
|
+
os.environ.get("HARNESS_VOICE_CASE", "2.1.2"),
|
|
636
|
+
on_exchange=live_exchange if on_exchange else None,
|
|
637
|
+
)
|
|
638
|
+
finally:
|
|
639
|
+
try:
|
|
640
|
+
webhook.stop()
|
|
641
|
+
except Exception:
|
|
642
|
+
logger.exception(
|
|
643
|
+
"webhook cleanup failed after scenario %s", scenario.name
|
|
644
|
+
)
|
|
645
|
+
if tunnel is not None:
|
|
646
|
+
try:
|
|
647
|
+
tunnel.terminate()
|
|
648
|
+
except Exception:
|
|
649
|
+
logger.exception(
|
|
650
|
+
"tunnel cleanup failed after scenario %s", scenario.name
|
|
651
|
+
)
|
|
652
|
+
if (Path(world_root) / "environment.json").exists():
|
|
653
|
+
try:
|
|
654
|
+
from ..provision import runtime_logs, stop_runtime
|
|
655
|
+
|
|
656
|
+
# Some third-party LiveKit agents execute every tool in-process. They do not
|
|
657
|
+
# call the harness webhook and may not implement HARNESS_AGENT_TOOL_TRACE,
|
|
658
|
+
# but the LiveKit worker emits structured execution lifecycle events. Read
|
|
659
|
+
# those events before removing the per-scenario container. Raw logs are not
|
|
660
|
+
# retained; only normalized tool evidence is kept below.
|
|
661
|
+
runtime_output = runtime_logs(world_root)
|
|
662
|
+
stop_runtime(world_root)
|
|
663
|
+
except Exception:
|
|
664
|
+
logger.exception(
|
|
665
|
+
"runtime cleanup failed after scenario %s", scenario.name
|
|
666
|
+
)
|
|
667
|
+
# Everything the runner recorded about this call, read from the report it wrote.
|
|
668
|
+
return code, newest_report(started, root=sdk_output), runtime_output
|
|
669
|
+
|
|
670
|
+
attempts = 1 + max(0, int(os.environ.get("HARNESS_VOICE_INFRA_RETRIES", "1")))
|
|
671
|
+
code, case, runtime_output = 1, {}, ""
|
|
672
|
+
attempts_used = 0
|
|
673
|
+
trace_path = folder / "agent-tool-calls.jsonl"
|
|
674
|
+
for attempt in range(attempts):
|
|
675
|
+
attempts_used = attempt + 1
|
|
676
|
+
# Every attempt owns its trace. A stale line from a failed attempt must not turn a later
|
|
677
|
+
# worker-join failure into something that looks like agent evidence.
|
|
678
|
+
trace_path.unlink(missing_ok=True)
|
|
679
|
+
# The webhook is the transport-level evidence fallback. Clear calls from a failed voice
|
|
680
|
+
# attempt before retrying so only the attempt whose transcript is graded can contribute.
|
|
681
|
+
world.calls = []
|
|
682
|
+
code, case, runtime_output = await asyncio.to_thread(placed_once)
|
|
683
|
+
attempt_calls = _semantic_calls(
|
|
684
|
+
trace_path, contract=contract
|
|
685
|
+
) or _livekit_log_calls(runtime_output, contract=contract)
|
|
686
|
+
if (
|
|
687
|
+
not _voice_attempt_should_retry(
|
|
688
|
+
code, case, has_agent_calls=bool(attempt_calls or world.calls)
|
|
689
|
+
)
|
|
690
|
+
or attempt + 1 >= attempts
|
|
691
|
+
):
|
|
692
|
+
break
|
|
693
|
+
logger.warning(
|
|
694
|
+
"retryable voice attempt ended early for %s; retrying attempt %s/%s",
|
|
695
|
+
scenario.name,
|
|
696
|
+
attempt + 2,
|
|
697
|
+
attempts,
|
|
698
|
+
)
|
|
699
|
+
semantic = _semantic_calls(trace_path, contract=contract) or _livekit_log_calls(
|
|
700
|
+
runtime_output, contract=contract
|
|
701
|
+
)
|
|
702
|
+
# A worker trace includes semantic/local actions that never cross HTTP and is preferred when
|
|
703
|
+
# available. In a hosted Docker runner, however, the job artifacts can live in a named volume
|
|
704
|
+
# whose container path cannot be bind-mounted by the host daemon into the submitted runtime.
|
|
705
|
+
# In that case the bound world is still exact evidence: setup calls were cleared in prepare,
|
|
706
|
+
# caller hydration uses record=False, and every remaining call arrived through this call's
|
|
707
|
+
# private webhook. Do not erase that evidence merely because the optional trace is absent.
|
|
708
|
+
if semantic:
|
|
709
|
+
world.calls = semantic
|
|
710
|
+
spoken = str(case.get("transcript") or "")
|
|
711
|
+
# Every track that exists, copied in beside the result so a run is self-contained and the
|
|
712
|
+
# page can fall back when the preferred one is missing.
|
|
713
|
+
kept = _keep_tracks(tracks_in(case), folder)
|
|
714
|
+
|
|
715
|
+
if _voice_infrastructure_failure(code, case, has_agent_calls=bool(semantic)):
|
|
716
|
+
status = str((case.get("metadata") or {}).get("status") or "failed")
|
|
717
|
+
result = Result(
|
|
718
|
+
scenario=scenario.name,
|
|
719
|
+
tests=scenario.tests,
|
|
720
|
+
problems=[
|
|
721
|
+
"voice infrastructure failed after retry: the target worker never joined or "
|
|
722
|
+
"produced an assistant turn; this is not graded as an agent failure"
|
|
723
|
+
],
|
|
724
|
+
ended=status,
|
|
725
|
+
turns=len([line for line in spoken.splitlines() if line.strip()]),
|
|
726
|
+
calls=0,
|
|
727
|
+
transcript=spoken,
|
|
728
|
+
recording=(kept[0]["path"] if kept else ""),
|
|
729
|
+
)
|
|
730
|
+
result.tracks = kept
|
|
731
|
+
result.measured = measured(case)
|
|
732
|
+
result.measured["voice_attempts"] = attempts_used
|
|
733
|
+
return result
|
|
734
|
+
|
|
735
|
+
catalogue = load_catalogue(world_root)
|
|
736
|
+
settled = grade_sub_goals(world, scenario, catalogue, world.calls)
|
|
737
|
+
# Judged the same way a typed run is. Without this a spoken scenario reports "1/2" when what
|
|
738
|
+
# happened is that one check passed and the other was never asked, which reads as the agent
|
|
739
|
+
# half-failing rather than as the suite not having looked.
|
|
740
|
+
spoken_transcript = Transcript(
|
|
741
|
+
exchanges=[_said(line) for line in spoken.splitlines() if line.strip()],
|
|
742
|
+
calls=list(world.calls),
|
|
743
|
+
ended=str((case.get("metadata") or {}).get("status") or "finished"),
|
|
744
|
+
)
|
|
745
|
+
judgements, judged_cost = await judge(
|
|
746
|
+
scenario,
|
|
747
|
+
spoken_transcript,
|
|
748
|
+
contract,
|
|
749
|
+
catalogue,
|
|
750
|
+
model=roles["judge"],
|
|
751
|
+
ending=", ".join(
|
|
752
|
+
f"{name}: {len(rows)} rows"
|
|
753
|
+
for name, rows in sorted(world.observe().state.items())
|
|
754
|
+
),
|
|
755
|
+
)
|
|
756
|
+
_require_action_evidence(judgements, scenario, world.calls)
|
|
757
|
+
suite_judgements = judge_suite_evals(
|
|
758
|
+
catalogue.suite_evals,
|
|
759
|
+
scenario,
|
|
760
|
+
spoken_transcript,
|
|
761
|
+
contract,
|
|
762
|
+
ending=", ".join(
|
|
763
|
+
f"{name}: {len(rows)} rows"
|
|
764
|
+
for name, rows in sorted(world.observe().state.items())
|
|
765
|
+
),
|
|
766
|
+
)
|
|
767
|
+
judgements += reconcile_task_completion(suite_judgements, settled, judgements)
|
|
768
|
+
result = Result(
|
|
769
|
+
scenario=scenario.name,
|
|
770
|
+
tests=scenario.tests,
|
|
771
|
+
problems=[
|
|
772
|
+
f"{name} is not in this catalogue, so nothing graded it"
|
|
773
|
+
for name in ungraded_sub_goals(scenario, catalogue)
|
|
774
|
+
],
|
|
775
|
+
state_failures=[f"{one.name}: {one.said}" for one in settled if not one.held],
|
|
776
|
+
conduct=judgements,
|
|
777
|
+
checkpoints=checkpoints(settled, judgements),
|
|
778
|
+
spent_usd=judged_cost,
|
|
779
|
+
ended=str((case.get("metadata") or {}).get("status") or "finished"),
|
|
780
|
+
turns=len([line for line in spoken.splitlines() if line.strip()]),
|
|
781
|
+
calls=len(world.calls),
|
|
782
|
+
transcript=spoken,
|
|
783
|
+
exchanges=_timed_exchanges(spoken_transcript.exchanges, spoken_times(case)),
|
|
784
|
+
recording=(kept[0]["path"] if kept else ""),
|
|
785
|
+
)
|
|
786
|
+
result.tracks = kept
|
|
787
|
+
result.measured = measured(case)
|
|
788
|
+
result.measured["voice_attempts"] = attempts_used
|
|
789
|
+
result.calls_detail = _calls_of(world.calls)
|
|
790
|
+
return result
|
|
791
|
+
|
|
792
|
+
|
|
793
|
+
def _normalize_live_exchange(turn: dict[str, Any]) -> dict[str, Any]:
|
|
794
|
+
"""Translate ALK's report roles to the harness UI's conversation roles.
|
|
795
|
+
|
|
796
|
+
The callback comes from the simulator's own AgentSession, so its roles are from that
|
|
797
|
+
session's point of view: ``assistant`` is the simulated customer and ``user`` is the tested
|
|
798
|
+
service agent. The completed report later translates them to the test's point of view.
|
|
799
|
+
"""
|
|
800
|
+
raw = str(turn.get("speaker") or turn.get("role") or "").strip().lower()
|
|
801
|
+
speaker = (
|
|
802
|
+
"customer"
|
|
803
|
+
if raw in {"assistant", "customer"}
|
|
804
|
+
else "agent"
|
|
805
|
+
if raw in {"user", "agent"}
|
|
806
|
+
else raw or "customer"
|
|
807
|
+
)
|
|
808
|
+
return {
|
|
809
|
+
**turn,
|
|
810
|
+
"speaker": speaker,
|
|
811
|
+
"text": turn.get("text") or turn.get("content") or "",
|
|
812
|
+
}
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def _require_action_evidence(
|
|
816
|
+
judgements: list[Judgement], scenario: Scenario, calls: list[Call]
|
|
817
|
+
) -> None:
|
|
818
|
+
"""Prevent conditional prose checks from passing vacuously when the agent did nothing.
|
|
819
|
+
|
|
820
|
+
A judge can reasonably say "surge was disclosed before confirmation" when neither event
|
|
821
|
+
occurred because the implication is technically vacuous. In an executable scenario with a
|
|
822
|
+
reference action sequence, no semantic calls means the scenario conduct was not satisfied.
|
|
823
|
+
"""
|
|
824
|
+
if not scenario.solution or calls:
|
|
825
|
+
return
|
|
826
|
+
for judgement in judgements:
|
|
827
|
+
if judgement.holds:
|
|
828
|
+
judgement.holds = False
|
|
829
|
+
judgement.why = (
|
|
830
|
+
"The agent made no semantic tool calls, so this scenario conduct cannot be "
|
|
831
|
+
"credited even if its ordering condition is vacuously true."
|
|
832
|
+
)
|
|
833
|
+
|
|
834
|
+
|
|
835
|
+
def _voice_infrastructure_failure(
|
|
836
|
+
code: int, case: dict[str, Any], *, has_agent_calls: bool = False
|
|
837
|
+
) -> bool:
|
|
838
|
+
"""A failed room before meaningful target activity is not agent evidence."""
|
|
839
|
+
if code == 0:
|
|
840
|
+
return False
|
|
841
|
+
transcript = str(case.get("transcript") or "")
|
|
842
|
+
lines = [line for line in transcript.splitlines() if line.strip()]
|
|
843
|
+
has_target_turn = any(
|
|
844
|
+
line.strip().lower().startswith(("assistant:", "agent:")) for line in lines
|
|
845
|
+
)
|
|
846
|
+
if not has_target_turn:
|
|
847
|
+
return True
|
|
848
|
+
failure = (case.get("metadata") or {}).get("failure") or case.get("failure") or {}
|
|
849
|
+
retryable = bool(failure.get("retryable")) if isinstance(failure, dict) else False
|
|
850
|
+
failure_code = str(failure.get("code") or "") if isinstance(failure, dict) else ""
|
|
851
|
+
transport_retryable = retryable and failure_code in {
|
|
852
|
+
"target_disconnected",
|
|
853
|
+
"target_not_found",
|
|
854
|
+
"provider_disconnected",
|
|
855
|
+
"room_connection_failed",
|
|
856
|
+
"room_not_ready",
|
|
857
|
+
}
|
|
858
|
+
target_lines = [
|
|
859
|
+
line.split(":", 1)[-1].strip()
|
|
860
|
+
for line in lines
|
|
861
|
+
if line.strip().lower().startswith(("assistant:", "agent:"))
|
|
862
|
+
]
|
|
863
|
+
truncated_target = any(
|
|
864
|
+
utterance and utterance[-1] not in ".?!" for utterance in target_lines
|
|
865
|
+
)
|
|
866
|
+
# ALK requires six alternating messages by default. A retryable disconnect before that,
|
|
867
|
+
# without one semantic action, is a room/worker lifecycle failure; a longer conversation or
|
|
868
|
+
# any tool trace is enough evidence to grade the agent normally.
|
|
869
|
+
return (
|
|
870
|
+
not has_agent_calls
|
|
871
|
+
and len(lines) < 6
|
|
872
|
+
and (transport_retryable or truncated_target)
|
|
873
|
+
)
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
def _voice_attempt_should_retry(
|
|
877
|
+
code: int, case: dict[str, Any], *, has_agent_calls: bool = False
|
|
878
|
+
) -> bool:
|
|
879
|
+
"""Retry one short silence without misclassifying the final result as infrastructure.
|
|
880
|
+
|
|
881
|
+
Live speech recognition occasionally drops the caller's first audio turn. The provider
|
|
882
|
+
labels that timeout retryable, but a greeting proves the worker joined, so if the retry also
|
|
883
|
+
fails it remains an agent-pipeline reliability failure. A bounded second attempt separates
|
|
884
|
+
a transient dropped turn from a reproducible weak branch while preserving both outcomes via
|
|
885
|
+
``voice_attempts``.
|
|
886
|
+
"""
|
|
887
|
+
if _voice_infrastructure_failure(code, case, has_agent_calls=has_agent_calls):
|
|
888
|
+
return True
|
|
889
|
+
if code == 0 or has_agent_calls:
|
|
890
|
+
return False
|
|
891
|
+
failure = (case.get("metadata") or {}).get("failure") or case.get("failure") or {}
|
|
892
|
+
if not isinstance(failure, dict):
|
|
893
|
+
return False
|
|
894
|
+
lines = [
|
|
895
|
+
line for line in str(case.get("transcript") or "").splitlines() if line.strip()
|
|
896
|
+
]
|
|
897
|
+
return (
|
|
898
|
+
bool(failure.get("retryable"))
|
|
899
|
+
and str(failure.get("code") or "") == "conversation_silence_timeout"
|
|
900
|
+
and len(lines) < 6
|
|
901
|
+
)
|
|
902
|
+
|
|
903
|
+
|
|
904
|
+
def _semantic_calls(path: Path, *, contract: AgentContract | None = None) -> list[Call]:
|
|
905
|
+
"""Read the submitted worker's agent-facing tool trace, tolerating a killed final line."""
|
|
906
|
+
if not path.exists():
|
|
907
|
+
return []
|
|
908
|
+
endpoint_names = {
|
|
909
|
+
entry.endpoint.strip("/"): entry.tool
|
|
910
|
+
for entry in (contract.tool_entrypoints if contract is not None else [])
|
|
911
|
+
if entry.endpoint.strip("/")
|
|
912
|
+
}
|
|
913
|
+
calls: list[Any] = []
|
|
914
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
915
|
+
try:
|
|
916
|
+
record = json.loads(line)
|
|
917
|
+
except (json.JSONDecodeError, TypeError):
|
|
918
|
+
continue
|
|
919
|
+
if not isinstance(record, dict) or not record.get("name"):
|
|
920
|
+
continue
|
|
921
|
+
output = record.get("output")
|
|
922
|
+
if isinstance(output, str):
|
|
923
|
+
try:
|
|
924
|
+
output = json.loads(output)
|
|
925
|
+
except json.JSONDecodeError:
|
|
926
|
+
pass
|
|
927
|
+
arguments = record.get("arguments") or {}
|
|
928
|
+
if isinstance(arguments, str):
|
|
929
|
+
try:
|
|
930
|
+
arguments = json.loads(arguments)
|
|
931
|
+
except json.JSONDecodeError:
|
|
932
|
+
arguments = {"raw": arguments}
|
|
933
|
+
if not isinstance(arguments, dict):
|
|
934
|
+
arguments = {"value": arguments}
|
|
935
|
+
failed = bool(record.get("is_error"))
|
|
936
|
+
recorded_name = str(record["name"]).strip("/")
|
|
937
|
+
call = Call(
|
|
938
|
+
name=endpoint_names.get(recorded_name, recorded_name),
|
|
939
|
+
arguments=arguments,
|
|
940
|
+
result=output,
|
|
941
|
+
ok=not failed,
|
|
942
|
+
refused=failed,
|
|
943
|
+
error=str(output) if failed else "",
|
|
944
|
+
at=float(record.get("at") or 0.0),
|
|
945
|
+
)
|
|
946
|
+
# A harness-aware worker may mirror a local state-machine action to the world for
|
|
947
|
+
# observability and then emit the authoritative function-completion event. They are one
|
|
948
|
+
# logical action. Prefer the completion result, but never collapse ordinary identical
|
|
949
|
+
# retries (payment-status polling is a legitimate example).
|
|
950
|
+
if calls and _telemetry_mirror(calls[-1], call):
|
|
951
|
+
calls[-1] = call
|
|
952
|
+
else:
|
|
953
|
+
calls.append(call)
|
|
954
|
+
return calls
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
_ANSI = re.compile(r"\x1b\[[0-?]*[ -/]*[@-~]")
|
|
958
|
+
|
|
959
|
+
|
|
960
|
+
def _livekit_log_calls(
|
|
961
|
+
output: str, *, contract: AgentContract | None = None
|
|
962
|
+
) -> list[Call]:
|
|
963
|
+
"""Normalize completed LiveKit Python tool executions from bounded runtime logs.
|
|
964
|
+
|
|
965
|
+
This fallback is intentionally narrow: a start event alone earns no evidence, and arbitrary
|
|
966
|
+
application prose is never interpreted as a call. LiveKit's structured ``executing tool``
|
|
967
|
+
and matching ``tools execution completed`` records are stable SDK lifecycle events. The
|
|
968
|
+
Successful result values are not present in those logs, so they are represented honestly as
|
|
969
|
+
completion evidence rather than fabricated output. Structured ``ToolError while executing
|
|
970
|
+
tool`` records do carry a refusal reason; correlate those with the matching start so a
|
|
971
|
+
truthful refusal is never normalized as success.
|
|
972
|
+
"""
|
|
973
|
+
endpoint_names = {
|
|
974
|
+
entry.endpoint.strip("/"): entry.tool
|
|
975
|
+
for entry in (contract.tool_entrypoints if contract is not None else [])
|
|
976
|
+
if entry.endpoint.strip("/")
|
|
977
|
+
}
|
|
978
|
+
contract_tools = (
|
|
979
|
+
{tool.name: tool for tool in contract.tools} if contract is not None else {}
|
|
980
|
+
)
|
|
981
|
+
decoder = json.JSONDecoder()
|
|
982
|
+
starts: list[dict[str, Any]] = []
|
|
983
|
+
completed: set[str] = set()
|
|
984
|
+
refusals: dict[tuple[str, str], str] = {}
|
|
985
|
+
for raw_line in str(output or "").splitlines():
|
|
986
|
+
line = _ANSI.sub("", raw_line)
|
|
987
|
+
marker = "executing tool"
|
|
988
|
+
completion = "tools execution completed"
|
|
989
|
+
record: Any = None
|
|
990
|
+
try:
|
|
991
|
+
structured = json.loads(line)
|
|
992
|
+
except (json.JSONDecodeError, TypeError):
|
|
993
|
+
structured = None
|
|
994
|
+
structured_message = (
|
|
995
|
+
str(structured.get("message") or "") if isinstance(structured, dict) else ""
|
|
996
|
+
)
|
|
997
|
+
refusal = structured_message.startswith("ToolError while executing tool:")
|
|
998
|
+
if isinstance(structured, dict) and (
|
|
999
|
+
structured_message in {marker, completion} or refusal
|
|
1000
|
+
):
|
|
1001
|
+
record = structured
|
|
1002
|
+
event = "refusal" if refusal else structured_message
|
|
1003
|
+
elif marker in line:
|
|
1004
|
+
brace = line.find("{", line.find(marker) + len(marker))
|
|
1005
|
+
if brace < 0:
|
|
1006
|
+
continue
|
|
1007
|
+
try:
|
|
1008
|
+
record, _ = decoder.raw_decode(line[brace:])
|
|
1009
|
+
except (json.JSONDecodeError, TypeError):
|
|
1010
|
+
continue
|
|
1011
|
+
event = marker
|
|
1012
|
+
elif completion in line:
|
|
1013
|
+
brace = line.find("{", line.find(completion) + len(completion))
|
|
1014
|
+
if brace < 0:
|
|
1015
|
+
continue
|
|
1016
|
+
try:
|
|
1017
|
+
record, _ = decoder.raw_decode(line[brace:])
|
|
1018
|
+
except (json.JSONDecodeError, TypeError):
|
|
1019
|
+
continue
|
|
1020
|
+
event = completion
|
|
1021
|
+
else:
|
|
1022
|
+
continue
|
|
1023
|
+
if not isinstance(record, dict):
|
|
1024
|
+
continue
|
|
1025
|
+
if event == "refusal":
|
|
1026
|
+
speech_id = str(record.get("speech_id") or "")
|
|
1027
|
+
function = str(record.get("function") or "").strip("/")
|
|
1028
|
+
if speech_id and function:
|
|
1029
|
+
refusals[(speech_id, function)] = structured_message.split(":", 1)[
|
|
1030
|
+
-1
|
|
1031
|
+
].strip()
|
|
1032
|
+
elif event == marker:
|
|
1033
|
+
if not record.get("function"):
|
|
1034
|
+
continue
|
|
1035
|
+
speech_id = str(record.get("speech_id") or "")
|
|
1036
|
+
recorded_name = str(record["function"]).strip("/")
|
|
1037
|
+
name = endpoint_names.get(recorded_name, recorded_name)
|
|
1038
|
+
arguments: Any = record.get("lk.pii.arguments") or {}
|
|
1039
|
+
if isinstance(arguments, str):
|
|
1040
|
+
try:
|
|
1041
|
+
arguments = json.loads(arguments)
|
|
1042
|
+
except json.JSONDecodeError:
|
|
1043
|
+
arguments = {"raw": arguments}
|
|
1044
|
+
if not isinstance(arguments, dict):
|
|
1045
|
+
arguments = {"value": arguments}
|
|
1046
|
+
if contract is not None:
|
|
1047
|
+
specification = contract_tools.get(name)
|
|
1048
|
+
# LiveKit also logs SDK/workflow functions which are not target tools. Only the
|
|
1049
|
+
# contract can make a runtime-log fallback authoritative target evidence.
|
|
1050
|
+
if specification is None:
|
|
1051
|
+
continue
|
|
1052
|
+
# A speech-level completion has no per-tool result. Do not turn a malformed start
|
|
1053
|
+
# into success merely because the surrounding speech completed.
|
|
1054
|
+
if any(argument not in arguments for argument in specification.args):
|
|
1055
|
+
continue
|
|
1056
|
+
starts.append(
|
|
1057
|
+
{
|
|
1058
|
+
"name": name,
|
|
1059
|
+
"recorded_name": recorded_name,
|
|
1060
|
+
"arguments": arguments,
|
|
1061
|
+
"speech_id": speech_id,
|
|
1062
|
+
"at": _log_timestamp(str(record.get("timestamp") or line)),
|
|
1063
|
+
}
|
|
1064
|
+
)
|
|
1065
|
+
elif event == completion:
|
|
1066
|
+
if record.get("speech_id"):
|
|
1067
|
+
completed.add(str(record["speech_id"]))
|
|
1068
|
+
calls: list[Call] = []
|
|
1069
|
+
starts_per_speech = {
|
|
1070
|
+
speech_id: sum(one["speech_id"] == speech_id for one in starts)
|
|
1071
|
+
for speech_id in completed
|
|
1072
|
+
}
|
|
1073
|
+
for one in starts:
|
|
1074
|
+
if not one["speech_id"] or one["speech_id"] not in completed:
|
|
1075
|
+
continue
|
|
1076
|
+
# A single LiveKit completion can close a batch of tool starts but cannot prove which
|
|
1077
|
+
# individual invocation succeeded. Native semantic traces remain authoritative for that
|
|
1078
|
+
# case; the bounded-log fallback deliberately emits no ambiguous evidence.
|
|
1079
|
+
if starts_per_speech.get(one["speech_id"]) != 1:
|
|
1080
|
+
continue
|
|
1081
|
+
error = refusals.get(
|
|
1082
|
+
(one["speech_id"], one["recorded_name"]), ""
|
|
1083
|
+
) or refusals.get((one["speech_id"], one["name"]), "")
|
|
1084
|
+
calls.append(
|
|
1085
|
+
Call(
|
|
1086
|
+
name=one["name"],
|
|
1087
|
+
arguments=one["arguments"],
|
|
1088
|
+
result=(
|
|
1089
|
+
error
|
|
1090
|
+
if error
|
|
1091
|
+
else {
|
|
1092
|
+
"evidence": "livekit_runtime_log",
|
|
1093
|
+
"execution": "completed",
|
|
1094
|
+
}
|
|
1095
|
+
),
|
|
1096
|
+
ok=not error,
|
|
1097
|
+
refused=bool(error),
|
|
1098
|
+
error=error,
|
|
1099
|
+
at=float(one["at"]),
|
|
1100
|
+
)
|
|
1101
|
+
)
|
|
1102
|
+
return calls
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _log_timestamp(line: str) -> float:
|
|
1106
|
+
value = line.strip()
|
|
1107
|
+
matched = re.match(r"^(\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2},\d{3})", value)
|
|
1108
|
+
if matched is not None:
|
|
1109
|
+
value = matched.group(1).replace(",", ".")
|
|
1110
|
+
try:
|
|
1111
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
1112
|
+
return (parsed if parsed.tzinfo else parsed.replace(tzinfo=UTC)).timestamp()
|
|
1113
|
+
except ValueError:
|
|
1114
|
+
return 0.0
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
def _telemetry_mirror(previous: Call, current: Call) -> bool:
|
|
1118
|
+
if previous.name != current.name or previous.arguments != current.arguments:
|
|
1119
|
+
return False
|
|
1120
|
+
result = previous.result
|
|
1121
|
+
if isinstance(result, dict):
|
|
1122
|
+
if result.get("execution") == "submitted_agent_runtime":
|
|
1123
|
+
return True
|
|
1124
|
+
text = str(result.get("result") or "").lower()
|
|
1125
|
+
else:
|
|
1126
|
+
text = str(result or "").lower()
|
|
1127
|
+
return "submitted service has no endpoint" in text
|
|
1128
|
+
|
|
1129
|
+
|
|
1130
|
+
def _timed_exchanges(
|
|
1131
|
+
exchanges: list[Any], times: list[dict[str, Any]]
|
|
1132
|
+
) -> list[dict[str, Any]]:
|
|
1133
|
+
"""The conversation with each turn's speech times attached, where they were measured.
|
|
1134
|
+
|
|
1135
|
+
Paired by position, and only when the two agree on how many turns there were. They come
|
|
1136
|
+
from the same call but by different routes, so a mismatch means one of them dropped a turn
|
|
1137
|
+
-- and pairing them anyway would hang every turn's timing on the wrong words.
|
|
1138
|
+
"""
|
|
1139
|
+
spoken = [{"speaker": turn.speaker, "text": turn.text} for turn in exchanges]
|
|
1140
|
+
if len(times) != len(spoken):
|
|
1141
|
+
return spoken
|
|
1142
|
+
for turn, when in zip(spoken, times, strict=True):
|
|
1143
|
+
if when.get("start_time_ms") is None:
|
|
1144
|
+
continue
|
|
1145
|
+
turn["start_time_ms"] = when["start_time_ms"]
|
|
1146
|
+
if when.get("end_time_ms") is not None:
|
|
1147
|
+
turn["end_time_ms"] = when["end_time_ms"]
|
|
1148
|
+
return spoken
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
def _keep_tracks(found: list[dict[str, str]], folder: Path) -> list[dict[str, str]]:
|
|
1152
|
+
"""Copy each recording into this run's folder, keeping the order it was offered in.
|
|
1153
|
+
|
|
1154
|
+
Copied rather than referenced, because the runner's own directory is transient and a run
|
|
1155
|
+
that cannot be listened to next week is a run that cannot be shown to anybody.
|
|
1156
|
+
"""
|
|
1157
|
+
import shutil
|
|
1158
|
+
|
|
1159
|
+
folder.mkdir(parents=True, exist_ok=True)
|
|
1160
|
+
kept: list[dict[str, str]] = []
|
|
1161
|
+
for track in found:
|
|
1162
|
+
source = Path(track["path"])
|
|
1163
|
+
if not source.exists():
|
|
1164
|
+
continue
|
|
1165
|
+
landed = folder / f"{track['label'].replace(':', '_')}{source.suffix}"
|
|
1166
|
+
try:
|
|
1167
|
+
shutil.copyfile(source, landed)
|
|
1168
|
+
except OSError:
|
|
1169
|
+
continue
|
|
1170
|
+
kept.append({"label": track["label"], "path": str(landed)})
|
|
1171
|
+
return kept
|
|
1172
|
+
|
|
1173
|
+
|
|
1174
|
+
def _averaged(measured: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
1175
|
+
"""Each metric's mean over the scenarios that reported it, carrying whether it applied.
|
|
1176
|
+
|
|
1177
|
+
A metric that had nothing to measure scores 1.0, so averaging the lot produces a suite
|
|
1178
|
+
summary in which two thirds of the numbers are perfect and none of them mean anything. The
|
|
1179
|
+
applicability travels with the average instead of being flattened away, so a reader is never
|
|
1180
|
+
shown "browser action safety 1.00" for a suite of phone calls without also being told there
|
|
1181
|
+
were no browser actions.
|
|
1182
|
+
"""
|
|
1183
|
+
gathered: dict[str, list[float]] = {}
|
|
1184
|
+
applies: dict[str, bool] = {}
|
|
1185
|
+
reasons: dict[str, str] = {}
|
|
1186
|
+
for one in measured:
|
|
1187
|
+
for metric in (one or {}).get("metrics") or []:
|
|
1188
|
+
name, value = metric.get("name"), metric.get("score")
|
|
1189
|
+
if not name or not isinstance(value, (int, float)):
|
|
1190
|
+
continue
|
|
1191
|
+
gathered.setdefault(name, []).append(float(value))
|
|
1192
|
+
# Applicable anywhere is applicable: one scenario exercising a capability is enough
|
|
1193
|
+
# to make the number worth reading across the suite.
|
|
1194
|
+
applies[name] = applies.get(name, False) or bool(
|
|
1195
|
+
metric.get("applicable", True)
|
|
1196
|
+
)
|
|
1197
|
+
if metric.get("reason") and name not in reasons:
|
|
1198
|
+
reasons[name] = str(metric["reason"])
|
|
1199
|
+
return [
|
|
1200
|
+
{
|
|
1201
|
+
"name": name,
|
|
1202
|
+
"score": round(sum(values) / len(values), 4),
|
|
1203
|
+
"applicable": applies.get(name, True),
|
|
1204
|
+
"reason": reasons.get(name, ""),
|
|
1205
|
+
"cases": len(values),
|
|
1206
|
+
}
|
|
1207
|
+
for name, values in sorted(gathered.items())
|
|
1208
|
+
if values
|
|
1209
|
+
]
|