agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,601 @@
|
|
|
1
|
+
"""The tools that run a scenario against the real agent, and record what happened.
|
|
2
|
+
|
|
3
|
+
Placing a call was a command before this existed, which made the last stage the only one you
|
|
4
|
+
could not simply ask for. Nothing about it needed to be a command: wiring the world to the
|
|
5
|
+
assistant and grading afterwards is already code, and choosing which scenario to run and reading
|
|
6
|
+
what came back is the part worth having judgement on.
|
|
7
|
+
|
|
8
|
+
So the same shape as every other stage. The tools do what must be exact — restore the world,
|
|
9
|
+
repoint the assistant's own tools, place the call through ALK, run the checks — and the stage
|
|
10
|
+
decides what to run and says what it means.
|
|
11
|
+
|
|
12
|
+
A run takes minutes, not seconds. The tool blocks for that long, and says so, because a stage
|
|
13
|
+
that fires a call and returns immediately would report on a conversation that has not happened.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import json
|
|
20
|
+
import os
|
|
21
|
+
import re
|
|
22
|
+
import shutil
|
|
23
|
+
import time
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from ..backends import tool, tool_server
|
|
28
|
+
|
|
29
|
+
from .. import platform
|
|
30
|
+
from ..catalogue import load_catalogue
|
|
31
|
+
from ..config import ARTIFACTS_ROOT
|
|
32
|
+
from ..scenario_tools import load_scenarios
|
|
33
|
+
from ..tools import schema
|
|
34
|
+
from ..world.snapshot import require_source_implementation
|
|
35
|
+
from .call import CASE, place_the_call
|
|
36
|
+
from .live import LiveRun, grade, wire
|
|
37
|
+
|
|
38
|
+
RUN_SERVER = "runs"
|
|
39
|
+
RESULTS = "runs.json"
|
|
40
|
+
|
|
41
|
+
# What a call needs before it can be placed at all, per transport. Checked up front rather than
|
|
42
|
+
# three minutes in, because the failure otherwise arrives after the expensive part.
|
|
43
|
+
#
|
|
44
|
+
# Which transport is in play is decided by the case: the 1.x cases reach a LiveKit worker, the
|
|
45
|
+
# 2.x cases a hosted Vapi assistant. Asking for the other one's credentials is how a working
|
|
46
|
+
# setup gets reported as broken.
|
|
47
|
+
REQUIRED_VAPI = ("VAPI_API_KEY", "VAPI_ASSISTANT_ID")
|
|
48
|
+
REQUIRED_LIVEKIT = (
|
|
49
|
+
"LIVEKIT_API_KEY",
|
|
50
|
+
"LIVEKIT_API_SECRET",
|
|
51
|
+
"LIVEKIT_TARGET_AGENT_NAME",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _livekit_case() -> bool:
|
|
56
|
+
"""Whether the case being run reaches a LiveKit worker rather than a hosted assistant."""
|
|
57
|
+
return os.environ.get("HARNESS_VOICE_CASE", CASE).strip().startswith("1.")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _ok(text: str) -> dict[str, Any]:
|
|
61
|
+
return {"content": [{"type": "text", "text": text}]}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _err(text: str) -> dict[str, Any]:
|
|
65
|
+
return {"content": [{"type": "text", "text": text}], "is_error": True}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def configure_source_voice(world_root: Path, contract: Any = None) -> bool:
|
|
69
|
+
"""Select and configure the zero-setup voice path for a provisioned source runtime."""
|
|
70
|
+
root = Path(world_root)
|
|
71
|
+
if not (root / "environment.json").exists():
|
|
72
|
+
return False
|
|
73
|
+
from ..provision import activate_voice_environment
|
|
74
|
+
|
|
75
|
+
prompt = str(getattr(contract, "system_prompt_excerpt", "") or "").strip()
|
|
76
|
+
activate_voice_environment(root, system_prompt=prompt)
|
|
77
|
+
if not os.environ.get("LIVEKIT_TARGET_AGENT_NAME", "").strip():
|
|
78
|
+
label = str(getattr(contract, "agent", "") or "harness-agent").lower()
|
|
79
|
+
label = re.sub(r"[^a-z0-9-]+", "-", label).strip("-") or "harness-agent"
|
|
80
|
+
# The worker setup replaces this with a scenario-scoped value immediately before start.
|
|
81
|
+
# This base merely lets valid unnamed AgentServer workers pass transport preflight.
|
|
82
|
+
os.environ["LIVEKIT_TARGET_AGENT_NAME"] = label[:48]
|
|
83
|
+
os.environ.setdefault("HARNESS_VOICE_CASE", "1.1.2")
|
|
84
|
+
# This is an inbound worker: it greets first. The scripted caller turns that one greeting
|
|
85
|
+
# into its opening request. Simulator-first races an independent opening timer against the
|
|
86
|
+
# greeting transcript and can emit two consecutive caller turns.
|
|
87
|
+
os.environ["HARNESS_CONVERSATION_DIRECTION"] = "agent_first"
|
|
88
|
+
return True
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def missing_prerequisites(
|
|
92
|
+
world_root: Path | None = None, contract: Any = None
|
|
93
|
+
) -> list[str]:
|
|
94
|
+
"""What would stop a live call, in the words of what to do about it."""
|
|
95
|
+
source_backed = bool(world_root and configure_source_voice(world_root, contract))
|
|
96
|
+
problems: list[str] = []
|
|
97
|
+
livekit = _livekit_case()
|
|
98
|
+
absent = [
|
|
99
|
+
name
|
|
100
|
+
for name in (REQUIRED_LIVEKIT if livekit else REQUIRED_VAPI)
|
|
101
|
+
if not os.environ.get(name)
|
|
102
|
+
]
|
|
103
|
+
if absent:
|
|
104
|
+
suffix = (
|
|
105
|
+
"The submitted runtime does not provide them and the platform has no workspace "
|
|
106
|
+
"values configured."
|
|
107
|
+
if source_backed
|
|
108
|
+
else "Load the environment that owns these credentials before running."
|
|
109
|
+
)
|
|
110
|
+
problems.append(
|
|
111
|
+
f"{', '.join(absent)} not set, so there is no way to reach the agent. {suffix}"
|
|
112
|
+
)
|
|
113
|
+
credential = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS", "").strip()
|
|
114
|
+
google_simulator = os.environ.get("SIMULATOR_LLM_PROVIDER", "google").lower() in {
|
|
115
|
+
"google",
|
|
116
|
+
"gemini",
|
|
117
|
+
"vertex",
|
|
118
|
+
}
|
|
119
|
+
if livekit and google_simulator and credential and not Path(credential).is_file():
|
|
120
|
+
problems.append(
|
|
121
|
+
"GOOGLE_APPLICATION_CREDENTIALS points to a file that does not exist in the "
|
|
122
|
+
"harness runtime. Configure the platform's simulator credential mount once; this "
|
|
123
|
+
"is not per-agent setup."
|
|
124
|
+
)
|
|
125
|
+
# A LiveKit worker we run ourselves calls the world directly on the network we share with it,
|
|
126
|
+
# so there is nothing to expose. Only a hosted assistant has to reach in from outside.
|
|
127
|
+
exposed = os.environ.get("HARNESS_WEBHOOK_URL") or shutil.which("cloudflared")
|
|
128
|
+
if not livekit and not exposed:
|
|
129
|
+
problems.append(
|
|
130
|
+
"no way to expose the webhook publicly. A hosted agent cannot reach loopback, so "
|
|
131
|
+
"either install cloudflared (brew install cloudflared) or set HARNESS_WEBHOOK_URL "
|
|
132
|
+
"to a tunnel that is already running."
|
|
133
|
+
)
|
|
134
|
+
return problems
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def save_results(results: list[dict[str, Any]], destination: Path) -> Path:
|
|
138
|
+
"""Keep every run, so a suite can be read after the fact rather than scrolled back to."""
|
|
139
|
+
destination = Path(destination)
|
|
140
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
141
|
+
path = destination / RESULTS
|
|
142
|
+
path.write_text(json.dumps(results, indent=2, ensure_ascii=False), encoding="utf-8")
|
|
143
|
+
return path
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def load_results(destination: Path) -> list[dict[str, Any]]:
|
|
147
|
+
path = Path(destination) / RESULTS
|
|
148
|
+
if not path.exists():
|
|
149
|
+
return []
|
|
150
|
+
try:
|
|
151
|
+
loaded = json.loads(path.read_text(encoding="utf-8"))
|
|
152
|
+
return loaded if isinstance(loaded, list) else []
|
|
153
|
+
except json.JSONDecodeError:
|
|
154
|
+
return []
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def as_record(run: LiveRun) -> dict[str, Any]:
|
|
158
|
+
return {
|
|
159
|
+
"scenario": run.scenario,
|
|
160
|
+
"passed": bool(run.settled)
|
|
161
|
+
and run.met == len(run.settled)
|
|
162
|
+
and not run.problems,
|
|
163
|
+
"met": run.met,
|
|
164
|
+
"of": len(run.settled),
|
|
165
|
+
"settled": [
|
|
166
|
+
{"name": one.name, "held": one.held, "said": one.said, "broken": one.broken}
|
|
167
|
+
for one in run.settled
|
|
168
|
+
],
|
|
169
|
+
"judged": list(run.judged),
|
|
170
|
+
"calls": list(run.calls),
|
|
171
|
+
"problems": list(run.problems),
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def transcript_since(started: float) -> str:
|
|
176
|
+
"""What was said on the call that just happened, from the voice runner's own report.
|
|
177
|
+
|
|
178
|
+
The voice case owns the call and writes its report where it always has; reaching into that
|
|
179
|
+
report is how the transcript gets onto the run record without the harness re-implementing
|
|
180
|
+
any of the call. Only a report written after this run started counts — the newest file on
|
|
181
|
+
disk is otherwise last week's call wearing today's verdict.
|
|
182
|
+
"""
|
|
183
|
+
root = ARTIFACTS_ROOT / "simulation-acceptance"
|
|
184
|
+
if not root.exists():
|
|
185
|
+
return ""
|
|
186
|
+
newest: tuple[float, Path] | None = None
|
|
187
|
+
for report in root.glob("run_*/*/report.json"):
|
|
188
|
+
written = report.stat().st_mtime
|
|
189
|
+
if written >= started and (newest is None or written > newest[0]):
|
|
190
|
+
newest = (written, report)
|
|
191
|
+
if newest is None:
|
|
192
|
+
return ""
|
|
193
|
+
try:
|
|
194
|
+
loaded = json.loads(newest[1].read_text(encoding="utf-8"))
|
|
195
|
+
except (json.JSONDecodeError, OSError):
|
|
196
|
+
return ""
|
|
197
|
+
for result in loaded.get("results") or []:
|
|
198
|
+
spoken = result.get("transcript")
|
|
199
|
+
if isinstance(spoken, str) and spoken.strip():
|
|
200
|
+
return spoken
|
|
201
|
+
return ""
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def report(run: LiveRun) -> str:
|
|
205
|
+
"""One run, as something worth reading rather than a score."""
|
|
206
|
+
lines = [run.line()]
|
|
207
|
+
lines += [one.line() for one in run.settled]
|
|
208
|
+
lines += [f" [?] {name} — judged, not settled by code" for name in run.judged]
|
|
209
|
+
if run.problems:
|
|
210
|
+
lines += [f" !! {problem}" for problem in run.problems]
|
|
211
|
+
lines.append("")
|
|
212
|
+
lines.append("what the agent actually did:")
|
|
213
|
+
lines += [
|
|
214
|
+
f" {call}" for call in run.calls or ["(no tool calls reached the world)"]
|
|
215
|
+
]
|
|
216
|
+
return "\n".join(lines)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def run_tools(
|
|
220
|
+
world_root: Path,
|
|
221
|
+
destination: Path,
|
|
222
|
+
*,
|
|
223
|
+
contract: Any = None,
|
|
224
|
+
case: str = "",
|
|
225
|
+
) -> Any:
|
|
226
|
+
"""A server for running one agent's scenarios against the real thing.
|
|
227
|
+
|
|
228
|
+
How a scenario runs is decided by what the agent is, not by this stage. A hosted voice agent
|
|
229
|
+
gets the live path — its own tools repointed at the world over a webhook, the call placed
|
|
230
|
+
through ALK. Anything else runs here: the agent stood up from its contract, conversing over
|
|
231
|
+
the same world, graded by the same checks. The scenarios, the world and the grading are
|
|
232
|
+
identical either way; only the transport changes.
|
|
233
|
+
"""
|
|
234
|
+
written = load_scenarios(destination)
|
|
235
|
+
catalogue = load_catalogue(destination)
|
|
236
|
+
results = load_results(destination)
|
|
237
|
+
live = bool(contract is not None and getattr(contract, "modality", "") == "voice")
|
|
238
|
+
if live:
|
|
239
|
+
configure_source_voice(world_root, contract)
|
|
240
|
+
voice_case = case or os.environ.get("HARNESS_VOICE_CASE", "2.1.2")
|
|
241
|
+
authenticity_error = ""
|
|
242
|
+
try:
|
|
243
|
+
require_source_implementation(world_root)
|
|
244
|
+
except (FileNotFoundError, RuntimeError) as failed:
|
|
245
|
+
authenticity_error = str(failed)
|
|
246
|
+
|
|
247
|
+
@tool(
|
|
248
|
+
"list_scenarios",
|
|
249
|
+
"The scenarios that can be run, what each one tests, and which of its sub-goals are "
|
|
250
|
+
"settled by code rather than left to a judge.",
|
|
251
|
+
schema({}, []),
|
|
252
|
+
)
|
|
253
|
+
async def list_scenarios(_args: dict[str, Any]) -> dict[str, Any]:
|
|
254
|
+
if not written:
|
|
255
|
+
return _err("no scenarios have been written for this agent yet")
|
|
256
|
+
lines: list[str] = []
|
|
257
|
+
for one in written:
|
|
258
|
+
settled = [
|
|
259
|
+
name
|
|
260
|
+
for name in one.sub_goals
|
|
261
|
+
if (found := catalogue.named(name)) and found.deterministic()
|
|
262
|
+
]
|
|
263
|
+
judged = [name for name in one.sub_goals if name not in settled]
|
|
264
|
+
ran = next((r for r in results if r["scenario"] == one.name), None)
|
|
265
|
+
mark = (
|
|
266
|
+
""
|
|
267
|
+
if ran is None
|
|
268
|
+
else (
|
|
269
|
+
" [last run: PASS]" if ran.get("passed") else " [last run: FAIL]"
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
lines.append(
|
|
273
|
+
f"{one.name}{mark}\n passes when: {one.tests or one.use_case or '—'}\n"
|
|
274
|
+
f" settled by code: {', '.join(settled) or 'none'}\n"
|
|
275
|
+
f" judged: {', '.join(judged) or 'none'}"
|
|
276
|
+
)
|
|
277
|
+
return _ok("\n".join(lines))
|
|
278
|
+
|
|
279
|
+
@tool(
|
|
280
|
+
"preflight",
|
|
281
|
+
"Check everything a run needs before spending one. For a hosted voice agent that is the "
|
|
282
|
+
"assistant's credentials and a way to expose the webhook publicly; for anything else "
|
|
283
|
+
"the run happens here and needs nothing external. Run this before the first run.",
|
|
284
|
+
schema({}, []),
|
|
285
|
+
)
|
|
286
|
+
async def preflight(_args: dict[str, Any]) -> dict[str, Any]:
|
|
287
|
+
if authenticity_error:
|
|
288
|
+
return _err(f"Not ready:\n - {authenticity_error}")
|
|
289
|
+
if not live:
|
|
290
|
+
return _err(
|
|
291
|
+
"Not ready: no shipped-runtime target is registered for this agent. The old "
|
|
292
|
+
"local target reconstructed it from the contract and is intentionally disabled."
|
|
293
|
+
)
|
|
294
|
+
problems = missing_prerequisites(world_root, contract)
|
|
295
|
+
if problems:
|
|
296
|
+
return _err("Not ready:\n - " + "\n - ".join(problems))
|
|
297
|
+
return _ok(
|
|
298
|
+
"Ready. Credentials are set and the webhook can be exposed. "
|
|
299
|
+
f"{len(written)} scenarios are available."
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
async def _run_here(scenario: Any) -> dict[str, Any]:
|
|
303
|
+
"""The scenario against the agent stood up from its contract, over the same world."""
|
|
304
|
+
from . import run_suite
|
|
305
|
+
|
|
306
|
+
if authenticity_error:
|
|
307
|
+
return _err(authenticity_error)
|
|
308
|
+
if contract is None:
|
|
309
|
+
return _err("no contract is loaded, so there is no agent to stand up")
|
|
310
|
+
graded = await run_suite([scenario], contract, world_root, out=destination)
|
|
311
|
+
results[:] = load_results(destination)
|
|
312
|
+
result = graded[0]
|
|
313
|
+
lines = [result.line()] + [check.line() for check in result.checkpoints]
|
|
314
|
+
if result.transcript:
|
|
315
|
+
lines += ["", "the conversation:", result.transcript]
|
|
316
|
+
answer = "\n".join(lines)
|
|
317
|
+
return _ok(answer) if result.passed else _err(answer)
|
|
318
|
+
|
|
319
|
+
@tool(
|
|
320
|
+
"run_simulation",
|
|
321
|
+
"Run the whole suite. One call: every scenario, each in its own copy of the world, "
|
|
322
|
+
"graded, and written out as one run you can come back to.\n\n"
|
|
323
|
+
"This is how a suite is run. Running scenarios one at a time is for looking into a "
|
|
324
|
+
"single failure afterwards, not for getting results.\n\n"
|
|
325
|
+
"`concurrency` is how many run at once. Leave it at 1 for a spoken agent, where every "
|
|
326
|
+
"scenario is a real call. It takes minutes and blocks until the whole suite is done.",
|
|
327
|
+
schema({"concurrency": int, "model": str}, []),
|
|
328
|
+
)
|
|
329
|
+
async def run_simulation(args: dict[str, Any]) -> dict[str, Any]:
|
|
330
|
+
from .simulation import simulate
|
|
331
|
+
|
|
332
|
+
if authenticity_error:
|
|
333
|
+
return _err(authenticity_error)
|
|
334
|
+
if not live:
|
|
335
|
+
return _err(
|
|
336
|
+
"no shipped-runtime target is registered for this agent; refusing to "
|
|
337
|
+
"reconstruct it from the contract"
|
|
338
|
+
)
|
|
339
|
+
if contract is None:
|
|
340
|
+
return _err("no contract is loaded, so there is no agent to run against")
|
|
341
|
+
if not written:
|
|
342
|
+
return _err("there are no scenarios to run")
|
|
343
|
+
# Kept as they finish, because reporting needs the graded results themselves and the
|
|
344
|
+
# summary carries only their rendering.
|
|
345
|
+
produced: list[Any] = []
|
|
346
|
+
summary = await simulate(
|
|
347
|
+
list(written),
|
|
348
|
+
contract,
|
|
349
|
+
world_root,
|
|
350
|
+
destination=destination,
|
|
351
|
+
model=str(args.get("model") or "") or None,
|
|
352
|
+
concurrency=max(1, int(args.get("concurrency") or 1)),
|
|
353
|
+
on_case_done=produced.append,
|
|
354
|
+
)
|
|
355
|
+
results[:] = load_results(destination)
|
|
356
|
+
lines = [
|
|
357
|
+
f"{summary['run_id']}: {summary['passed']}/{summary['scenarios']} passed "
|
|
358
|
+
f"in {summary['seconds']}s, ${summary['spent_usd']}",
|
|
359
|
+
"",
|
|
360
|
+
]
|
|
361
|
+
for one in summary["results"]:
|
|
362
|
+
mark = "PASS" if one["passed"] else "FAIL"
|
|
363
|
+
note = f" {one['problems'][0]}" if one["problems"] else ""
|
|
364
|
+
audio = " [recording]" if one["recording"] else ""
|
|
365
|
+
lines.append(
|
|
366
|
+
f" {mark} {one['scenario']} {one['met']}/{one['of']}{audio}{note}"
|
|
367
|
+
)
|
|
368
|
+
# Reported here too, not only from the run button: a run that reaches the platform only
|
|
369
|
+
# when it was started one particular way leaves the page an unreliable record of what
|
|
370
|
+
# has been run.
|
|
371
|
+
_, said = platform.deliver(
|
|
372
|
+
produced, list(written), destination, modality=contract.modality or "text"
|
|
373
|
+
)
|
|
374
|
+
lines += ["", *said]
|
|
375
|
+
lines += [
|
|
376
|
+
"",
|
|
377
|
+
"read_run gives any one of these in full: the conversation, every tool call with "
|
|
378
|
+
"its arguments, and what each check decided.",
|
|
379
|
+
]
|
|
380
|
+
return _ok("\n".join(lines))
|
|
381
|
+
|
|
382
|
+
@tool(
|
|
383
|
+
"read_run",
|
|
384
|
+
"One run in full, or the list of runs when no id is given. A run holds every scenario's "
|
|
385
|
+
"conversation, every tool call with its arguments and result, and what each check "
|
|
386
|
+
"decided — which is what a failure is diagnosed from.",
|
|
387
|
+
schema({"run_id": str, "scenario": str}, []),
|
|
388
|
+
)
|
|
389
|
+
async def read_run(args: dict[str, Any]) -> dict[str, Any]:
|
|
390
|
+
from .simulation import every_run, read_run as load_run
|
|
391
|
+
|
|
392
|
+
run_id = str(args.get("run_id") or "")
|
|
393
|
+
if not run_id:
|
|
394
|
+
runs = every_run(destination)
|
|
395
|
+
if not runs:
|
|
396
|
+
return _ok("No runs yet. run_simulation makes one.")
|
|
397
|
+
return _ok(
|
|
398
|
+
"\n".join(
|
|
399
|
+
f" {one['run_id']} {one.get('passed', 0)}/{one.get('scenarios', 0)} "
|
|
400
|
+
f"passed {one.get('seconds', 0)}s"
|
|
401
|
+
for one in runs
|
|
402
|
+
)
|
|
403
|
+
)
|
|
404
|
+
try:
|
|
405
|
+
whole = load_run(destination, run_id)
|
|
406
|
+
except FileNotFoundError as missing:
|
|
407
|
+
return _err(str(missing))
|
|
408
|
+
wanted = str(args.get("scenario") or "")
|
|
409
|
+
cases = [
|
|
410
|
+
one
|
|
411
|
+
for one in whole.get("scenarios", [])
|
|
412
|
+
if not wanted or one.get("scenario") == wanted
|
|
413
|
+
]
|
|
414
|
+
if not cases:
|
|
415
|
+
return _err(f"{run_id} has no scenario called {wanted!r}")
|
|
416
|
+
return _ok(json.dumps(cases if wanted else whole, indent=2, default=str)[:6000])
|
|
417
|
+
|
|
418
|
+
@tool(
|
|
419
|
+
"run_scenario",
|
|
420
|
+
"Run one scenario against the agent and grade it.\n\n"
|
|
421
|
+
"The world is restored and the scenario's setup applied first. A hosted voice agent is "
|
|
422
|
+
"reached live — its OWN tools are pointed at the world over a webhook and the call is "
|
|
423
|
+
"placed; any other agent is stood up here from its contract and conversed with. Either "
|
|
424
|
+
"way the sub-goals' checks run against what the world holds afterwards plus the calls "
|
|
425
|
+
"that were made.\n\n"
|
|
426
|
+
"It can take minutes and blocks until the run is over. Run one at a time and read what "
|
|
427
|
+
"comes back before running the next.",
|
|
428
|
+
# Both spellings accepted: every model that has driven this stage has guessed
|
|
429
|
+
# `scenario` at least once, and a retry on an argument name is a wasted turn.
|
|
430
|
+
schema({"name": str, "scenario": str}, []),
|
|
431
|
+
)
|
|
432
|
+
async def run_scenario(args: dict[str, Any]) -> dict[str, Any]:
|
|
433
|
+
if authenticity_error:
|
|
434
|
+
return _err(authenticity_error)
|
|
435
|
+
name = str(args.get("name") or args.get("scenario") or "")
|
|
436
|
+
scenario = next((one for one in written if one.name == name), None)
|
|
437
|
+
if scenario is None:
|
|
438
|
+
return _err(
|
|
439
|
+
f"no scenario called {name!r}. There is: "
|
|
440
|
+
+ ", ".join(one.name for one in written)
|
|
441
|
+
)
|
|
442
|
+
if not live:
|
|
443
|
+
return await _run_here(scenario)
|
|
444
|
+
problems = missing_prerequisites(world_root, contract)
|
|
445
|
+
if problems:
|
|
446
|
+
return _err(
|
|
447
|
+
"Cannot place a call:\n - "
|
|
448
|
+
+ "\n - ".join(problems)
|
|
449
|
+
+ "\nThis is the environment this harness is running in, not something to fix "
|
|
450
|
+
"in the scenario."
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
def placed() -> tuple[LiveRun, str, list[str], str]:
|
|
454
|
+
"""The whole call, off the event loop.
|
|
455
|
+
|
|
456
|
+
Wiring reads a subprocess's stdout and placing the call blocks for minutes; run
|
|
457
|
+
inline they freeze whatever loop is hosting this tool, which for the web UI means
|
|
458
|
+
the stream, the status endpoint and the stop button all die for the duration.
|
|
459
|
+
"""
|
|
460
|
+
world, instruction, webhook, tunnel, url, moved = wire(scenario, world_root)
|
|
461
|
+
started = time.time()
|
|
462
|
+
try:
|
|
463
|
+
# The caller's instruction reaches the voice case through the environment, so
|
|
464
|
+
# how a simulated caller behaves is not decided in two places.
|
|
465
|
+
os.environ["HARNESS_INSTRUCTION"] = instruction
|
|
466
|
+
os.environ["HARNESS_SCENARIO"] = scenario.name
|
|
467
|
+
# The caller is never handed the grader's pass question. `tests` is written about
|
|
468
|
+
# the agent in the third person, so as an objective it reads as a rubric rather
|
|
469
|
+
# than a motive. What this person wants is already in the instruction.
|
|
470
|
+
os.environ.pop("HARNESS_OUTCOME", None)
|
|
471
|
+
os.environ["HARNESS_PERSONA"] = json.dumps(
|
|
472
|
+
scenario.persona.model_dump(exclude_none=True)
|
|
473
|
+
if scenario.persona is not None
|
|
474
|
+
else {"name": "customer"}
|
|
475
|
+
)
|
|
476
|
+
os.environ["HARNESS_INITIAL_MESSAGE"] = (
|
|
477
|
+
scenario.persona.initial_message
|
|
478
|
+
if scenario.persona is not None
|
|
479
|
+
else ""
|
|
480
|
+
)
|
|
481
|
+
code = place_the_call(voice_case)
|
|
482
|
+
run = grade(scenario, world, world_root)
|
|
483
|
+
if code != 0 and not run.calls:
|
|
484
|
+
run.problems.append(
|
|
485
|
+
f"the voice runner exited {code} and no tool call reached the world, "
|
|
486
|
+
"so this says nothing about the agent"
|
|
487
|
+
)
|
|
488
|
+
finally:
|
|
489
|
+
webhook.stop()
|
|
490
|
+
if tunnel is not None:
|
|
491
|
+
tunnel.terminate()
|
|
492
|
+
if (Path(world_root) / "environment.json").exists():
|
|
493
|
+
from ..provision import stop_runtime
|
|
494
|
+
|
|
495
|
+
stop_runtime(world_root)
|
|
496
|
+
world.close()
|
|
497
|
+
return run, url, moved, transcript_since(started)
|
|
498
|
+
|
|
499
|
+
run, url, moved, spoken = await asyncio.to_thread(placed)
|
|
500
|
+
|
|
501
|
+
record = as_record(run)
|
|
502
|
+
record["instruction"] = scenario.instruction
|
|
503
|
+
record["transcript"] = spoken
|
|
504
|
+
# Re-read before writing: the local suite writes the same file, and a list loaded when
|
|
505
|
+
# this stage opened would silently roll back anything recorded since.
|
|
506
|
+
results[:] = [
|
|
507
|
+
r for r in load_results(destination) if r.get("scenario") != scenario.name
|
|
508
|
+
]
|
|
509
|
+
results.append(record)
|
|
510
|
+
save_results(results, destination)
|
|
511
|
+
answer = f"webhook: {url}/tool\nrepointed: {', '.join(moved)}\n\n{report(run)}"
|
|
512
|
+
return _ok(answer) if not run.problems else _err(answer)
|
|
513
|
+
|
|
514
|
+
@tool(
|
|
515
|
+
"read_results",
|
|
516
|
+
"What every scenario did the last time it was run, without running anything.",
|
|
517
|
+
schema({}, []),
|
|
518
|
+
)
|
|
519
|
+
async def read_results(_args: dict[str, Any]) -> dict[str, Any]:
|
|
520
|
+
if not results:
|
|
521
|
+
return _ok("nothing has been run yet")
|
|
522
|
+
lines = []
|
|
523
|
+
for record in results:
|
|
524
|
+
mark = "PASS" if record.get("passed") else "FAIL"
|
|
525
|
+
# Two record shapes share this file: live runs carry settled/judged, local runs
|
|
526
|
+
# carry checkpoints. Both say what failed, and both deserve to be read.
|
|
527
|
+
failed = [
|
|
528
|
+
f"{one.get('name')}: {one.get('said') or one.get('detail') or ''}"
|
|
529
|
+
for one in (record.get("settled") or record.get("checkpoints") or [])
|
|
530
|
+
if not (one.get("held") if "held" in one else one.get("passed"))
|
|
531
|
+
]
|
|
532
|
+
met = record.get("met", record.get("checkpoints_met", "?"))
|
|
533
|
+
of = record.get("of")
|
|
534
|
+
scored = f"{met}/{of}" if of is not None else str(met)
|
|
535
|
+
lines.append(
|
|
536
|
+
f"{mark} {record.get('scenario')} {scored}"
|
|
537
|
+
+ ("\n - " + "\n - ".join(failed) if failed else "")
|
|
538
|
+
)
|
|
539
|
+
passed = sum(1 for record in results if record.get("passed"))
|
|
540
|
+
return _ok("\n".join(lines) + f"\n\n{passed} of {len(results)} passed")
|
|
541
|
+
|
|
542
|
+
server = tool_server(
|
|
543
|
+
name=RUN_SERVER,
|
|
544
|
+
version="0.1.0",
|
|
545
|
+
tools=[
|
|
546
|
+
list_scenarios,
|
|
547
|
+
preflight,
|
|
548
|
+
run_simulation,
|
|
549
|
+
read_run,
|
|
550
|
+
run_scenario,
|
|
551
|
+
read_results,
|
|
552
|
+
],
|
|
553
|
+
)
|
|
554
|
+
return server
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
TOOL_NAMES = (
|
|
558
|
+
"list_scenarios",
|
|
559
|
+
"preflight",
|
|
560
|
+
"run_simulation",
|
|
561
|
+
"read_run",
|
|
562
|
+
"run_scenario",
|
|
563
|
+
"read_results",
|
|
564
|
+
)
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
# Which of the several recordings a call leaves behind is the one worth keeping. Both sides on
|
|
568
|
+
# one track, because the question asked of a spoken run is nearly always about the interaction:
|
|
569
|
+
# whether the agent talked over the caller, how long it left them waiting, what it heard.
|
|
570
|
+
PREFERRED = ("_stereo.wav", "stereo.wav", "combined.wav")
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
def recording_since(started: float, into: Path) -> str:
|
|
574
|
+
"""Copy the audio from the call that just happened into this run's folder.
|
|
575
|
+
|
|
576
|
+
ALK records already and writes several tracks under its own artifacts directory. Rather than
|
|
577
|
+
tell it where to put them — which it takes from its manifest, not from the environment — the
|
|
578
|
+
files it wrote are found the same way the transcript is, by being newer than the moment this
|
|
579
|
+
run began, and the one worth keeping is copied in beside the result.
|
|
580
|
+
"""
|
|
581
|
+
root = ARTIFACTS_ROOT / "simulation-acceptance"
|
|
582
|
+
if not root.exists():
|
|
583
|
+
return ""
|
|
584
|
+
fresh = [
|
|
585
|
+
path
|
|
586
|
+
for path in root.rglob("*")
|
|
587
|
+
if path.is_file()
|
|
588
|
+
and path.suffix.lower() in (".wav", ".mp3", ".ogg")
|
|
589
|
+
and path.stat().st_mtime >= started
|
|
590
|
+
]
|
|
591
|
+
if not fresh:
|
|
592
|
+
return ""
|
|
593
|
+
chosen = next(
|
|
594
|
+
(one for mark in PREFERRED for one in fresh if one.name.endswith(mark)),
|
|
595
|
+
max(fresh, key=lambda one: one.stat().st_size),
|
|
596
|
+
)
|
|
597
|
+
into = Path(into)
|
|
598
|
+
into.mkdir(parents=True, exist_ok=True)
|
|
599
|
+
landed = into / f"recording{chosen.suffix}"
|
|
600
|
+
shutil.copyfile(chosen, landed)
|
|
601
|
+
return str(landed)
|